"""Offline generator for backend/cities.json — the finite set of cities that get crawlable climate pages (/climate/). Run occasionally to refresh the list: python gen_cities.py [N] # default N=500 top metros by population It reuses the GeoNames index that places.py already downloads/parses (calling places._load() synchronously fills places._data), takes the top-N places by population, and assigns each a stable, unique, URL-safe slug. Committing the output keeps the routable city set explicit and reviewable, and decouples page-serving from the async place-name loader. """ import json import os import re import sys import unicodedata import places OUT_PATH = os.path.join(os.path.dirname(__file__), "cities.json") # GeoNames entry tuple layout (see places._load): the fields we keep. _NAME, _ADMIN1, _COUNTRY, _CC, _LAT, _LON, _POP = 1, 2, 3, 4, 5, 6, 7 def slugify(*parts: str) -> str: """ASCII, lowercase, hyphenated slug from name/admin/country parts.""" text = " ".join(p for p in parts if p) text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode() text = re.sub(r"[^a-zA-Z0-9]+", "-", text).strip("-").lower() return re.sub(r"-{2,}", "-", text) # Countries where English is the primary/official language of web search. Used to # top up the population-ranked global list (which skews to Asia) with the # high-search-demand English-market cities that would otherwise be missed. ENGLISH_CC = {"US", "GB", "CA", "AU", "NZ", "IE", "ZA"} def _to_city(e, seen_slugs: set[str]) -> dict | None: name, admin1, country, cc = e[_NAME], e[_ADMIN1], e[_COUNTRY], e[_CC] # Drop admin1 from the slug when it just repeats the city name # (e.g. Tokyo/Tokyo, Singapore/Singapore) to avoid "tokyo-tokyo-jp". admin_part = admin1 if admin1 and slugify(admin1) != slugify(name) else "" base = slugify(name, admin_part, cc or "") if not base: return None slug = base i = 2 while slug in seen_slugs: # disambiguate the rare collision slug = f"{base}-{i}" i += 1 seen_slugs.add(slug) return { "slug": slug, "name": name, "admin1": admin1, "country": country, "country_code": cc, "lat": round(e[_LAT], 5), "lon": round(e[_LON], 5), "population": e[_POP], } def build(n_global: int = 500, n_english: int = 250) -> list[dict]: """Top n_global cities worldwide by population, then up to n_english more from English-speaking countries that weren't already in that global set.""" places._load() # synchronous parse; fills places._data (entries are pop-desc) if not places._data: raise SystemExit("GeoNames index failed to load (see logs); cannot generate cities.") entries = places._data[0] out: list[dict] = [] seen_slugs: set[str] = set() for e in entries: if len(out) >= n_global: break c = _to_city(e, seen_slugs) if c: out.append(c) # Identity of the cities already chosen, so the English top-up skips them. chosen = {(c["name"], c["admin1"], c["country_code"]) for c in out} added = 0 for e in entries: if added >= n_english: break if e[_CC] not in ENGLISH_CC: continue if (e[_NAME], e[_ADMIN1], e[_CC]) in chosen: continue c = _to_city(e, seen_slugs) if c: out.append(c) added += 1 return out def main() -> None: n_global = int(sys.argv[1]) if len(sys.argv) > 1 else 500 n_english = int(sys.argv[2]) if len(sys.argv) > 2 else 250 cities = build(n_global, n_english) with open(OUT_PATH, "w", encoding="utf-8") as f: json.dump(cities, f, ensure_ascii=False, indent=0, separators=(",", ":")) f.write("\n") print(f"wrote {len(cities)} cities ({n_global} global + up to {n_english} English-market) -> {OUT_PATH}") print("sample:", ", ".join(c["slug"] for c in cities[:8])) if __name__ == "__main__": main()