83 lines
2.9 KiB
Python
83 lines
2.9 KiB
Python
|
|
"""Offline generator for backend/cities.json — the finite set of cities that get
|
||
|
|
crawlable climate pages (/climate/<slug>). Run occasionally to refresh the list:
|
||
|
|
|
||
|
|
python gen_cities.py [N] # default N=500 top metros by population
|
||
|
|
|
||
|
|
It reuses the GeoNames index that places.py already downloads/parses (calling
|
||
|
|
places._load() synchronously fills places._data), takes the top-N places by
|
||
|
|
population, and assigns each a stable, unique, URL-safe slug. Committing the output
|
||
|
|
keeps the routable city set explicit and reviewable, and decouples page-serving
|
||
|
|
from the async place-name loader.
|
||
|
|
"""
|
||
|
|
import json
|
||
|
|
import os
|
||
|
|
import re
|
||
|
|
import sys
|
||
|
|
import unicodedata
|
||
|
|
|
||
|
|
import places
|
||
|
|
|
||
|
|
OUT_PATH = os.path.join(os.path.dirname(__file__), "cities.json")
|
||
|
|
|
||
|
|
# GeoNames entry tuple layout (see places._load): the fields we keep.
|
||
|
|
_NAME, _ADMIN1, _COUNTRY, _CC, _LAT, _LON, _POP = 1, 2, 3, 4, 5, 6, 7
|
||
|
|
|
||
|
|
|
||
|
|
def slugify(*parts: str) -> str:
|
||
|
|
"""ASCII, lowercase, hyphenated slug from name/admin/country parts."""
|
||
|
|
text = " ".join(p for p in parts if p)
|
||
|
|
text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
|
||
|
|
text = re.sub(r"[^a-zA-Z0-9]+", "-", text).strip("-").lower()
|
||
|
|
return re.sub(r"-{2,}", "-", text)
|
||
|
|
|
||
|
|
|
||
|
|
def build(n: int = 500) -> list[dict]:
|
||
|
|
places._load() # synchronous parse; fills places._data (entries are pop-desc)
|
||
|
|
if not places._data:
|
||
|
|
raise SystemExit("GeoNames index failed to load (see logs); cannot generate cities.")
|
||
|
|
entries = places._data[0]
|
||
|
|
|
||
|
|
out: list[dict] = []
|
||
|
|
seen_slugs: set[str] = set()
|
||
|
|
for e in entries:
|
||
|
|
if len(out) >= n:
|
||
|
|
break
|
||
|
|
name, admin1, country, cc = e[_NAME], e[_ADMIN1], e[_COUNTRY], e[_CC]
|
||
|
|
# Drop admin1 from the slug when it just repeats the city name
|
||
|
|
# (e.g. Tokyo/Tokyo, Singapore/Singapore) to avoid "tokyo-tokyo-jp".
|
||
|
|
admin_part = admin1 if admin1 and slugify(admin1) != slugify(name) else ""
|
||
|
|
base = slugify(name, admin_part, cc or "")
|
||
|
|
if not base:
|
||
|
|
continue
|
||
|
|
slug = base
|
||
|
|
i = 2
|
||
|
|
while slug in seen_slugs: # disambiguate the rare collision
|
||
|
|
slug = f"{base}-{i}"
|
||
|
|
i += 1
|
||
|
|
seen_slugs.add(slug)
|
||
|
|
out.append({
|
||
|
|
"slug": slug,
|
||
|
|
"name": name,
|
||
|
|
"admin1": admin1,
|
||
|
|
"country": country,
|
||
|
|
"country_code": cc,
|
||
|
|
"lat": round(e[_LAT], 5),
|
||
|
|
"lon": round(e[_LON], 5),
|
||
|
|
"population": e[_POP],
|
||
|
|
})
|
||
|
|
return out
|
||
|
|
|
||
|
|
|
||
|
|
def main() -> None:
|
||
|
|
n = int(sys.argv[1]) if len(sys.argv) > 1 else 500
|
||
|
|
cities = build(n)
|
||
|
|
with open(OUT_PATH, "w", encoding="utf-8") as f:
|
||
|
|
json.dump(cities, f, ensure_ascii=False, indent=0, separators=(",", ":"))
|
||
|
|
f.write("\n")
|
||
|
|
print(f"wrote {len(cities)} cities -> {OUT_PATH}")
|
||
|
|
print("sample:", ", ".join(c["slug"] for c in cities[:8]))
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
main()
|