thermograph/gen_cities.py
Emi Griffith 3bbd819d1d SEO: add 250 English-market city pages; auto-warm archives on deploy (#97)
The population-ranked global top-500 skewed to Asian megacities and missed
high-English-search-demand cities. gen_cities.py now tops up with the top ~250
cities from English-speaking countries (US/GB/CA/AU/NZ/IE/ZA) not already in the
global set, so US coverage goes 13->146, GB 2->42, CA 3->29, etc. (Seattle, Boston,
Manchester, Melbourne, Auckland, Dublin, ...). cities.json regenerated to 750.

Both deploy scripts now launch warm_cities.py automatically after the health check,
detached (dev: a systemd --user transient unit; prod: setsid/nohup), so the city
pages serve from cache without a manual step; idempotent, so only the first deploy
does the full warm. DEPLOY.md updated.
2026-07-16 00:11:14 +00:00

108 lines
3.9 KiB
Python

"""Offline generator for backend/cities.json — the finite set of cities that get
crawlable climate pages (/climate/<slug>). Run occasionally to refresh the list:
python gen_cities.py [N] # default N=500 top metros by population
It reuses the GeoNames index that places.py already downloads/parses (calling
places._load() synchronously fills places._data), takes the top-N places by
population, and assigns each a stable, unique, URL-safe slug. Committing the output
keeps the routable city set explicit and reviewable, and decouples page-serving
from the async place-name loader.
"""
import json
import os
import re
import sys
import unicodedata
import places
OUT_PATH = os.path.join(os.path.dirname(__file__), "cities.json")
# GeoNames entry tuple layout (see places._load): the fields we keep.
_NAME, _ADMIN1, _COUNTRY, _CC, _LAT, _LON, _POP = 1, 2, 3, 4, 5, 6, 7
def slugify(*parts: str) -> str:
"""ASCII, lowercase, hyphenated slug from name/admin/country parts."""
text = " ".join(p for p in parts if p)
text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
text = re.sub(r"[^a-zA-Z0-9]+", "-", text).strip("-").lower()
return re.sub(r"-{2,}", "-", text)
# Countries where English is the primary/official language of web search. Used to
# top up the population-ranked global list (which skews to Asia) with the
# high-search-demand English-market cities that would otherwise be missed.
ENGLISH_CC = {"US", "GB", "CA", "AU", "NZ", "IE", "ZA"}
def _to_city(e, seen_slugs: set[str]) -> dict | None:
name, admin1, country, cc = e[_NAME], e[_ADMIN1], e[_COUNTRY], e[_CC]
# Drop admin1 from the slug when it just repeats the city name
# (e.g. Tokyo/Tokyo, Singapore/Singapore) to avoid "tokyo-tokyo-jp".
admin_part = admin1 if admin1 and slugify(admin1) != slugify(name) else ""
base = slugify(name, admin_part, cc or "")
if not base:
return None
slug = base
i = 2
while slug in seen_slugs: # disambiguate the rare collision
slug = f"{base}-{i}"
i += 1
seen_slugs.add(slug)
return {
"slug": slug, "name": name, "admin1": admin1,
"country": country, "country_code": cc,
"lat": round(e[_LAT], 5), "lon": round(e[_LON], 5),
"population": e[_POP],
}
def build(n_global: int = 500, n_english: int = 250) -> list[dict]:
"""Top n_global cities worldwide by population, then up to n_english more from
English-speaking countries that weren't already in that global set."""
places._load() # synchronous parse; fills places._data (entries are pop-desc)
if not places._data:
raise SystemExit("GeoNames index failed to load (see logs); cannot generate cities.")
entries = places._data[0]
out: list[dict] = []
seen_slugs: set[str] = set()
for e in entries:
if len(out) >= n_global:
break
c = _to_city(e, seen_slugs)
if c:
out.append(c)
# Identity of the cities already chosen, so the English top-up skips them.
chosen = {(c["name"], c["admin1"], c["country_code"]) for c in out}
added = 0
for e in entries:
if added >= n_english:
break
if e[_CC] not in ENGLISH_CC:
continue
if (e[_NAME], e[_ADMIN1], e[_CC]) in chosen:
continue
c = _to_city(e, seen_slugs)
if c:
out.append(c)
added += 1
return out
def main() -> None:
n_global = int(sys.argv[1]) if len(sys.argv) > 1 else 500
n_english = int(sys.argv[2]) if len(sys.argv) > 2 else 250
cities = build(n_global, n_english)
with open(OUT_PATH, "w", encoding="utf-8") as f:
json.dump(cities, f, ensure_ascii=False, indent=0, separators=(",", ":"))
f.write("\n")
print(f"wrote {len(cities)} cities ({n_global} global + up to {n_english} English-market) -> {OUT_PATH}")
print("sample:", ", ".join(c["slug"] for c in cities[:8]))
if __name__ == "__main__":
main()