thermograph/gen_cities.py
Emi Griffith d17ac794fd Split the backend into domain packages (#217)
* Centralize filesystem paths in a single module

Add paths.py, which resolves the repo root once and derives the cache,
accounts DB, logs, templates, frontend and bundled-city-data locations
from it. Replace the 13 per-module `dirname(__file__)/..` anchors with
references to it, so a module's location no longer determines where the
app reads its data. Env overrides (accounts DB, VAPID, IndexNow) are
unchanged; every resolved path is byte-identical to before.

Groundwork for moving modules into packages without re-pointing paths.

Claude-Session: https://claude.ai/code/session_01XXxmNFy9cZ6Gh8Y9thZn62

* Split the backend into domain packages

Group the flat backend modules into packages that mirror their concerns:
  data/          climate, grading, scoring, grid, places, cities,
                 city_events, store
  web/           app, views, homepage, content, schemas
  notifications/ notify, digest, push, mailer, discord,
                 discord_interactions, discord_link
  accounts/      models, users, api_accounts, db
  core/          metrics, singleton, audit

Intra-project imports are rewritten to the package-qualified form. The
entry scripts (indexnow, warm_cities, migrate, gen_cities, gen_flavor)
and paths.py stay at the backend/ root, and backend/app.py becomes a
shim re-exporting web.app:app so the launch target stays `app:app` —
run.sh, the systemd units, and CI need no change.

Verified: full suite (318) passes, `uvicorn app:app` boots and serves
the home/SEO/static/API surfaces, and every root script imports clean.

Claude-Session: https://claude.ai/code/session_01XXxmNFy9cZ6Gh8Y9thZn62
2026-07-20 05:31:03 +00:00

164 lines
7.1 KiB
Python

"""Offline generator for backend/cities.json — the finite set of cities that get
crawlable climate pages (/climate/<slug>). Run occasionally to refresh the list:
python gen_cities.py [N] # default N=500 top metros by population
It reuses the GeoNames index that places.py already downloads/parses (calling
places._load() synchronously fills places._data), takes the top-N places by
population, and assigns each a stable, unique, URL-safe slug. Committing the output
keeps the routable city set explicit and reviewable, and decouples page-serving
from the async place-name loader.
"""
import json
import os
import re
import sys
import unicodedata
import paths
from data import places
OUT_PATH = paths.CITIES_JSON
# GeoNames entry tuple layout (see places._load): the fields we keep.
_NAME, _ADMIN1, _COUNTRY, _CC, _LAT, _LON, _POP = 1, 2, 3, 4, 5, 6, 7
def slugify(*parts: str) -> str:
"""ASCII, lowercase, hyphenated slug from name/admin/country parts."""
text = " ".join(p for p in parts if p)
text = unicodedata.normalize("NFKD", text).encode("ascii", "ignore").decode()
text = re.sub(r"[^a-zA-Z0-9]+", "-", text).strip("-").lower()
return re.sub(r"-{2,}", "-", text)
# Core English-speaking countries — the first English top-up tier.
ENGLISH_CC = {"US", "GB", "CA", "AU", "NZ", "IE", "ZA"}
# A second English top-up tier: the remaining countries where English is an official
# language, plus countries where >35% of the population speaks English (Eurobarometer
# 2012 "can hold a conversation in English" / EF EPI). Both drive English-language
# search demand. Editable — this is a judgment call, not a hard rule.
ENGLISH_EXTENDED_CC = {
# English official (beyond the core seven)
"IN", "PK", "PH", "SG", "HK", "MY", "LK", "PG", "FJ",
"NG", "KE", "GH", "UG", "TZ", "ZW", "ZM", "MW", "BW", "NA", "RW", "SL", "LR", "MU", "SS", "SZ", "LS", "GM", "SC",
"JM", "TT", "BB", "BS", "BZ", "GY", "GD", "LC", "VC", "AG", "DM", "KN", "MT",
# >35% English proficiency (non-official)
"NL", "SE", "DK", "NO", "IS", "FI", "DE", "AT", "BE", "CH", "LU", "CY", "SI", "GR", "EE", "LV", "LT", "FR", "IL",
}
# Values filter: exclude cities in countries that criminalize LGBTQ+ people (plus
# Pakistan and Indonesia by request), EXCEPT the individually-kept notable/tourist/
# high-English hubs in KEEP_SLUGS. Editable — regenerate cities.json after changes.
EXCLUDE_CC = {
"NG", "KE", "GH", "UG", "TZ", "ZM", "ZW", "MW", "SL", "LR", "GM", "ET", "SN", "CM", "EG",
"SA", "AE", "QA", "KW", "OM", "YE", "IR", "IQ", "SY", "AF", "BD", "LK", "MY", "BN", "MM",
"JM", "GD", "DM", "LC", "VC", "KN", "GY", "DZ", "MA", "LY", "SD", "SS", "ER", "MR",
"PK", "ID", "CN",
}
KEEP_SLUGS = {
"lagos-ng", "abuja-fct-ng", "kuala-lumpur-my", "malacca-melaka-my",
"kota-kinabalu-sabah-my", "kuching-sarawak-my", "nairobi-nairobi-county-ke",
"mombasa-mombasa-county-ke", "dar-es-salaam-dar-es-salaam-region-tz",
"zanzibar-zanzibar-urban-west-tz", "arusha-tz", "cairo-eg", "giza-eg", "alexandria-eg",
"dubai-ae", "abu-dhabi-ae", "casablanca-casablanca-settat-ma", "rabat-rabat-sale-kenitra-ma",
"fes-fes-meknes-ma", "accra-greater-accra-gh", "kumasi-ashanti-gh", "dhaka-dhaka-division-bd",
"kampala-central-region-ug", "dakar-sn", "lusaka-lusaka-province-zm", "harare-zw",
"yangon-mm", "colombo-western-province-lk", "kingston-jm", "addis-ababa-et", "tehran-ir",
"jeddah-mecca-region-sa", "baghdad-iq", "kabul-af", "karachi-sindh-pk", "multan-punjab-pk",
"lahore-punjab-pk", "hyderabad-sindh-pk", "islamabad-pk", "faisalabad-punjab-pk",
"rawalpindi-punjab-pk", "jakarta-id", "surabaya-east-java-id", "bekasi-west-java-id",
# China — only the 5 most English-friendly cities (per a where-to-speak-English guide).
"shanghai-cn", "beijing-cn", "shenzhen-guangdong-cn", "guangzhou-guangdong-cn", "chengdu-sichuan-cn",
}
def _base_slug(e) -> str:
name, admin1 = e[_NAME], e[_ADMIN1]
# Drop admin1 from the slug when it just repeats the city name
# (e.g. Tokyo/Tokyo, Singapore/Singapore) to avoid "tokyo-tokyo-jp".
admin_part = admin1 if admin1 and slugify(admin1) != slugify(name) else ""
return slugify(name, admin_part, e[_CC] or "")
def _excluded(e) -> bool:
return e[_CC] in EXCLUDE_CC and _base_slug(e) not in KEEP_SLUGS
def _to_city(e, seen_slugs: set[str]) -> dict | None:
name, admin1, country, cc = e[_NAME], e[_ADMIN1], e[_COUNTRY], e[_CC]
base = _base_slug(e)
if not base:
return None
slug = base
i = 2
while slug in seen_slugs: # disambiguate the rare collision
slug = f"{base}-{i}"
i += 1
seen_slugs.add(slug)
return {
"slug": slug, "name": name, "admin1": admin1,
"country": country, "country_code": cc,
"lat": round(e[_LAT], 5), "lon": round(e[_LON], 5),
"population": e[_POP],
}
def build(n_global: int = 500, n_english: int = 250, n_extended: int = 250) -> list[dict]:
"""Three tiers, population-descending, de-duplicated:
1. top n_global cities worldwide,
2. up to n_english more from core English-speaking countries,
3. up to n_extended more from the remaining English-official + >35%-English
countries — all not already chosen."""
places._load() # synchronous parse; fills places._data (entries are pop-desc)
if not places._data:
raise SystemExit("GeoNames index failed to load (see logs); cannot generate cities.")
entries = places._data[0]
out: list[dict] = []
seen_slugs: set[str] = set()
chosen_ids: set = set() # (name, admin1, cc) already added, so tiers don't overlap
def add_from(pred, limit: int) -> int:
added = 0
for e in entries:
if added >= limit:
break
if not pred(e):
continue
if _excluded(e): # values filter (keeps KEEP_SLUGS); backfills the rest
continue
ident = (e[_NAME], e[_ADMIN1], e[_CC])
if ident in chosen_ids:
continue
c = _to_city(e, seen_slugs)
if c:
out.append(c)
chosen_ids.add(ident)
added += 1
return added
add_from(lambda e: True, n_global) # tier 1: global
add_from(lambda e: e[_CC] in ENGLISH_CC, n_english) # tier 2: core English
add_from(lambda e: e[_CC] in ENGLISH_EXTENDED_CC, n_extended) # tier 3: extended English
return out
def main() -> None:
a = sys.argv[1:]
n_global = int(a[0]) if len(a) > 0 else 500
n_english = int(a[1]) if len(a) > 1 else 250
n_extended = int(a[2]) if len(a) > 2 else 250
cities = build(n_global, n_english, n_extended)
with open(OUT_PATH, "w", encoding="utf-8") as f:
json.dump(cities, f, ensure_ascii=False, indent=0, separators=(",", ":"))
f.write("\n")
print(f"wrote {len(cities)} cities ({n_global} global + {n_english} core-English + "
f"{n_extended} extended-English) -> {OUT_PATH}")
print("sample:", ", ".join(c["slug"] for c in cities[:8]))
if __name__ == "__main__":
main()