"""Pre-warm the archives for the curated city set (backend/cities.json) so the crawlable /climate pages render from cache and a search-engine crawl never bursts the archive API quota. Run at/after deploy: python warm_cities.py [--limit N] [--pace SECONDS] Idempotent: a cell whose archive is already cached is skipped. Fetches are paced (default 2s) to stay well under the archive API's rate limit. A cell that still has no cached archive when its page is first requested self-heals via get_history, so this is an optimization, not a hard dependency. """ import sys import time import cities import climate import grid def main(limit: int | None = None, pace: float = 2.0) -> None: todo = cities.all_cities() if limit: todo = todo[:limit] fetched = skipped = failed = 0 for i, c in enumerate(todo, 1): cell = grid.snap(c["lat"], c["lon"]) cached = climate.load_cached_history(cell) if cached is not None and not cached.is_empty(): skipped += 1 continue try: climate.get_history(cell) # fetch + cache the ~45-yr archive climate.get_recent_forecast(cell) # + the recent/forecast bundle (today block) fetched += 1 print(f"[{i}/{len(todo)}] warmed {c['slug']} ({cell['id']})") time.sleep(pace) except Exception as e: # noqa: BLE001 - keep going; the page self-heals later failed += 1 print(f"[{i}/{len(todo)}] FAILED {c['slug']}: {e}") time.sleep(pace) print(f"done: fetched={fetched} skipped(cached)={skipped} failed={failed}") if __name__ == "__main__": args = sys.argv[1:] lim = int(args[args.index("--limit") + 1]) if "--limit" in args else None pc = float(args[args.index("--pace") + 1]) if "--pace" in args else 2.0 main(limit=lim, pace=pc)