thermograph/warm_cities.py

80 lines
3.5 KiB
Python

"""Pre-warm the archives for the curated city set (backend/cities.json) so the
crawlable /climate pages render from cache and a search-engine crawl never bursts
the archive API quota. Run at/after deploy:
python warm_cities.py [--limit N] [--pace SECONDS]
Idempotent: a cell whose archive is already cached is skipped. Fetches are paced
(default 2s) to stay well under the archive API's rate limit. A cell that still
has no cached archive when its page is first requested self-heals via get_history,
so this is an optimization, not a hard dependency.
Deploy launches this script detached on every deploy with no overlap check, so a
slow previous run (still mid-warm) can still be going when the next deploy's
invocation starts. Two overlapping runs would double-spend archive-fetch quota on
the same cells and race climate._write_cache's parquet writes, so the CLI entry
point below claims a single-instance flock (see LOCK_PATH) before calling main().
"""
import os
import sys
import time
from api import homepage
from core import singleton
from data import cities
from data import climate
from data import grid
import paths
# core/singleton.py's flock guard, reused here for cross-*process* (not
# cross-worker) exclusion: the first invocation holds this for its process
# lifetime; a second one (an overlapping deploy) fails the non-blocking flock
# and stands down immediately.
LOCK_PATH = os.path.join(paths.DATA_DIR, "warm_cities.lock")
def main(limit: int | None = None, pace: float = 2.0) -> None:
todo = cities.all_cities()
if limit:
todo = todo[:limit]
fetched = skipped = failed = 0
for i, c in enumerate(todo, 1):
cell = grid.snap(c["lat"], c["lon"])
cached = climate.load_cached_history(cell)
if cached is not None and not cached.is_empty():
skipped += 1
continue
try:
climate.get_history(cell) # fetch + cache the ~45-yr archive
climate.get_recent_forecast(cell) # + the recent/forecast bundle (today block)
fetched += 1
print(f"[{i}/{len(todo)}] warmed {c['slug']} ({cell['id']})")
time.sleep(pace)
except Exception as e: # noqa: BLE001 - keep going; the page self-heals later
failed += 1
print(f"[{i}/{len(todo)}] FAILED {c['slug']}: {e}")
time.sleep(pace)
print(f"done: fetched={fetched} skipped(cached)={skipped} failed={failed}")
# Rebuild the homepage's "unusual right now" feed from whatever is now warm.
# Runs here as well as on the notifier loop so a fresh deploy has a populated
# feed immediately, and so it still refreshes when the notifier is disabled.
try:
feed = homepage.refresh()
print(f"homepage feed: {feed['considered']} cities graded, "
f"{len(feed['ranked'])} ranked")
except Exception as e: # noqa: BLE001 - the homepage renders without the feed
print(f"homepage feed: FAILED {e}")
if __name__ == "__main__":
if not singleton.claim(LOCK_PATH):
# Not an error: an overlapping run just means a previous deploy's warm is
# still in flight. Exit 0 so the deploy script doesn't treat this as a
# failure.
print("warm_cities: another instance is already running -- exiting")
sys.exit(0)
args = sys.argv[1:]
lim = int(args[args.index("--limit") + 1]) if "--limit" in args else None
pc = float(args[args.index("--pace") + 1]) if "--pace" in args else 2.0
main(limit=lim, pace=pc)