"""Persistent derived-data store — SQLite (WAL) at data/thermograph.sqlite. The raw per-cell parquet records stay the source of truth; this database holds only what's *derived* from them, so expensive work is done once per cell and survives restarts/deploys instead of being recomputed per request, per view: * ``derived`` — finished API response payloads (grade / calendar / day / forecast), zlib-compressed JSON keyed by (kind, cell, request key) and guarded by a validity ``token``. The token encodes everything the payload depends on (payload schema version, history end date, recent-fetch stamp…); a row whose stored token doesn't match the caller's current token is simply a miss, so stale data can never be served — freshness is driven by the token advancing, never by clock-based expiry. * ``revgeo`` — reverse-geocode labels ("City, State" per ~cell), previously an in-memory dict that re-hit Nominatim after every restart. The store is a pure accelerator: every reader falls back to recomputing from parquet when the database is missing, locked, or corrupt (all helpers swallow their own errors), and deleting data/thermograph.sqlite is a safe full reset. """ import json import os import sqlite3 import threading import time import zlib DB_PATH = os.path.join(os.path.dirname(__file__), "..", "data", "thermograph.sqlite") _SCHEMA = """ CREATE TABLE IF NOT EXISTS derived ( kind TEXT NOT NULL, -- payload family: grade | calendar | day | forecast cell_id TEXT NOT NULL, key TEXT NOT NULL, -- request identity within the kind (dates/spans/params) token TEXT NOT NULL, -- validity: mismatched token == miss (see module docstring) payload BLOB NOT NULL, -- zlib-compressed JSON bytes updated_at REAL NOT NULL, PRIMARY KEY (kind, cell_id, key) ); CREATE TABLE IF NOT EXISTS revgeo ( key TEXT PRIMARY KEY, -- "lat,lon" rounded to ~cell precision label TEXT, -- NULL == lookup ran but found no label updated_at REAL NOT NULL ); """ # SQLite connections aren't shareable across threads; uvicorn serves requests on # a thread pool, so each thread lazily opens its own connection. WAL lets those # readers run concurrently with the (serialized) writers. _local = threading.local() def _conn() -> sqlite3.Connection | None: conn = getattr(_local, "conn", None) if conn is not None: return conn try: os.makedirs(os.path.dirname(DB_PATH), exist_ok=True) conn = sqlite3.connect(DB_PATH, timeout=5.0) conn.execute("PRAGMA journal_mode=WAL") conn.execute("PRAGMA synchronous=NORMAL") conn.executescript(_SCHEMA) _local.conn = conn return conn except Exception: # noqa: BLE001 - the store is optional; readers fall back return None # --- derived payloads ------------------------------------------------------- def get_payload(kind: str, cell_id: str, key: str, token: str) -> bytes | None: """Raw JSON bytes of a cached payload, or None when absent or token-invalid.""" conn = _conn() if conn is None: return None try: row = conn.execute( "SELECT token, payload FROM derived WHERE kind=? AND cell_id=? AND key=?", (kind, cell_id, key), ).fetchone() if row is None or row[0] != token: return None return zlib.decompress(row[1]) except Exception: # noqa: BLE001 return None def get_json(kind: str, cell_id: str, key: str, token: str) -> dict | None: """Like get_payload but decoded — for callers assembling composite responses.""" body = get_payload(kind, cell_id, key, token) return json.loads(body) if body is not None else None def put_payload(kind: str, cell_id: str, key: str, token: str, payload: dict) -> bytes: """Encode ``payload`` to compact JSON, persist it, and return the encoded bytes (so the caller can serve the exact same bytes without re-serializing). allow_nan=False mirrors Starlette's JSONResponse: a NaN slipping through grading should fail loudly here, not be cached as JS-unparseable `NaN`.""" body = json.dumps(payload, default=str, allow_nan=False, separators=(",", ":")).encode() conn = _conn() if conn is not None: try: conn.execute( "INSERT OR REPLACE INTO derived VALUES (?,?,?,?,?,?)", (kind, cell_id, key, token, zlib.compress(body, 6), time.time()), ) conn.commit() except Exception: # noqa: BLE001 - failing to cache must not fail the request pass return body # --- reverse-geocode labels -------------------------------------------------- REVGEO_MISS_TTL = 86400.0 # a "no label" result may be transient — retry after a day def revgeo_key(lat: float, lon: float) -> str: return f"{round(lat, 3)},{round(lon, 3)}" def get_revgeo(key: str) -> tuple[bool, str | None]: """(found, label). A stored NULL label counts as found (a cached miss) until it ages past REVGEO_MISS_TTL, when the lookup becomes retryable.""" conn = _conn() if conn is None: return (False, None) try: row = conn.execute("SELECT label, updated_at FROM revgeo WHERE key=?", (key,)).fetchone() if row is None: return (False, None) if row[0] is None and time.time() - row[1] > REVGEO_MISS_TTL: return (False, None) return (True, row[0]) except Exception: # noqa: BLE001 return (False, None) def put_revgeo(key: str, label: str | None) -> None: conn = _conn() if conn is None: return try: conn.execute("INSERT OR REPLACE INTO revgeo VALUES (?,?,?)", (key, label, time.time())) conn.commit() except Exception: # noqa: BLE001 pass def stats() -> dict: """Row counts + on-disk size, for the migrate script's summary output.""" conn = _conn() out = {"db_path": os.path.abspath(DB_PATH), "derived": 0, "revgeo": 0, "bytes": 0} if conn is None: return out try: out["derived"] = conn.execute("SELECT COUNT(*) FROM derived").fetchone()[0] out["revgeo"] = conn.execute("SELECT COUNT(*) FROM revgeo").fetchone()[0] # Recent writes may still live in the WAL sidecar — count both files. out["bytes"] = sum(os.path.getsize(p) for p in (DB_PATH, DB_PATH + "-wal") if os.path.exists(p)) except Exception: # noqa: BLE001 pass return out