thermograph/store.py
Emi Griffith 576a239723 Compare/calendar mobile polish: overflow fix, sheet load button, 6-yr default, country in names (#80)
- Fix the mobile width blow-out: the distribution grid item (.cmp-dist-row) had no
  min-width:0, so each row expanded to the 9–10-column connected bar chart's
  min-content and widened the whole page (zoom-out, clipped filter sheet). Add
  grid-template-columns: minmax(0,1fr) + min-width:0 so the strip scrolls inside its
  own .ct-strip, and tighten .ct-col to 34px on phones.
- Add a Load/Refresh button inside the compare filter sheet's date-range block, wired
  to the same refresh()/isDirty(); editing the range on mobile no longer needs the
  sheet closed. The existing button stays in the controls (loads added places).
- Default date range on both pages is now January six years back → the present. On
  the calendar this is an explicit chunked range (the months=24 path is server-capped
  at ~2yr).
- Names now include country: reverse_geocode appends it to the place label; revgeo
  cache key gets a v2 prefix and PAYLOAD_VER bumps to p2 so labels/payloads
  repopulate.
- Location names read as a hierarchy: compact chips show just the lead segment (full
  name in the title/aria), and each ranked card shows the lead segment bold over a
  muted "city · region · country" line.

Frontend + a small backend name/cache change; no schema migration.
2026-07-12 06:32:10 +00:00

174 lines
6.8 KiB
Python

"""Persistent derived-data store — SQLite (WAL) at data/thermograph.sqlite.
The raw per-cell parquet records stay the source of truth; this database holds
only what's *derived* from them, so expensive work is done once per cell and
survives restarts/deploys instead of being recomputed per request, per view:
* ``derived`` — finished API response payloads (grade / calendar / day /
forecast), zlib-compressed JSON keyed by (kind, cell, request key) and
guarded by a validity ``token``. The token encodes everything the payload
depends on (payload schema version, history end date, recent-fetch stamp…);
a row whose stored token doesn't match the caller's current token is simply
a miss, so stale data can never be served — freshness is driven by the token
advancing, never by clock-based expiry.
* ``revgeo`` — reverse-geocode labels ("City, State" per ~cell), previously an
in-memory dict that re-hit Nominatim after every restart.
The store is a pure accelerator: every reader falls back to recomputing from
parquet when the database is missing, locked, or corrupt (all helpers swallow
their own errors), and deleting data/thermograph.sqlite is a safe full reset.
"""
import json
import os
import sqlite3
import threading
import time
import zlib
DB_PATH = os.path.join(os.path.dirname(__file__), "..", "data", "thermograph.sqlite")
_SCHEMA = """
CREATE TABLE IF NOT EXISTS derived (
kind TEXT NOT NULL, -- payload family: grade | calendar | day | forecast
cell_id TEXT NOT NULL,
key TEXT NOT NULL, -- request identity within the kind (dates/spans/params)
token TEXT NOT NULL, -- validity: mismatched token == miss (see module docstring)
payload BLOB NOT NULL, -- zlib-compressed JSON bytes
updated_at REAL NOT NULL,
PRIMARY KEY (kind, cell_id, key)
);
CREATE TABLE IF NOT EXISTS revgeo (
key TEXT PRIMARY KEY, -- "lat,lon" rounded to ~cell precision
label TEXT, -- NULL == lookup ran but found no label
updated_at REAL NOT NULL
);
"""
# SQLite connections aren't shareable across threads; uvicorn serves requests on
# a thread pool, so each thread lazily opens its own connection. WAL lets those
# readers run concurrently with the (serialized) writers.
_local = threading.local()
def _conn() -> sqlite3.Connection | None:
conn = getattr(_local, "conn", None)
if conn is not None:
return conn
try:
os.makedirs(os.path.dirname(DB_PATH), exist_ok=True)
conn = sqlite3.connect(DB_PATH, timeout=5.0)
conn.execute("PRAGMA journal_mode=WAL")
conn.execute("PRAGMA synchronous=NORMAL")
conn.executescript(_SCHEMA)
_local.conn = conn
return conn
except Exception: # noqa: BLE001 - the store is optional; readers fall back
return None
# --- derived payloads -------------------------------------------------------
def get_payload(kind: str, cell_id: str, key: str, token: str) -> bytes | None:
"""Raw JSON bytes of a cached payload, or None when absent or token-invalid."""
conn = _conn()
if conn is None:
return None
try:
row = conn.execute(
"SELECT token, payload FROM derived WHERE kind=? AND cell_id=? AND key=?",
(kind, cell_id, key),
).fetchone()
if row is None or row[0] != token:
return None
return zlib.decompress(row[1])
except Exception: # noqa: BLE001
return None
def get_json(kind: str, cell_id: str, key: str, token: str) -> dict | None:
"""Like get_payload but decoded — for callers assembling composite responses."""
body = get_payload(kind, cell_id, key, token)
return json.loads(body) if body is not None else None
def put_payload(kind: str, cell_id: str, key: str, token: str, payload: dict,
cache: bool = True) -> bytes:
"""Encode ``payload`` to compact JSON, persist it, and return the encoded bytes
(so the caller can serve the exact same bytes without re-serializing).
Pass ``cache=False`` to serve the payload without persisting it — used when a
best-effort field (the reverse-geocoded place name) failed to resolve, so the
incomplete result isn't cached for the life of the token.
allow_nan=False mirrors Starlette's JSONResponse: a NaN slipping through
grading should fail loudly here, not be cached as JS-unparseable `NaN`."""
body = json.dumps(payload, default=str, allow_nan=False,
separators=(",", ":")).encode()
conn = _conn()
if cache and conn is not None:
try:
conn.execute(
"INSERT OR REPLACE INTO derived VALUES (?,?,?,?,?,?)",
(kind, cell_id, key, token, zlib.compress(body, 6), time.time()),
)
conn.commit()
except Exception: # noqa: BLE001 - failing to cache must not fail the request
pass
return body
# --- reverse-geocode labels --------------------------------------------------
REVGEO_MISS_TTL = 3600.0 # a "no label" result may be transient (rate limit) — retry hourly
def revgeo_key(lat: float, lon: float) -> str:
# The "v2:" prefix versions the label format: v1 rows (no country) are bypassed
# and re-fetched so names pick up the trailing country component.
return f"v2:{round(lat, 3)},{round(lon, 3)}"
def get_revgeo(key: str) -> tuple[bool, str | None]:
"""(found, label). A stored NULL label counts as found (a cached miss) until
it ages past REVGEO_MISS_TTL, when the lookup becomes retryable."""
conn = _conn()
if conn is None:
return (False, None)
try:
row = conn.execute("SELECT label, updated_at FROM revgeo WHERE key=?", (key,)).fetchone()
if row is None:
return (False, None)
if row[0] is None and time.time() - row[1] > REVGEO_MISS_TTL:
return (False, None)
return (True, row[0])
except Exception: # noqa: BLE001
return (False, None)
def put_revgeo(key: str, label: str | None) -> None:
conn = _conn()
if conn is None:
return
try:
conn.execute("INSERT OR REPLACE INTO revgeo VALUES (?,?,?)", (key, label, time.time()))
conn.commit()
except Exception: # noqa: BLE001
pass
def stats() -> dict:
"""Row counts + on-disk size, for the migrate script's summary output."""
conn = _conn()
out = {"db_path": os.path.abspath(DB_PATH), "derived": 0, "revgeo": 0, "bytes": 0}
if conn is None:
return out
try:
out["derived"] = conn.execute("SELECT COUNT(*) FROM derived").fetchone()[0]
out["revgeo"] = conn.execute("SELECT COUNT(*) FROM revgeo").fetchone()[0]
# Recent writes may still live in the WAL sidecar — count both files.
out["bytes"] = sum(os.path.getsize(p) for p in (DB_PATH, DB_PATH + "-wal")
if os.path.exists(p))
except Exception: # noqa: BLE001
pass
return out