thermograph/tests/conftest.py
Emi Griffith c3bacdce0c Migrate backend dataframe layer from pandas to polars (#90)
* Migrate backend dataframe layer from pandas to polars

Replace pandas with polars across the backend, dropping both pandas and its
pyarrow parquet engine from the dependency set. numpy stays (the grading
percentile math is unchanged).

- climate.py: parquet IO, source→frame mappings, cache read/topup on polars.
  New _normalize_read casts the cached `date` column to pl.Date (older files
  were written by pandas as datetime64[ns]); frames now unify missing values as
  null so the grading boundary drops them consistently across sources.
- grading.py: keep the numpy percentile core; swap the frame→numpy bridge to
  .to_numpy()/.drop_nulls(), day-of-year/year to polars dt expressions, and the
  per-row loop to iter_rows(named=True).
- views.py: filter/anti-join/concat replace boolean-mask, isin and pd.concat;
  scalar dates are stdlib datetime.date; a local _months_before helper replaces
  DateOffset(months=) for the calendar-range default.
- app.py, migrate.py: request-date parsing uses datetime.date, removing pandas
  from the endpoint and migrate layers entirely.
- The date column is pl.Date end to end, eliminating the pandas normalize() calls
  and comparing cleanly against stdlib dates.

Payloads are unchanged: calendar, day, grade and forecast responses are
byte-for-byte identical to the pandas implementation on the same cached record.

Tests ported to polars fixtures, with added coverage for the combined feels-like
fallback, calendar month-offset (month-end/leap), and the concat/dedup
"fresher source wins" rule.

* Port notify.py to polars after merging dev's account system

Merge origin/dev (accounts + notification subscriptions) and carry the pandas→
polars migration into the newly added notify.py, which the merge brought in still
using pandas — with pandas removed from requirements this broke its import.

- notify.py: _candidate_rows filters/sorts the recent bundle with polars
  expressions and returns iter_rows dicts; date scalars are datetime.date;
  history/recent emptiness via is_empty().
- test_notify.py: synthetic history/rows built with polars + datetime.
2026-07-15 19:07:38 +00:00

109 lines
3.9 KiB
Python

"""Shared test setup: import path, offline guards, and synthetic weather data.
Everything here keeps the suite hermetic — no Open-Meteo, no Nominatim, no
GeoNames download, and no writes into the repo's data/ or logs/ folders.
"""
import datetime
import os
import sys
import tempfile
import threading
import numpy as np
import polars as pl
import pytest
BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, BACKEND)
import places # noqa: E402
# app.py calls places.start_loading() at import; pretend it already ran so no
# background GeoNames download starts. Tests build their own tiny index.
places._load_started = True
_TMP = tempfile.mkdtemp(prefix="thermograph-tests-")
import store # noqa: E402
# Keep derived-store writes out of the repo's data/ folder. Individual store
# tests re-point this per test; everything else shares one throwaway DB.
store.DB_PATH = os.path.join(_TMP, "store.sqlite")
import audit # noqa: E402
audit.AUDIT_DIR = os.path.join(_TMP, "logs", "audit")
audit.ERROR_DIR = os.path.join(_TMP, "logs", "errors")
def _years_before(d: datetime.date, years: int) -> datetime.date:
"""`d` shifted back `years` years, same month/day (Feb 29 → Feb 28)."""
try:
return d.replace(year=d.year - years)
except ValueError:
return d.replace(year=d.year - years, day=28)
def _daily(start: datetime.date, end: datetime.date) -> list[datetime.date]:
"""Inclusive daily date list from `start` to `end`."""
return [start + datetime.timedelta(days=i) for i in range((end - start).days + 1)]
def _with_doy(df: pl.DataFrame) -> pl.DataFrame:
return df.with_columns(pl.col("date").dt.ordinal_day().cast(pl.Int16).alias("doy"))
def make_history(years: int = 20, end: str | None = None, seed: int = 7) -> pl.DataFrame:
"""A plausible daily record (tmax/tmin/precip + doy), matching the columns
and dtypes climate.get_history returns for an older cache."""
end_ts = (datetime.date.fromisoformat(end) if end
else datetime.date.today() - datetime.timedelta(days=6))
dates = _daily(_years_before(end_ts, years), end_ts)
rng = np.random.default_rng(seed)
doy = np.array([d.timetuple().tm_yday for d in dates])
seasonal = 55 + 30 * np.sin((doy - 100) / 366.0 * 2 * np.pi)
tmax = seasonal + rng.normal(0, 8, len(dates))
tmin = tmax - 15 + rng.normal(0, 3, len(dates))
precip = np.where(rng.random(len(dates)) < 0.3, rng.gamma(1.5, 0.2, len(dates)), 0.0)
return _with_doy(pl.DataFrame({
"date": dates,
"tmax": np.round(tmax, 1),
"tmin": np.round(tmin, 1),
"precip": np.round(precip, 2),
}))
def make_recent(history: pl.DataFrame, future_days: int = 7, seed: int = 11) -> pl.DataFrame:
"""A recent+forecast bundle: from a couple of weeks before the archive's end
through `future_days` past today — the shape climate.get_recent_forecast returns."""
today = datetime.date.today()
start = history["date"].max() - datetime.timedelta(days=14)
dates = _daily(start, today + datetime.timedelta(days=future_days))
rng = np.random.default_rng(seed)
doy = np.array([d.timetuple().tm_yday for d in dates])
seasonal = 55 + 30 * np.sin((doy - 100) / 366.0 * 2 * np.pi)
tmax = seasonal + rng.normal(0, 8, len(dates))
return _with_doy(pl.DataFrame({
"date": dates,
"tmax": np.round(tmax, 1),
"tmin": np.round(tmax - 15, 1),
"precip": np.where(rng.random(len(dates)) < 0.3, 0.15, 0.0),
}))
@pytest.fixture(scope="session")
def history():
return make_history()
@pytest.fixture(scope="session")
def recent(history):
return make_recent(history)
@pytest.fixture
def tmp_store(tmp_path, monkeypatch):
"""store.py against a fresh database file, isolated per test."""
monkeypatch.setattr(store, "DB_PATH", str(tmp_path / "store.sqlite"))
monkeypatch.setattr(store, "_local", threading.local())
return store