* Migrate backend dataframe layer from pandas to polars Replace pandas with polars across the backend, dropping both pandas and its pyarrow parquet engine from the dependency set. numpy stays (the grading percentile math is unchanged). - climate.py: parquet IO, source→frame mappings, cache read/topup on polars. New _normalize_read casts the cached `date` column to pl.Date (older files were written by pandas as datetime64[ns]); frames now unify missing values as null so the grading boundary drops them consistently across sources. - grading.py: keep the numpy percentile core; swap the frame→numpy bridge to .to_numpy()/.drop_nulls(), day-of-year/year to polars dt expressions, and the per-row loop to iter_rows(named=True). - views.py: filter/anti-join/concat replace boolean-mask, isin and pd.concat; scalar dates are stdlib datetime.date; a local _months_before helper replaces DateOffset(months=) for the calendar-range default. - app.py, migrate.py: request-date parsing uses datetime.date, removing pandas from the endpoint and migrate layers entirely. - The date column is pl.Date end to end, eliminating the pandas normalize() calls and comparing cleanly against stdlib dates. Payloads are unchanged: calendar, day, grade and forecast responses are byte-for-byte identical to the pandas implementation on the same cached record. Tests ported to polars fixtures, with added coverage for the combined feels-like fallback, calendar month-offset (month-end/leap), and the concat/dedup "fresher source wins" rule. * Port notify.py to polars after merging dev's account system Merge origin/dev (accounts + notification subscriptions) and carry the pandas→ polars migration into the newly added notify.py, which the merge brought in still using pandas — with pandas removed from requirements this broke its import. - notify.py: _candidate_rows filters/sorts the recent bundle with polars expressions and returns iter_rows dicts; date scalars are datetime.date; history/recent emptiness via is_empty(). - test_notify.py: synthetic history/rows built with polars + datetime.
143 lines
6.5 KiB
Python
143 lines
6.5 KiB
Python
"""The payload layer: pure builders, span clamping, and the layering guarantee
|
|
that offline callers (migrate) can use it without dragging in the web stack."""
|
|
import datetime
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
|
|
import polars as pl
|
|
import pytest
|
|
|
|
import climate
|
|
import views
|
|
|
|
CELL = {"id": "1642_-4223", "center_lat": 47.6087, "center_lon": -122.29377}
|
|
|
|
|
|
def test_views_and_migrate_import_without_the_web_stack():
|
|
"""migrate.py must stay runnable offline: importing the payload layer may
|
|
not construct the FastAPI app or start the places-index download."""
|
|
code = ("import sys; import views, migrate; "
|
|
"assert 'fastapi' not in sys.modules, 'views/migrate pulled in FastAPI'; "
|
|
"assert 'app' not in sys.modules, 'views/migrate imported the web app'; "
|
|
"import places; assert places._load_started is False")
|
|
backend = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
subprocess.run([sys.executable, "-c", code], cwd=backend, check=True)
|
|
|
|
|
|
# ---- cache identity --------------------------------------------------------------
|
|
# These formats are shared by the endpoints, the /cell bundle and migrate; pin
|
|
# them so a change is always deliberate (an accidental drift silently strands
|
|
# every existing derived row — change them only alongside a PAYLOAD_VER bump).
|
|
|
|
def test_cache_identity_formats_are_pinned(history):
|
|
t = datetime.date(2026, 6, 15)
|
|
assert views.grade_key(t, 14, 7) == "2026-06-15:14:7"
|
|
assert views.calendar_key(t, datetime.date(2026, 6, 20), 24) == "2026-06-15:2026-06-20:24"
|
|
assert views.day_key(t) == "2026-06-15"
|
|
assert views.forecast_key(t, 7) == "2026-06-15:7"
|
|
assert views.history_token(history) == f"{views.PAYLOAD_VER}:{views.hist_end(history)}"
|
|
|
|
|
|
def test_recent_token_composes_history_and_stamp(history, monkeypatch):
|
|
monkeypatch.setattr(climate, "recent_stamp", lambda cid: "stamp")
|
|
assert views.recent_token(history, "1_2") == f"{views.history_token(history)}:stamp"
|
|
|
|
|
|
def test_day_token_expires_hourly_only_beyond_the_archive(history):
|
|
last = history["date"].max()
|
|
assert views.day_token(history, last) == views.history_token(history)
|
|
future = views.day_token(history, last + datetime.timedelta(days=1))
|
|
assert future.startswith(views.history_token(history) + ":h")
|
|
|
|
|
|
# ---- cal_span clamping ---------------------------------------------------------
|
|
|
|
def _hist(start, end):
|
|
# cal_span only reads the record's min/max date, so two rows pin the span.
|
|
return pl.DataFrame({"date": [datetime.date.fromisoformat(start),
|
|
datetime.date.fromisoformat(end)]})
|
|
|
|
|
|
def test_cal_span_defaults_to_months_back_same_day_of_month():
|
|
start_ts, end_ts = views.cal_span(_hist("2020-01-01", "2026-06-29"), None, None, 24)
|
|
assert end_ts == datetime.date(2026, 6, 29)
|
|
assert start_ts == datetime.date(2024, 6, 29)
|
|
|
|
|
|
def test_cal_span_clamps_to_the_record():
|
|
h = _hist("2025-03-01", "2026-06-29") # record shorter than the 2-year cap
|
|
start_ts, end_ts = views.cal_span(h, "2020-01-01", None, 24)
|
|
assert start_ts == datetime.date(2025, 3, 1) # can't start before the record
|
|
_, end_ts = views.cal_span(h, None, "2030-01-01", 24)
|
|
assert end_ts == datetime.date(2026, 6, 29) # nor end past it
|
|
|
|
|
|
def test_cal_span_caps_at_two_years():
|
|
start_ts, end_ts = views.cal_span(_hist("2018-01-01", "2026-06-29"),
|
|
"2018-01-01", "2026-06-29", 24)
|
|
assert (end_ts - start_ts).days == views.CAL_MAX_SPAN_DAYS - 1
|
|
|
|
|
|
def test_cal_span_never_inverts():
|
|
start_ts, end_ts = views.cal_span(_hist("2020-01-01", "2026-06-29"),
|
|
"2026-06-01", "2021-01-01", 24)
|
|
assert start_ts == end_ts
|
|
|
|
|
|
def test_months_before_clamps_month_end_and_leap():
|
|
"""Calendar-aware month subtraction (replaces pandas DateOffset): land on the
|
|
same day-of-month, clamped to the target month's last valid day."""
|
|
assert views._months_before(datetime.date(2026, 3, 31), 1) == datetime.date(2026, 2, 28)
|
|
assert views._months_before(datetime.date(2024, 3, 31), 1) == datetime.date(2024, 2, 29)
|
|
assert views._months_before(datetime.date(2026, 1, 15), 14) == datetime.date(2024, 11, 15)
|
|
|
|
|
|
def test_attach_dry_streaks_prefers_the_fresher_source_on_shared_dates():
|
|
"""De-dup keeps the recent/forecast row over the archive one for a shared date
|
|
(concat archive-first, keep='last') — the fresher precip drives the streak."""
|
|
d = datetime.date(2026, 1, 1)
|
|
hist = pl.DataFrame({"date": [d], "precip": [1.0]}) # archive: it rained (streak resets)
|
|
rec = pl.DataFrame({"date": [d], "precip": [0.0]}) # fresher: dry (streak counts)
|
|
graded = [{"date": d.isoformat()}]
|
|
views._attach_dry_streaks(graded, hist, rec) # archive first, recent last
|
|
assert graded[0]["dsr"] == 1 # recent (dry) won
|
|
|
|
|
|
# ---- builders -------------------------------------------------------------------
|
|
|
|
def test_build_grade_window_and_shape(history, recent):
|
|
target = datetime.date.today()
|
|
payload = views.build_grade(CELL, target, 14, history, recent,
|
|
{"cached": True}, "Testville")
|
|
assert payload["target_date"] == target.isoformat()
|
|
days = [d["date"] for d in payload["recent"]]
|
|
assert days == sorted(days, reverse=True) # newest first
|
|
assert payload["climatology"]["tmax"] is not None
|
|
assert all(d["dsr"] is not None for d in payload["recent"])
|
|
|
|
|
|
def test_build_day_pulls_future_obs_from_recent(history, recent):
|
|
today = datetime.date.today()
|
|
payload = views.build_day(CELL, history, today, "Testville", recent=recent)
|
|
assert payload["detail"]["date"] == today.isoformat()
|
|
assert payload["detail"]["metrics"]["tmax"]["obs"] is not None
|
|
|
|
|
|
def test_build_day_survives_recent_fetch_failure(history, monkeypatch):
|
|
def boom(cell):
|
|
raise RuntimeError("upstream down")
|
|
monkeypatch.setattr(climate, "get_recent_forecast", boom)
|
|
today = datetime.date.today()
|
|
payload = views.build_day(CELL, history, today, "Testville")
|
|
assert payload["detail"]["metrics"]["tmax"]["obs"] is None # climatology only
|
|
assert payload["detail"]["metrics"]["tmax"]["ladder"] is not None
|
|
|
|
|
|
def test_build_forecast_only_future_days(history, recent):
|
|
today = datetime.date.today()
|
|
payload = views.build_forecast(CELL, 7, history, recent, today, None)
|
|
days = [d["date"] for d in payload["recent"]]
|
|
assert days == sorted(days, reverse=True) # furthest-out first
|
|
assert min(days) > today.isoformat()
|
|
assert payload["forecast"] is True
|