thermograph/backend/tests/data/test_climate.py
Emi Griffith 5569bcbc0a
All checks were successful
secrets-guard / encrypted (pull_request) Successful in 7s
Flip recent/forecast leg off Open-Meteo (NASA past + MET Norway forward)
The recent-observations + forward-forecast bundle is now built without Open-Meteo:
the recent observed window comes from a NASA POWER range (measured, via the range
fetch added for the history flip) and the forward days from MET Norway, merged by
date. Meteostat fills the gusts neither source carries — measured for the observed
days, estimated from wind for the forecast days.

Open-Meteo's forecast API is demoted to the fallback, still gated by its existing
_forecast_cooldown_until rate-limit cooldown; then a stale cache. MET Norway
switched to /compact (same fields, smaller payload) and the bundle TTL relaxed
1h -> 4h. NASA's near-real-time lag leaves a 1-2 day recent-edge gap that grades
as missing, bracketed by history behind and forecast ahead.

Tests updated to the new source order (NASA+MET merge, Open-Meteo fallback, and
the forecast cooldown now gating that fallback); deletion of Open-Meteo is not
done — it stays as the dormant fallback.
2026-07-23 06:32:31 -07:00

579 lines
26 KiB
Python

"""The source→frame mappings and cache-write path (pure parts of climate.py —
no network)."""
import datetime
import polars as pl
import pytest
from data import climate
def _om_daily(n=3):
"""A minimal Open-Meteo daily response."""
return {
"time": [f"2026-06-{d:02d}" for d in range(1, n + 1)],
"temperature_2m_max": [80.0, 90.0, None],
"temperature_2m_min": [60.0, 70.0, 55.0],
"precipitation_sum": [0.0, 0.25, 0.1],
"wind_speed_10m_max": [10.0, 20.0, 5.0],
"wind_gusts_10m_max": [15.0, 30.0, 8.0],
"apparent_temperature_max": [82.0, 95.0, 70.0],
"apparent_temperature_min": [58.0, 68.0, 50.0],
"relative_humidity_2m_mean": [50.0, 60.0, 70.0],
# Sunshine vs daylight seconds -> `sun` fraction (full sun, half, quarter).
"sunshine_duration": [43200.0, 21600.0, 10800.0],
"daylight_duration": [43200.0, 43200.0, 43200.0],
}
def test_to_frame_schema_and_day_filter():
df = climate._to_frame(_om_daily())
# The None tmax day is dropped: a usable climate day needs a real high/low.
assert len(df) == 2
assert df["doy"].dtype == pl.Int16
assert set(df.columns) == {"date", "tmax", "tmin", "precip", "wind", "gust",
"humid", "fmax", "fmin", "feels", "sun", "doy"}
def test_to_frame_derives_sun_fraction():
df = climate._to_frame(_om_daily())
# sun = sunshine_duration / daylight_duration, clamped to [0, 1].
assert df["sun"].to_list() == [1.0, 0.5]
def test_to_frame_sun_null_without_sunshine():
daily = _om_daily()
del daily["sunshine_duration"], daily["daylight_duration"]
df = climate._to_frame(daily)
assert "sun" in df.columns and df["sun"].is_null().all()
def test_to_frame_tolerates_missing_series():
daily = _om_daily()
del daily["wind_gusts_10m_max"], daily["relative_humidity_2m_mean"]
df = climate._to_frame(daily)
assert df["gust"].is_null().all() and df["humid"].is_null().all()
assert df["tmax"].is_not_null().all() # required series unaffected
def test_finalize_frame_backfills_absent_optional_columns():
# A source frame carrying only the required base columns is finalized into the
# full schema, every absent optional metric added as an all-null column.
bare = pl.DataFrame({
"date": [datetime.date(2026, 6, 1), datetime.date(2026, 6, 2)],
"tmax": [80.0, 82.0], "tmin": [60.0, 61.0], "precip": [0.0, 0.1],
})
df = climate._finalize_frame(bare)
for c in climate.OPTIONAL_COLS:
assert c in df.columns and df[c].is_null().all()
assert all(c in df.columns for c in climate.NEW_COLS) # incl. derived `feels`
def test_combined_feels_picks_the_extreme_side():
df = climate._to_frame(_om_daily())
# Day 2: fmax 95 is 30 past the 65°F comfort point vs fmin 68 only 3 below —
# the hot side wins; both days here are hot-side days.
assert df.filter(pl.col("date") == datetime.date(2026, 6, 2))["feels"].item() == 95.0
def test_combined_feels_falls_back_to_the_present_side():
"""When one apparent-temperature side is missing, `feels` uses whichever side
is present (the null-vs-value coalesce must reproduce the old NaN fallback)."""
daily = _om_daily()
daily["apparent_temperature_max"] = [None, None, None] # hot side absent
df = climate._to_frame(daily)
row = df.filter(pl.col("date") == datetime.date(2026, 6, 1)).row(0, named=True)
assert row["feels"] == 58.0 # falls back to apparent low
def test_nasa_to_frame_converts_units_and_fills():
param = {
"T2M_MAX": {"20260601": 30.0, "20260602": -999.0}, # °C; -999 = missing
"T2M_MIN": {"20260601": 20.0, "20260602": 15.0},
"PRECTOTCORR": {"20260601": 25.4, "20260602": 0.0}, # mm
"WS10M_MAX": {"20260601": 10.0, "20260602": 5.0}, # m/s
"RH2M": {"20260601": 50.0, "20260602": 60.0},
}
df = climate._nasa_to_frame(param)
assert len(df) == 1 # the missing-tmax day dropped
row = df.row(0, named=True)
assert row["tmax"] == 86.0 # 30°C
assert row["precip"] == 1.0 # 25.4mm = 1in
assert round(row["wind"], 1) == 22.4 # 10 m/s in mph
assert row["gust"] is None # POWER has no gusts (missing -> null)
assert row["fmax"] > row["tmax"] # ≥80°F: the NWS heat index applies
def _metno_step(time, t, rh, wind, p1=None, p6=None):
"""One MET Norway timeseries step (instant details + optional precip blocks)."""
data = {"instant": {"details": {
"air_temperature": t, "relative_humidity": rh, "wind_speed": wind}}}
if p1 is not None:
data["next_1_hours"] = {"details": {"precipitation_amount": p1}}
if p6 is not None:
data["next_6_hours"] = {"details": {"precipitation_amount": p6}}
return {"time": time, "data": data}
def test_metno_to_frame_aggregates_daily_and_converts_units():
props = {"timeseries": [
# Day 1: two hourly steps. The first also carries a next_6_hours block, which
# must be IGNORED (next_1_hours wins) so precip isn't double-counted.
_metno_step("2026-07-16T00:00:00Z", 20.0, 50.0, 2.0, p1=0.5, p6=3.0),
_metno_step("2026-07-16T01:00:00Z", 25.0, 60.0, 4.0, p1=0.5),
# Day 2: a 6-hourly step (only next_6_hours) + an instant-only tail step.
_metno_step("2026-07-17T00:00:00Z", 10.0, 80.0, 10.0, p6=6.0),
_metno_step("2026-07-17T06:00:00Z", 12.0, 70.0, 8.0),
]}
df = climate._metno_to_frame(props)
assert len(df) == 2
d1 = df.filter(pl.col("date") == datetime.date(2026, 7, 16)).row(0, named=True)
assert d1["tmax"] == 77.0 and d1["tmin"] == 68.0 # 25°C / 20°C -> °F
assert round(d1["precip"], 4) == round(1.0 / 25.4, 4) # 0.5+0.5 mm (not +3.0) -> in
assert round(d1["wind"], 1) == 8.9 # max 4 m/s -> mph
assert d1["gust"] is None # MET Norway has no gusts
assert d1["feels"] is not None
d2 = df.filter(pl.col("date") == datetime.date(2026, 7, 17)).row(0, named=True)
assert round(d2["precip"], 4) == round(6.0 / 25.4, 4) # only the 6-hour block
def test_recent_forecast_merges_nasa_past_and_metno_forward(monkeypatch, tmp_path):
"""The recent/forecast bundle is built from a NASA POWER recent-past range plus the
MET Norway forward forecast, merged by date (Open-Meteo not consulted)."""
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
cell = {"id": "rf_primary", "center_lat": 47.6, "center_lon": -122.3}
past = climate._finalize_approximated(pl.DataFrame({
"date": [datetime.date(2026, 7, 10), datetime.date(2026, 7, 11)],
"tmax": [70.0, 72.0], "tmin": [50.0, 51.0], "precip": [0.0, 0.1],
"wind": [5.0, 6.0], "humid": [60.0, 55.0],
}))
fwd = climate._finalize_approximated(pl.DataFrame({
"date": [datetime.date(2026, 7, 16), datetime.date(2026, 7, 17)],
"tmax": [80.0, 82.0], "tmin": [60.0, 61.0], "precip": [0.0, 0.0],
"wind": [3.0, 4.0], "humid": [40.0, 45.0],
}))
monkeypatch.setattr(climate, "_fetch_history_nasa", lambda c, start, end: past)
monkeypatch.setattr(climate, "_fetch_forecast_metno", lambda c: fwd)
monkeypatch.setattr(climate.meteostat, "fill_gusts", lambda lat, lon, df: df)
def _om_should_not_run(c): raise AssertionError("Open-Meteo forecast should not run")
monkeypatch.setattr(climate, "_fetch_recent_forecast_om", _om_should_not_run)
df = climate._load_recent_forecast(cell)
assert df["date"].to_list() == [
datetime.date(2026, 7, 10), datetime.date(2026, 7, 11),
datetime.date(2026, 7, 16), datetime.date(2026, 7, 17)]
def test_recent_forecast_falls_back_to_open_meteo(monkeypatch, tmp_path):
"""When the NASA + MET Norway primary fails, the Open-Meteo forecast API serves."""
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_forecast_cooldown_until", 0.0)
cell = {"id": "rf_fallback", "center_lat": 47.6, "center_lon": -122.3}
def _primary_down(c): raise RuntimeError("NASA + MET Norway down")
monkeypatch.setattr(climate, "_fetch_recent_forecast", _primary_down)
om = climate._to_frame(_om_daily(3))
monkeypatch.setattr(climate, "_fetch_recent_forecast_om", lambda c: om)
df = climate._load_recent_forecast(cell)
assert df.height == om.height
def test_recent_forecast_serves_stale_cache_when_all_sources_fail(monkeypatch, tmp_path):
"""With every source down (the NASA + MET Norway primary and the Open-Meteo
fallback), an existing (stale) cache is served rather than failing — and it is NOT
rewritten, so its mtime stays old and the next request still retries upstream first."""
import os
import time
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
cell = {"id": "stale_cell", "center_lat": 47.6, "center_lon": -122.3}
# Seed an rf cache, then age it well past the TTL so it counts as stale.
stale = climate._to_frame(_om_daily())
path = climate._rf_cache_path(cell["id"])
climate._write_cache(stale, path)
old = time.time() - (climate.FORECAST_TTL_HOURS + 5) * 3600
os.utime(path, (old, old))
def all_down(url, params, timeout, *, phase, headers=None, attempts=climate.MAX_ATTEMPTS):
raise RuntimeError(f"{phase} down")
monkeypatch.setattr(climate, "_request", all_down)
df = climate._load_recent_forecast(cell)
assert df.height == stale.height # served the stale cache
assert abs(os.path.getmtime(path) - old) < 2 # not rewritten (mtime unchanged)
def test_om_daily_params_carries_the_window():
cell = {"center_lat": 47.6, "center_lon": -122.3}
p = climate._om_daily_params(cell, start_date="2026-01-01", end_date="2026-02-01")
assert p["latitude"] == 47.6 and p["daily"] == climate.DAILY_VARS
assert p["temperature_unit"] == "fahrenheit"
assert p["start_date"] == "2026-01-01" and p["end_date"] == "2026-02-01"
p2 = climate._om_daily_params(cell, past_days=25, forecast_days=8)
assert p2["past_days"] == 25 and "start_date" not in p2
def test_write_cache_strips_the_derived_doy(tmp_path):
df = climate._to_frame(_om_daily())
path = str(tmp_path / "cell.parquet")
climate._write_cache(df, path)
stored = pl.read_parquet(path)
assert "doy" not in stored.columns
assert len(stored) == len(df)
def _wb_frame(tmax, humid, tmin=None):
"""Frame with raw RH (`humid`) for the wet-bulb / humidity derivations."""
n = len(tmax)
return pl.DataFrame({
"date": [datetime.date(2026, 7, 1) + datetime.timedelta(days=i) for i in range(n)],
"tmax": [float(x) for x in tmax],
"tmin": [float(t) for t in (tmin or [x - 15 for x in tmax])],
"humid": [float(h) for h in humid],
})
def test_wetbulb_stull_spot_values():
# Stull (2011): 20 °C / 50% -> ~13.7 °C; 30 °C / 80% -> ~27.2 °C. In °F here.
df = climate._derive_wetbulb(_wb_frame([68.0, 86.0], [50.0, 80.0]))
wb = df["wetbulb"].to_list()
assert wb[0] == pytest.approx((13.7 * 9 / 5) + 32, abs=0.6)
assert wb[1] == pytest.approx((27.2 * 9 / 5) + 32, abs=0.6)
def test_wetbulb_never_exceeds_dry_bulb():
df = climate._derive_wetbulb(_wb_frame([40.0, 60.0, 85.0, 100.0], [20.0, 55.0, 70.0, 95.0]))
for tmax, wb in zip(df["tmax"], df["wetbulb"]):
assert wb is not None and wb <= tmax + 0.05
def test_wetbulb_null_outside_validity_range():
# RH 3% (< 5) and a scorching 130 °F (> 50 °C) both fall outside Stull's fit.
df = climate._derive_wetbulb(_wb_frame([70.0, 130.0], [3.0, 40.0]))
assert df["wetbulb"].to_list() == [None, None]
def test_wetbulb_noop_without_humidity():
df = pl.DataFrame({"tmax": [70.0], "tmin": [55.0]})
assert "wetbulb" not in climate._derive_wetbulb(df).columns
def test_derive_metrics_adds_wetbulb_and_absolute_humidity():
out = climate._derive_metrics(_wb_frame([68.0], [50.0]))
assert "wetbulb" in out.columns
# _derive_humidity replaces raw RH (50) with absolute humidity (g/m³, ~8-9).
assert out["humid"][0] < 30 and out["humid"][0] != 50.0
def test_history_fetch_targets_archive_url_with_seamless_model(monkeypatch):
"""Historical fetches hit ARCHIVE_URL (env-overridable for self-hosting) and pin
the ERA5 seamless model, so a self-hosted Open-Meteo serves the same 0.1° blend."""
seen = {}
class Resp:
def json(self): return {"daily": _om_daily(3)}
def fake_request(url, params, timeout, *, phase, headers=None, attempts=climate.MAX_ATTEMPTS):
seen["url"], seen["params"], seen["phase"] = url, dict(params), phase
return Resp()
monkeypatch.setattr(climate, "_request", fake_request)
cell = {"center_lat": 47.6, "center_lon": -122.3}
climate._fetch_history(cell)
assert seen["url"] == climate.ARCHIVE_URL
assert seen["params"]["models"] == "era5_seamless"
assert seen["phase"] == "history_fetch"
climate._fetch_history_range(cell, "2026-06-01", "2026-06-10") # tail top-up
assert seen["url"] == climate.ARCHIVE_URL
assert seen["params"]["models"] == "era5_seamless"
def test_daily_params_carry_no_model_by_default():
# The shared param builder (also used by the forecast path) stays model-free;
# only the archive fetches add models=era5_seamless.
p = climate._om_daily_params({"center_lat": 1.0, "center_lon": 2.0}, past_days=7)
assert "models" not in p
assert climate.ARCHIVE_URL.endswith("/v1/archive") # default (no env override in tests)
def _full_history_frame(n=4000):
"""A plausibly-full archive frame (>= MIN_ARCHIVE_DAYS rows), standard columns."""
start = datetime.date(1990, 1, 1)
dates = [start + datetime.timedelta(days=i) for i in range(n)]
return climate._finalize_frame(pl.DataFrame({
"date": dates,
"tmax": [70.0] * n, "tmin": [50.0] * n, "precip": [0.0] * n,
"wind": [5.0] * n, "gust": [8.0] * n, "humid": [60.0] * n,
"fmax": [70.0] * n, "fmin": [50.0] * n,
}))
def test_nasa_is_primary_history_source(monkeypatch, tmp_path):
"""NASA POWER is the primary history source: a full NASA frame is accepted and
Open-Meteo is never consulted."""
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_archive_cooldown_until", 0.0)
cell = {"id": "nasa_primary", "center_lat": 47.6, "center_lon": -122.3}
full = _full_history_frame()
monkeypatch.setattr(climate, "_fetch_history_nasa", lambda c: full)
def _om_should_not_run(c): raise AssertionError("Open-Meteo should not be called")
monkeypatch.setattr(climate, "_fetch_history", _om_should_not_run)
df, meta = climate._load_history(cell)
assert meta["source"] == "nasa-power"
assert df.height == full.height
def test_falls_back_to_open_meteo_when_nasa_unavailable(monkeypatch, tmp_path):
"""When NASA POWER fails, the load falls back to the Open-Meteo archive."""
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_archive_cooldown_until", 0.0)
cell = {"id": "om_fallback", "center_lat": 47.6, "center_lon": -122.3}
full = _full_history_frame()
def _nasa_down(c): raise RuntimeError("NASA POWER outage")
monkeypatch.setattr(climate, "_fetch_history_nasa", _nasa_down)
monkeypatch.setattr(climate, "_fetch_history", lambda c: full)
df, meta = climate._load_history(cell)
assert meta["source"] == "open-meteo"
assert df.height == full.height
def test_short_nasa_is_rejected_and_falls_back_to_open_meteo(monkeypatch, tmp_path):
"""A too-short NASA response (a partial/unavailable point) is rejected rather than
cached as a complete record, and Open-Meteo serves instead."""
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_archive_cooldown_until", 0.0)
cell = {"id": "short_nasa", "center_lat": 47.6, "center_lon": -122.3}
short = climate._to_frame(_om_daily(3)) # 2 usable days « threshold
full = _full_history_frame() # >= MIN_ARCHIVE_DAYS
assert short.height < climate.MIN_ARCHIVE_DAYS <= full.height
monkeypatch.setattr(climate, "_fetch_history_nasa", lambda c: short)
monkeypatch.setattr(climate, "_fetch_history", lambda c: full)
df, meta = climate._load_history(cell)
assert meta["source"] == "open-meteo" # rejected the short NASA response
assert df.height == full.height
def test_history_tail_prefers_nasa(monkeypatch):
"""The tail top-up fetches NASA POWER first; Open-Meteo's range fetch is not
touched when NASA succeeds."""
cell = {"center_lat": 1.0, "center_lon": 2.0}
sentinel = object()
monkeypatch.setattr(climate, "_fetch_history_nasa",
lambda c, start=None, end=None: sentinel)
def _om_should_not_run(c, s, e): raise AssertionError("Open-Meteo tail should not run")
monkeypatch.setattr(climate, "_fetch_history_range", _om_should_not_run)
assert climate._fetch_history_tail(cell, "2026-06-01", "2026-06-10") is sentinel
def test_history_tail_falls_back_to_open_meteo(monkeypatch):
"""When NASA fails and Open-Meteo isn't cooling down, the tail falls back to the
Open-Meteo archive range."""
monkeypatch.setattr(climate, "_archive_cooldown_until", 0.0)
cell = {"center_lat": 1.0, "center_lon": 2.0}
def _nasa_down(c, start=None, end=None): raise RuntimeError("NASA down")
monkeypatch.setattr(climate, "_fetch_history_nasa", _nasa_down)
sentinel = object()
monkeypatch.setattr(climate, "_fetch_history_range", lambda c, s, e: sentinel)
assert climate._fetch_history_tail(cell, "2026-06-01", "2026-06-10") is sentinel
# --- forecast cooldown (mirrors the archive path's, tracked separately) -----
def test_forecast_cooldown_skips_the_open_meteo_fallback(monkeypatch, tmp_path):
"""While the forecast cooldown is active, the Open-Meteo FALLBACK is skipped: with
the keyless primary (NASA + MET Norway) also down and no cache, the load surfaces
WeatherUnavailable without ever calling Open-Meteo."""
import time as time_mod
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_forecast_cooldown_until", time_mod.time() + 60)
cell = {"id": "fc_cooldown_cell", "center_lat": 47.6, "center_lon": -122.3}
def _primary_down(c): raise RuntimeError("NASA + MET Norway down")
monkeypatch.setattr(climate, "_fetch_recent_forecast", _primary_down)
def _om_should_not_run(c):
raise AssertionError("Open-Meteo fallback must not run during cooldown")
monkeypatch.setattr(climate, "_fetch_recent_forecast_om", _om_should_not_run)
with pytest.raises(climate.WeatherUnavailable):
climate._load_recent_forecast(cell)
def test_forecast_429_sets_its_own_cooldown(monkeypatch, tmp_path):
"""A 429 from the forecast endpoint sets _forecast_cooldown_until (NOT
_archive_cooldown_until -- the two upstream endpoints have independent
quotas) and surfaces as WeatherUnavailable when the backup is also down."""
import time as time_mod
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_forecast_cooldown_until", 0.0)
monkeypatch.setattr(climate, "_archive_cooldown_until", 0.0)
cell = {"id": "fc_429_cell", "center_lat": 47.6, "center_lon": -122.3}
class FakeResponse:
status_code = 429
def json(self): return {"reason": "rate limited"}
class FakeRateLimitError(Exception):
def __init__(self):
self.response = FakeResponse()
def fake_request(url, params, timeout, *, phase, headers=None, attempts=climate.MAX_ATTEMPTS):
if url == climate.FORECAST_URL:
raise FakeRateLimitError()
raise RuntimeError("metno also down")
monkeypatch.setattr(climate, "_request", fake_request)
with pytest.raises(climate.WeatherUnavailable):
climate._load_recent_forecast(cell)
assert climate._forecast_cooldown_until > time_mod.time()
assert climate._archive_cooldown_until == 0.0 # the archive cooldown is untouched
def test_forecast_cooldown_does_not_block_archive_fetch(monkeypatch, tmp_path):
"""The two cooldowns are independent state: an active forecast cooldown must
not gate _load_history's archive fetch."""
import time as time_mod
monkeypatch.setattr(climate, "CACHE_DIR", str(tmp_path))
monkeypatch.setattr(climate, "_forecast_cooldown_until", time_mod.time() + 60)
monkeypatch.setattr(climate, "_archive_cooldown_until", 0.0)
cell = {"id": "fc_indep_cell", "center_lat": 47.6, "center_lon": -122.3}
full = _full_history_frame()
monkeypatch.setattr(climate, "_fetch_history", lambda c: full)
def _nasa_should_not_run(c): raise AssertionError("NASA should not be called")
monkeypatch.setattr(climate, "_fetch_history_nasa", _nasa_should_not_run)
df, meta = climate._load_history(cell)
assert meta["source"] == "open-meteo"
# --- per-attempt timeouts (the cell-lock finding) ---------------------------
def test_request_applies_a_per_attempt_timeout_sequence(monkeypatch):
"""A timeout sequence (e.g. ARCHIVE_FETCH_TIMEOUTS) is applied per attempt,
not the same value on every retry -- the early attempts fail fast and only
the last gets the longer allowance, bounding how long a caller holding a
lock across every attempt (see _load_history) can be pinned."""
import time as time_mod
seen = []
class FakeResp:
def raise_for_status(self):
pass
class FakeClient:
def get(self, url, params=None, timeout=None, headers=None):
seen.append(timeout)
if len(seen) < 3:
raise RuntimeError("simulated transient failure")
return FakeResp()
monkeypatch.setattr(climate, "_client", FakeClient())
monkeypatch.setattr(time_mod, "sleep", lambda s: None) # skip retry backoff
climate._request("http://example.invalid", {}, (60, 60, 150), phase="test")
assert seen == [60, 60, 150]
def test_request_scalar_timeout_is_unchanged_for_every_attempt(monkeypatch):
"""Every other _request caller passes a single number -- confirm that path is
untouched (same value on every attempt, not just the first)."""
import time as time_mod
seen = []
class FakeResp:
def raise_for_status(self):
pass
class FakeClient:
def get(self, url, params=None, timeout=None, headers=None):
seen.append(timeout)
if len(seen) < 3:
raise RuntimeError("simulated transient failure")
return FakeResp()
monkeypatch.setattr(climate, "_client", FakeClient())
monkeypatch.setattr(time_mod, "sleep", lambda s: None)
climate._request("http://example.invalid", {}, 30, phase="test")
assert seen == [30, 30, 30]
# --- atomic parquet writes (warm_cities.py overlap-run finding) -------------
def test_write_cache_leaves_no_tempfile_behind_on_success(tmp_path):
df = climate._to_frame(_om_daily())
path = str(tmp_path / "cell.parquet")
climate._write_cache(df, path)
import os
assert sorted(os.listdir(tmp_path)) == ["cell.parquet"]
def test_write_cache_cleans_up_tempfile_and_leaves_no_partial_target_on_failure(
monkeypatch, tmp_path):
"""A write failure (disk full, killed process, ...) must not leave a
half-written file at the real path, and must not leak the tempfile either --
a concurrent reader (or an overlapping warm_cities.py run) must only ever see
the old complete file or the new complete file, never a partial one."""
df = climate._to_frame(_om_daily())
path = str(tmp_path / "cell.parquet")
def boom(self, *a, **kw):
raise RuntimeError("disk full")
monkeypatch.setattr(pl.DataFrame, "write_parquet", boom)
with pytest.raises(RuntimeError):
climate._write_cache(df, path)
import os
assert not os.path.exists(path)
assert os.listdir(tmp_path) == []
# --- reverse-geocode off the shared threadpool ------------------------------
def test_reverse_geocode_cache_hit_never_touches_the_worker_queue(monkeypatch):
key = climate.store.revgeo_key(10.0, 20.0)
monkeypatch.setitem(climate._REVGEO_CACHE, key, "Cached Place")
def boom(*a, **kw):
raise AssertionError("a cache hit must not enqueue a worker request")
monkeypatch.setattr(climate._REVGEO_QUEUE, "put", boom)
assert climate.reverse_geocode(10.0, 20.0) == "Cached Place"
def test_reverse_geocode_miss_resolves_via_the_worker_thread(monkeypatch):
"""An uncached lookup is answered by the dedicated worker thread (not the
calling thread), and the result is cached for the next call."""
monkeypatch.setattr(climate, "_fetch_revgeo_label", lambda lat, lon: "Worker Place")
label = climate.reverse_geocode(11.111, 22.222)
assert label == "Worker Place"
found, cached = climate.reverse_geocode_cached(11.111, 22.222)
assert found and cached == "Worker Place"
def test_reverse_geocode_timeout_returns_none_without_blocking_the_caller(monkeypatch):
"""A caller waits only up to _REVGEO_WAIT_TIMEOUT for ITS OWN request, even if
the worker is still busy on it -- it must not block for as long as the fetch
itself takes (that was exactly the old threadpool-pinning problem)."""
import time as time_mod
monkeypatch.setattr(climate, "_REVGEO_WAIT_TIMEOUT", 0.05)
def slow_fetch(lat, lon):
time_mod.sleep(0.3)
return "Too Slow"
monkeypatch.setattr(climate, "_fetch_revgeo_label", slow_fetch)
t0 = time_mod.monotonic()
label = climate.reverse_geocode(33.333, 44.444)
elapsed = time_mod.monotonic() - t0
assert label is None
assert elapsed < 0.2 # returned near the wait timeout, not after the 0.3s fetch