Port e77460e Plain-language weather insight sentences (backend half) from monorepo (drift reconciliation)
The frontend/ half of this commit (calendar.js, calendar.html, app.js, style.css) is out of scope for this repo; a separate agent handles the split frontend repo. Claude-Session: https://claude.ai/code/session_01RdARHDJaYC1wSQRTYM6t3Z
This commit is contained in:
parent
cc0b1e298f
commit
4fe750f10c
6 changed files with 128 additions and 9 deletions
|
|
@ -241,6 +241,9 @@ def build_calendar(cell, history, start_ts, end_ts, months, place, run=None) ->
|
||||||
"range": {"start": start_ts.isoformat(), "end": end_ts.isoformat()},
|
"range": {"start": start_ts.isoformat(), "end": end_ts.isoformat()},
|
||||||
"months": months,
|
"months": months,
|
||||||
"days": days,
|
"days": days,
|
||||||
|
# Per-month climatology over the full history — drives the plain-language
|
||||||
|
# insight strip and cross-month rain comparisons in the calendar UI.
|
||||||
|
"months_climate": grading.month_climate(history),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -68,20 +68,23 @@ DAILY_VARS = (
|
||||||
"temperature_2m_max,temperature_2m_min,precipitation_sum,"
|
"temperature_2m_max,temperature_2m_min,precipitation_sum,"
|
||||||
"wind_speed_10m_max,wind_gusts_10m_max,"
|
"wind_speed_10m_max,wind_gusts_10m_max,"
|
||||||
"apparent_temperature_max,apparent_temperature_min,"
|
"apparent_temperature_max,apparent_temperature_min,"
|
||||||
"relative_humidity_2m_mean"
|
"relative_humidity_2m_mean,"
|
||||||
|
"sunshine_duration,daylight_duration"
|
||||||
)
|
)
|
||||||
|
|
||||||
# Optional metric columns beyond the required date/tmax/tmin/precip schema. A
|
# Optional metric columns beyond the required date/tmax/tmin/precip schema. A
|
||||||
# source that lacks one gets an all-null column in _finalize_frame, so every
|
# source that lacks one gets an all-null column in _finalize_frame, so every
|
||||||
# finalized frame has the same schema without each builder stubbing missing
|
# finalized frame has the same schema without each builder stubbing missing
|
||||||
# columns by hand.
|
# columns by hand. `sun` (the sunshine fraction) only comes from Open-Meteo, so
|
||||||
OPTIONAL_COLS = ("wind", "gust", "humid", "fmax", "fmin")
|
# the NASA/MET fallbacks pick up its null column here too.
|
||||||
|
OPTIONAL_COLS = ("wind", "gust", "humid", "fmax", "fmin", "sun")
|
||||||
|
|
||||||
# Columns added after the original tmax/tmin/precip schema. A cached parquet
|
# Columns added after the original tmax/tmin/precip schema. A cached parquet
|
||||||
# missing any of these predates the wind/feels/humidity features (or the split
|
# missing any of these predates the wind/feels/humidity features (or the split
|
||||||
# apparent high/low) and is refetched so the new metrics show up immediately
|
# apparent high/low, or the sunshine fraction) and is refetched so the new
|
||||||
# instead of after the 30-day cache TTL. `feels` is derived in _finalize_frame;
|
# metrics show up immediately instead of after the 30-day cache TTL. `feels` is
|
||||||
# the rest are the optional raw inputs, so this stays in lockstep with them.
|
# derived in _finalize_frame; the rest are the optional raw inputs, so this stays
|
||||||
|
# in lockstep with them.
|
||||||
NEW_COLS = (*OPTIONAL_COLS, "feels")
|
NEW_COLS = (*OPTIONAL_COLS, "feels")
|
||||||
|
|
||||||
# Thermoneutral baseline (°F). The combined "feels like" metric reports whichever
|
# Thermoneutral baseline (°F). The combined "feels like" metric reports whichever
|
||||||
|
|
@ -433,13 +436,26 @@ def _to_frame(daily: dict) -> pl.DataFrame:
|
||||||
# from the fixed comfort baseline), which the weekly/day views still use.
|
# from the fixed comfort baseline), which the weekly/day views still use.
|
||||||
"fmax": col("apparent_temperature_max"),
|
"fmax": col("apparent_temperature_max"),
|
||||||
"fmin": col("apparent_temperature_min"),
|
"fmin": col("apparent_temperature_min"),
|
||||||
|
# Raw seconds of sunshine and of daylight; combined below into `sun`.
|
||||||
|
"sunsec": col("sunshine_duration"),
|
||||||
|
"daysec": col("daylight_duration"),
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
df = df.with_columns(
|
df = df.with_columns(
|
||||||
pl.col("date").str.to_date(),
|
pl.col("date").str.to_date(),
|
||||||
*[pl.col(c).cast(pl.Float64, strict=False)
|
*[pl.col(c).cast(pl.Float64, strict=False)
|
||||||
for c in ("tmax", "tmin", "precip", "wind", "gust", "fmax", "fmin")],
|
for c in ("tmax", "tmin", "precip", "wind", "gust", "fmax", "fmin",
|
||||||
|
"sunsec", "daysec")],
|
||||||
)
|
)
|
||||||
|
# `sun` = fraction of the day's daylight that was sunshine, clamped to [0, 1]
|
||||||
|
# (sunshine_duration can slightly exceed daylight near the poles). Null when
|
||||||
|
# either series is missing or daylight is zero (polar night).
|
||||||
|
df = df.with_columns(
|
||||||
|
pl.when((pl.col("daysec") > 0) & pl.col("sunsec").is_not_null())
|
||||||
|
.then((pl.col("sunsec") / pl.col("daysec")).clip(0.0, 1.0))
|
||||||
|
.otherwise(None)
|
||||||
|
.alias("sun")
|
||||||
|
).drop("sunsec", "daysec")
|
||||||
return _finalize_frame(df)
|
return _finalize_frame(df)
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -15,6 +15,7 @@ import polars as pl
|
||||||
|
|
||||||
HALF_WINDOW = 7 # +/- 7 days -> a 15-day seasonal window
|
HALF_WINDOW = 7 # +/- 7 days -> a 15-day seasonal window
|
||||||
RAIN_THRESHOLD = 0.01 # inches; a day with < this much precip counts as "dry"
|
RAIN_THRESHOLD = 0.01 # inches; a day with < this much precip counts as "dry"
|
||||||
|
SUN_THRESHOLD = 0.5 # `sun` fraction >= this counts as a "sunny" day (>=50% of daylight)
|
||||||
DSR_LOOKBACK_MIN_DAYS = 14 # min history required before a window to seed dry streaks
|
DSR_LOOKBACK_MIN_DAYS = 14 # min history required before a window to seed dry streaks
|
||||||
|
|
||||||
# Metrics graded on the diverging temperature scale (percentile -> TEMP_BANDS):
|
# Metrics graded on the diverging temperature scale (percentile -> TEMP_BANDS):
|
||||||
|
|
@ -192,6 +193,67 @@ def climatology(df: pl.DataFrame, target_doy: int) -> dict:
|
||||||
"max": round(float(np.max(v)), 2),
|
"max": round(float(np.max(v)), 2),
|
||||||
"mean": round(float(np.mean(v)), 1),
|
"mean": round(float(np.mean(v)), 1),
|
||||||
}
|
}
|
||||||
|
# Plain-language "insight" frequencies over the same window: the share of days
|
||||||
|
# that see measurable rain, and the share that are sunny. Powers the friendly
|
||||||
|
# summary sentences on the home page.
|
||||||
|
out["rain_freq"] = _rain_frac(sub)
|
||||||
|
out["sun_freq"] = _sun_frac(sub)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _rain_frac(df: pl.DataFrame) -> float | None:
|
||||||
|
"""Fraction of days with measurable rain (>= RAIN_THRESHOLD), or None if no
|
||||||
|
usable precip data."""
|
||||||
|
if "precip" not in df.columns:
|
||||||
|
return None
|
||||||
|
v = _finite(df["precip"])
|
||||||
|
if v.size == 0:
|
||||||
|
return None
|
||||||
|
return round(float(np.mean(v >= RAIN_THRESHOLD)), 3)
|
||||||
|
|
||||||
|
|
||||||
|
def _sun_frac(df: pl.DataFrame) -> float | None:
|
||||||
|
"""Fraction of days that are sunny (`sun` fraction >= SUN_THRESHOLD), or None
|
||||||
|
if no usable sunshine data (e.g. NASA/MET-sourced cells)."""
|
||||||
|
if "sun" not in df.columns:
|
||||||
|
return None
|
||||||
|
v = _finite(df["sun"])
|
||||||
|
if v.size == 0:
|
||||||
|
return None
|
||||||
|
return round(float(np.mean(v >= SUN_THRESHOLD)), 3)
|
||||||
|
|
||||||
|
|
||||||
|
# Days per calendar month (non-leap); used to turn a rain fraction into an
|
||||||
|
# average count of rainy days for the month.
|
||||||
|
_DAYS_IN_MONTH = (31, 28, 31, 30, 31, 30, 31, 31, 30, 31, 30, 31)
|
||||||
|
|
||||||
|
|
||||||
|
def month_climate(df: pl.DataFrame) -> dict:
|
||||||
|
"""Per calendar month (1..12), a compact climatological summary across the full
|
||||||
|
history: typical high/low (median), the share of days that see rain / are
|
||||||
|
sunny, and the implied average count of rainy days. Feeds the calendar view's
|
||||||
|
plain-language insight strip and cross-month ("Nx more likely") comparisons."""
|
||||||
|
if "date" not in df.columns:
|
||||||
|
return {}
|
||||||
|
months = df["date"].dt.month().to_numpy()
|
||||||
|
out: dict = {}
|
||||||
|
for m in range(1, 13):
|
||||||
|
sub = df.filter(pl.Series(months == m))
|
||||||
|
if sub.is_empty():
|
||||||
|
continue
|
||||||
|
tmax = _finite(sub["tmax"]) if "tmax" in sub.columns else np.array([])
|
||||||
|
tmin = _finite(sub["tmin"]) if "tmin" in sub.columns else np.array([])
|
||||||
|
rain_freq = _rain_frac(sub)
|
||||||
|
entry = {
|
||||||
|
"n_samples": int(len(sub)),
|
||||||
|
"tmax_med": round(float(np.median(tmax)), 1) if tmax.size else None,
|
||||||
|
"tmin_med": round(float(np.median(tmin)), 1) if tmin.size else None,
|
||||||
|
"rain_freq": rain_freq,
|
||||||
|
"sun_freq": _sun_frac(sub),
|
||||||
|
"rainy_days": (round(rain_freq * _DAYS_IN_MONTH[m - 1], 1)
|
||||||
|
if rain_freq is not None else None),
|
||||||
|
}
|
||||||
|
out[str(m)] = entry
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -20,6 +20,9 @@ def _om_daily(n=3):
|
||||||
"apparent_temperature_max": [82.0, 95.0, 70.0],
|
"apparent_temperature_max": [82.0, 95.0, 70.0],
|
||||||
"apparent_temperature_min": [58.0, 68.0, 50.0],
|
"apparent_temperature_min": [58.0, 68.0, 50.0],
|
||||||
"relative_humidity_2m_mean": [50.0, 60.0, 70.0],
|
"relative_humidity_2m_mean": [50.0, 60.0, 70.0],
|
||||||
|
# Sunshine vs daylight seconds -> `sun` fraction (full sun, half, quarter).
|
||||||
|
"sunshine_duration": [43200.0, 21600.0, 10800.0],
|
||||||
|
"daylight_duration": [43200.0, 43200.0, 43200.0],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -29,7 +32,20 @@ def test_to_frame_schema_and_day_filter():
|
||||||
assert len(df) == 2
|
assert len(df) == 2
|
||||||
assert df["doy"].dtype == pl.Int16
|
assert df["doy"].dtype == pl.Int16
|
||||||
assert set(df.columns) == {"date", "tmax", "tmin", "precip", "wind", "gust",
|
assert set(df.columns) == {"date", "tmax", "tmin", "precip", "wind", "gust",
|
||||||
"humid", "fmax", "fmin", "feels", "doy"}
|
"humid", "fmax", "fmin", "feels", "sun", "doy"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_to_frame_derives_sun_fraction():
|
||||||
|
df = climate._to_frame(_om_daily())
|
||||||
|
# sun = sunshine_duration / daylight_duration, clamped to [0, 1].
|
||||||
|
assert df["sun"].to_list() == [1.0, 0.5]
|
||||||
|
|
||||||
|
|
||||||
|
def test_to_frame_sun_null_without_sunshine():
|
||||||
|
daily = _om_daily()
|
||||||
|
del daily["sunshine_duration"], daily["daylight_duration"]
|
||||||
|
df = climate._to_frame(daily)
|
||||||
|
assert "sun" in df.columns and df["sun"].is_null().all()
|
||||||
|
|
||||||
|
|
||||||
def test_to_frame_tolerates_missing_series():
|
def test_to_frame_tolerates_missing_series():
|
||||||
|
|
|
||||||
|
|
@ -25,7 +25,7 @@ def _hist_frame(n=5, tmax0=70.0):
|
||||||
"""A raw daily frame with the persisted columns (no doy — the write path drops it)."""
|
"""A raw daily frame with the persisted columns (no doy — the write path drops it)."""
|
||||||
d0 = datetime.date(1995, 1, 1)
|
d0 = datetime.date(1995, 1, 1)
|
||||||
cols = {"date": [d0 + datetime.timedelta(days=i) for i in range(n)]}
|
cols = {"date": [d0 + datetime.timedelta(days=i) for i in range(n)]}
|
||||||
for c in ("tmax", "tmin", "precip", "wind", "gust", "humid", "fmax", "fmin", "feels"):
|
for c in ("tmax", "tmin", "precip", "wind", "gust", "humid", "fmax", "fmin", "feels", "sun"):
|
||||||
cols[c] = [float(tmax0 + i) for i in range(n)]
|
cols[c] = [float(tmax0 + i) for i in range(n)]
|
||||||
return pl.DataFrame(cols)
|
return pl.DataFrame(cols)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -164,3 +164,25 @@ def test_climatology_summary(history):
|
||||||
assert climo["n_samples"] >= 15 * 19
|
assert climo["n_samples"] >= 15 * 19
|
||||||
assert climo["tmax"]["p40"] <= climo["tmax"]["p50"] <= climo["tmax"]["p60"]
|
assert climo["tmax"]["p40"] <= climo["tmax"]["p50"] <= climo["tmax"]["p60"]
|
||||||
assert climo["feels"] is None # column absent from this record
|
assert climo["feels"] is None # column absent from this record
|
||||||
|
|
||||||
|
|
||||||
|
def test_climatology_frequencies(history):
|
||||||
|
climo = grading.climatology(history, 180)
|
||||||
|
# ~30% of days carry rain in the fixture; sunshine is absent from this record.
|
||||||
|
assert 0.0 < climo["rain_freq"] < 1.0
|
||||||
|
assert climo["sun_freq"] is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_month_climate(history):
|
||||||
|
mc = grading.month_climate(history)
|
||||||
|
assert sorted(int(k) for k in mc) == list(range(1, 13))
|
||||||
|
for k, e in mc.items():
|
||||||
|
assert 0.0 <= e["rain_freq"] <= 1.0
|
||||||
|
assert e["sun_freq"] is None # no sunshine column in the fixture
|
||||||
|
assert e["tmax_med"] is not None
|
||||||
|
# rainy_days is the rain fraction scaled to the month's length, stored
|
||||||
|
# rounded to one decimal (so allow for that 0.1 rounding).
|
||||||
|
assert e["rainy_days"] == pytest.approx(
|
||||||
|
e["rain_freq"] * grading._DAYS_IN_MONTH[int(k) - 1], abs=0.06)
|
||||||
|
# Summer (Jul) is warmer than winter (Jan) given the seasonal fixture.
|
||||||
|
assert mc["7"]["tmax_med"] > mc["1"]["tmax_med"]
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue