thermograph/tests/data/test_grading.py
Emi Griffith af57b89e4b Split Very Heavy into Very Heavy + Severe; make Dry mean zero rain (#216)
Rain intensity gains a Severe tier and Dry becomes strictly no-rain:

- The top half of Very Heavy (95th–99th rain-day percentile) becomes a new
  "Severe" tier; Very Heavy keeps the 90–95 band. Eight rain tiers now — Trace /
  Light / Brisk / Typical / Heavy / Very Heavy / Severe / Extreme — which refill
  the wet-2..wet-9 colour ramp contiguously (no gap), so Heavy/Very Heavy shift
  one shade lighter and Severe takes the second-darkest.
- A day is Dry only when it didn't rain at all; any measurable rain, however
  slight, is at least Trace. _grade_precip splits on > 0 rather than the 0.01"
  threshold (which still governs the separate dry-streak metric).
- The distribution strip drops the range under the Dry column — every dry day is
  zero, so a "0–0" span was noise.

_precip_ladder derives its percentile marks from RAIN_BANDS now, so adding or
splitting a tier can't leave a hard-coded list behind (that was the bug the 95th
mark would have hit). The detail-view ladder derives from the band table as before.

Claude-Session: https://claude.ai/code/session_013dRZmX9D3JEntfMKWMTWZ8
2026-07-20 05:09:22 +00:00

166 lines
6.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import datetime
import numpy as np
import polars as pl
import pytest
import grading
# ---- empirical percentile ----------------------------------------------------
def test_percentile_mid_rank_handles_ties():
samples = np.array([1.0, 2.0, 2.0, 3.0])
# less=1, equal=2 -> (1 + 0.5*2) / 4 = 50%
assert grading.empirical_percentile(samples, 2.0) == 50.0
def test_percentile_extremes_and_empties():
samples = np.array([1.0, 2.0, 3.0])
assert grading.empirical_percentile(samples, 0.0) == 0.0
assert grading.empirical_percentile(samples, 4.0) == 100.0
assert grading.empirical_percentile(np.array([]), 1.0) is None
assert grading.empirical_percentile(samples, None) is None
assert grading.empirical_percentile(samples, float("nan")) is None
# ---- tier bands ---------------------------------------------------------------
@pytest.mark.parametrize("pct,label,css", [
(99.5, "Near Record", "rec-hot"), # top tier is strict: > 99 only
(99.0, "Very High", "very-hot"), # p99 exactly is NOT near-record
(90.0, "Very High", "very-hot"),
(75.0, "High", "hot"),
(60.0, "Above Normal", "warm"),
(59.9, "Normal", "normal"),
(40.0, "Normal", "normal"),
(39.9, "Below Normal", "cool"),
(10.0, "Low", "cold"),
(1.0, "Very Low", "very-cold"),
(0.5, "Near Record", "rec-cold"), # strictly below the 1st percentile
])
def test_temp_band_boundaries(pct, label, css):
assert grading._band(pct, grading.TEMP_BANDS) == (label, css)
def test_ladders_stay_aligned_with_bands():
"""The detail-view ladders re-encode the band tables by hand; catch drift."""
for bands, ladder in [(grading.TEMP_BANDS, grading._TEMP_LADDER),
(grading.RAIN_BANDS, grading._RAIN_LADDER)]:
assert len(bands) == len(ladder)
for (_, label, css), (lcss, llabel, *_rest) in zip(bands, ladder):
assert (label, css) == (llabel, lcss)
# ---- seasonal window ----------------------------------------------------------
def test_window_mask_wraps_across_year_end():
doys = np.array([1, 180, 360, 366])
mask = grading.window_mask(doys, target_doy=1, half=7)
assert mask.tolist() == [True, False, True, True]
# ---- precip grading -----------------------------------------------------------
def test_only_zero_precip_is_dry():
# Dry means no rain at all; any measurable rain, however slight, is at least a
# Trace day (not Dry).
dry = grading._grade_precip(np.array([0.0, 0.5, 1.0]), 0.0)
assert dry["class"] == "dry" and dry["percentile"] is None and dry["grade"] == "Dry"
trace = grading._grade_precip(np.array([0.0, 0.5, 1.0]), 0.005)
assert trace["class"] != "dry" and trace["grade"] == "Trace"
def test_rain_percentile_ranks_among_rain_days_only():
# 7 dry days + rain days [0.1, 0.2, 0.4]; 0.2 ranks among the 3 rain days:
# less=1, equal=1 -> (1 + 0.5) / 3 = 50% -> Typical, unaffected by the dry mass.
samples = np.array([0.0] * 7 + [0.1, 0.2, 0.4])
g = grading._grade_precip(samples, 0.2)
assert g["percentile"] == 50.0
assert g["grade"] == "Typical"
def test_rain_with_no_historical_rain_days_is_extreme():
g = grading._grade_precip(np.zeros(10), 0.3)
assert g["percentile"] == 100.0 and g["class"] == "wet-9"
def test_rain_scale_eight_tiers_with_severe():
# Eight tiers filling the wet-2..wet-9 ramp: Trace / Light / Brisk / Typical /
# Heavy / Very Heavy / Severe / Extreme. Very Heavy was split, its top half
# becoming Severe (9599).
assert [b[1] for b in grading.RAIN_BANDS] == \
["Extreme", "Severe", "Very Heavy", "Heavy", "Typical", "Brisk", "Light", "Trace"]
assert [b[2] for b in grading.RAIN_BANDS] == \
["wet-9", "wet-8", "wet-7", "wet-6", "wet-5", "wet-4", "wet-3", "wet-2"]
rain = np.arange(1, 201, dtype=float) / 100.0 # 200 rain days: 0.01 .. 2.00
samples = np.concatenate([np.zeros(20), rain])
def grade(v):
g = grading._grade_precip(samples, v)
return g["grade"], g["class"]
assert grade(0.01) == ("Trace", "wet-2") # <1st pct -> the bottom tier
assert grade(1.35) == ("Heavy", "wet-6") # ~67th pct
assert grade(1.85) == ("Very Heavy", "wet-7") # ~92nd pct (lower half)
assert grade(1.96) == ("Severe", "wet-8") # ~98th pct (top half -> Severe)
assert grade(2.00) == ("Extreme", "wet-9") # >99th pct
# ---- dry streaks ---------------------------------------------------------------
def test_dry_streaks_walk():
dates = [datetime.date(2024, 1, 1) + datetime.timedelta(days=i) for i in range(5)]
precips = [0.5, 0.0, float("nan"), 0.02, 0.005]
out = grading.dry_streaks(dates, precips)
assert list(out.values()) == [0, 1, 2, 0, 1] # NaN counts as dry
# ---- range + day grading over a synthetic record --------------------------------
def test_grade_range_compact_shape(history):
end = history["date"].max()
start = end - datetime.timedelta(days=30)
days = grading.grade_range(history, start, end)
assert len(days) == 31
first = days[0]
assert set(first) == {"date", "dsr", *grading.TEMP_METRICS, "precip"}
assert first["dsr"] >= 0
g = first["tmax"]
assert set(g) == {"v", "pct", "c", "g"}
for rec in days: # dry days carry no rain percentile, wet days always do
if rec["precip"]["c"] == "dry":
assert rec["precip"]["pct"] is None
else:
assert rec["precip"]["pct"] is not None
def test_grade_day_normals_and_departure(history):
target = history["date"].max()
row = history.filter(pl.col("date") == target).row(0, named=True)
result = grading.grade_day(history, target, {"tmax": row["tmax"], "tmin": row["tmin"],
"precip": row["precip"]})
assert set(result["normals"]["tmax"]) == {"p1", "p10", "p25", "p40", "p50", "p60",
"p75", "p90", "p99"}
expected = max(abs(result["tmax"]["percentile"] - 50), abs(result["tmin"]["percentile"] - 50))
assert result["departure"] == round(expected, 1)
def test_day_detail_with_and_without_observation(history):
target = history["date"].max()
detail = grading.day_detail(history, target, {"tmax": 60.0, "precip": 0.0})
assert detail["metrics"]["tmax"]["obs"]["value"] == 60.0
assert detail["metrics"]["tmax"]["ladder"]["tiers"][0]["c"] == "rec-hot"
assert detail["metrics"]["precip"]["ladder"]["tiers"][-1]["c"] == "dry"
bare = grading.day_detail(history, target, None)
assert bare["metrics"]["tmax"]["obs"] is None
assert bare["metrics"]["tmax"]["ladder"] is not None
def test_climatology_summary(history):
climo = grading.climatology(history, 180)
# ±7-day window over ~20 years -> ~15 samples per year.
assert climo["n_samples"] >= 15 * 19
assert climo["tmax"]["p40"] <= climo["tmax"]["p50"] <= climo["tmax"]["p60"]
assert climo["feels"] is None # column absent from this record