All checks were successful
secrets-guard / encrypted (push) Successful in 8s
Build + push images (Forgejo registry) / build-push (frontend) (push) Successful in 14s
shell-lint / shellcheck (push) Successful in 19s
Deploy / deploy (frontend) (push) Successful in 24s
Build + push images (Forgejo registry) / build-push (backend) (push) Successful in 43s
Deploy / deploy (backend) (push) Successful in 1m7s
Co-authored-by: admin_emi <admin_emi@noreply.dev.jinemi.com> Co-committed-by: admin_emi <admin_emi@noreply.dev.jinemi.com>
325 lines
15 KiB
Python
325 lines
15 KiB
Python
"""Plain-English grade narratives — a local LLM phrases one graded day.
|
||
|
||
Thermograph's payloads say a day sits in the 92nd percentile and earns the
|
||
"Very High" tier. That is precise and completely inert to read. This module
|
||
turns one graded day into a sentence a person actually absorbs, using a model
|
||
served by Ollama on the operator's desktop.
|
||
|
||
Three properties hold this together, and every one of them is load-bearing:
|
||
|
||
* **Generation never happens on the request path.** Nothing here is imported by
|
||
a route handler that has to answer quickly. ``scripts/generate_narratives.py``
|
||
fills the cache; ``api/narrative_routes.py`` only ever *reads* it. A public
|
||
endpoint therefore has no dependency on the desktop being powered on, the
|
||
mesh being up, or a model being loaded — the worst case is a cache miss, which
|
||
serves ``null`` and renders as nothing.
|
||
|
||
* **The model never sees a number it could misreport.** ``facts_from_grade``
|
||
hands it tier labels and percentile ordinals only — never a temperature, never
|
||
a raw reading. A model that hallucinates cannot invent "38°C" because it was
|
||
never told a degree value, and the sentence it writes is checked against
|
||
:data:`_BANNED` for units it had no business emitting. This is the whole
|
||
reason narratives are unit-free rather than rendered per-country: it removes
|
||
an entire class of wrong output, and as a bonus there is exactly one cached
|
||
narrative per (cell, date) instead of a °C and a °F variant.
|
||
|
||
* **The framing stays relative.** Thermograph grades how *unusual* weather is
|
||
against ~45 years of that exact location's history — it is not a forecast and
|
||
not a thermometer. The prompt says so, and the sanitizer drops output that
|
||
drifts into prediction or advice.
|
||
|
||
Configuration is entirely environmental, so the same code runs on the desktop
|
||
(Ollama on localhost) and on a host that reaches it across WireGuard:
|
||
|
||
* ``THERMOGRAPH_OLLAMA_URL`` — default ``http://127.0.0.1:11434``
|
||
* ``THERMOGRAPH_NARRATIVE_MODEL`` — default ``qwen3:14b``
|
||
* ``THERMOGRAPH_NARRATIVE_TIMEOUT`` — seconds, default 120 (generous on purpose:
|
||
a CPU-only fallback is roughly an order of magnitude slower than the GPU, and
|
||
a batch job would rather wait than lose the row)
|
||
"""
|
||
import datetime
|
||
import math
|
||
import os
|
||
import re
|
||
|
||
import httpx
|
||
|
||
# Bump to invalidate every stored narrative at once — it is half of the validity
|
||
# token (the model name is the other half), so a prompt or house-style change
|
||
# here orphans the old rows rather than silently mixing two generations of text.
|
||
# n2: fact lines state the direction ("warmer than 82% of nights…") instead of a
|
||
# bare "82nd percentile (High)", which models could and did read backwards.
|
||
NARRATIVE_VER = "n2"
|
||
|
||
OLLAMA_URL = os.environ.get("THERMOGRAPH_OLLAMA_URL", "http://127.0.0.1:11434").rstrip("/")
|
||
MODEL = os.environ.get("THERMOGRAPH_NARRATIVE_MODEL", "qwen3:14b").strip()
|
||
TIMEOUT = float(os.environ.get("THERMOGRAPH_NARRATIVE_TIMEOUT", "120"))
|
||
|
||
# Hard ceiling on what we will store. The prompt asks for one sentence; this is
|
||
# the backstop for a model that ignores it. Sentence-truncated, not chopped
|
||
# mid-word (see _sanitize).
|
||
MAX_CHARS = 240
|
||
|
||
# The metrics a narrative may mention, in prompt order: payload key, the label
|
||
# used in the prompt, and the noun for what the percentile ranks against.
|
||
# Anything absent from the graded day is simply omitted — a cell with no
|
||
# feels-like reading yields a shorter fact list, never a "None" in the prompt.
|
||
#
|
||
# The noun matters more than it looks. An earlier version of this prompt wrote
|
||
# "Overnight low: 82nd percentile (High)", which a model read as "the low was
|
||
# high" and phrased as a *cool* night — an exact inversion of the fact, and one
|
||
# no output filter can catch because nothing about the sentence looks wrong.
|
||
# Stating the direction outright ("warmer than 82% of nights…") removes the
|
||
# ambiguity at the source; the tier is then along for corroboration, not
|
||
# interpretation.
|
||
_METRICS = (
|
||
("tmax", "Daytime high", "days"),
|
||
("tmin", "Overnight low", "nights"),
|
||
("feels", "Feels-like", "days"),
|
||
("precip", "Rain", "rainy days"),
|
||
)
|
||
|
||
# Units and forecast vocabulary the model was never given and must not produce.
|
||
# Matched case-insensitively against the finished sentence; a hit voids the
|
||
# generation rather than storing something that reads authoritative and is not.
|
||
_BANNED = re.compile(
|
||
r"(\d+\s*(°|deg|celsius|fahrenheit|c\b|f\b)" # any degree reading
|
||
r"|\b(mm|cm|inches|inch|mph|km/?h|knots?)\b" # any measured quantity
|
||
r"|\b(tomorrow|forecast|expect|will be|should be)\b" # prediction
|
||
r"|\b(wear|bring|advise|recommend|stay hydrated)\b)", # advice
|
||
re.I,
|
||
)
|
||
|
||
# qwen3 is a thinking model. We pass think=False, but a reasoning block leaking
|
||
# into the response is exactly the kind of thing that changes between releases,
|
||
# so it is stripped defensively too.
|
||
_THINK = re.compile(r"<think\b[^>]*>.*?</think\s*>", re.I | re.S)
|
||
|
||
# Models open with a conversational lead-in more often than they don't, and it
|
||
# arrives in two shapes: a bare interjection ("Sure!", "Okay,") and a labelled
|
||
# hand-off ("Here's a sentence:"), frequently both at once. Each is anchored to a
|
||
# known opener rather than to "anything before a colon", so a real narrative that
|
||
# happens to use a colon — "One thing stands out: the night was extraordinary" —
|
||
# survives untouched.
|
||
_LEADIN = re.compile(r"^\s*(?:sure|certainly|of course|okay|ok|absolutely|got it)"
|
||
r"\b[!,.…]*\s*", re.I)
|
||
_LABEL = re.compile(r"^\s*(?:here(?:'|’)?s|here is|sentence|summary|answer)"
|
||
r"\b[^:\n]{0,60}:\s*", re.I)
|
||
|
||
|
||
def _pct_int(pct) -> int:
|
||
"""A percentile as a whole number for the prompt, clamped to 1..99.
|
||
|
||
Same floor(x + 0.5) and same clamp as grading.pct_ordinal, deliberately: a
|
||
narrative saying "warmer than 92%" sitting beside a UI reading "92nd" is the
|
||
least confusing outcome, and the clamp keeps "warmer than 100% of days" —
|
||
which reads as a measurement error — off the page."""
|
||
return min(99, max(1, math.floor(float(pct) + 0.5)))
|
||
|
||
|
||
def narrative_key(target: datetime.date) -> str:
|
||
"""Derived-store key: one narrative per cell per date."""
|
||
return target.isoformat()
|
||
|
||
|
||
def narrative_token(model: str | None = None) -> str:
|
||
"""Validity token. Deliberately independent of the climate sync stamps.
|
||
|
||
The obvious move — folding ``recent_synced_at`` in, the way
|
||
``payloads.recent_token`` does — would invalidate every narrative every time
|
||
the forecast bundle refreshes (~4h). Regenerating the whole grid that often
|
||
is not affordable against a single desktop GPU, let alone the CPU fallback,
|
||
and it buys nothing: a sentence about a day being "cooler than four nights in
|
||
five" does not stop being true because the high moved a tenth of a degree.
|
||
|
||
Freshness is instead the generator's job — it re-runs for the current day on
|
||
its own cadence and overwrites in place (put_narrative upserts). This token
|
||
exists to orphan rows when the *phrasing contract* changes: the prompt/house
|
||
style (NARRATIVE_VER) or the model behind it."""
|
||
return f"{NARRATIVE_VER}:{model or MODEL}"
|
||
|
||
|
||
def _day_row(payload: dict, target: datetime.date) -> dict | None:
|
||
"""The graded row for ``target`` inside a /grade payload's ``recent`` array."""
|
||
want = target.isoformat()
|
||
for row in payload.get("recent") or ():
|
||
if row.get("date") == want:
|
||
return row
|
||
return None
|
||
|
||
|
||
def facts_from_grade(payload: dict, target: datetime.date,
|
||
place: str | None = None) -> dict | None:
|
||
"""Reduce a /grade payload to the handful of facts the model may use.
|
||
|
||
Returns None when the target day carries no gradeable metric at all — there
|
||
is nothing to say about it, and an empty fact list would invite the model to
|
||
invent something. Note what is *not* in the result: no temperatures, no
|
||
cell coordinates, no climatology arrays. See the module docstring.
|
||
|
||
``place`` overrides the payload's own label. The payload carries a
|
||
*reverse-geocoded* name for the ~2-mile cell, which in a dense city is a
|
||
neighbourhood — grading Shanghai yields "Nanjingdonglu Subdistrict", and a
|
||
sentence opening on that reads like it is about somewhere else. A caller that
|
||
already knows which city it asked about should say so."""
|
||
row = _day_row(payload, target)
|
||
if not row:
|
||
return None
|
||
|
||
metrics = []
|
||
for key, label, noun in _METRICS:
|
||
graded = row.get(key)
|
||
if not isinstance(graded, dict):
|
||
continue
|
||
grade = graded.get("grade")
|
||
if not grade:
|
||
continue
|
||
pct = graded.get("percentile")
|
||
# A dry day has a tier ("Dry") but no percentile — it is ranked among rain
|
||
# days only, and it isn't one. Carry pct=None and let build_prompt say so
|
||
# in words rather than emitting a percentage that doesn't exist.
|
||
metrics.append({"label": label, "grade": grade, "noun": noun,
|
||
"pct": _pct_int(pct) if pct is not None else None})
|
||
if not metrics:
|
||
return None
|
||
|
||
if place is None:
|
||
raw = payload.get("place")
|
||
place = (raw or {}).get("label") if isinstance(raw, dict) else raw
|
||
return {"place": place, "date": target.isoformat(), "metrics": metrics}
|
||
|
||
|
||
def build_prompt(facts: dict) -> str:
|
||
"""The full prompt for one narrative. Pure function of ``facts`` — same facts
|
||
in, same prompt out, which is what makes a regeneration reproducible."""
|
||
when = datetime.date.fromisoformat(facts["date"]).strftime("%-d %B")
|
||
lines = [
|
||
"You write one-sentence summaries for Thermograph, a service that grades "
|
||
"how unusual a day's weather is against about 45 years of that exact "
|
||
"location's own history.",
|
||
"",
|
||
"Rules:",
|
||
"- Describe how UNUSUAL the day is for this place at this time of year.",
|
||
"- Use only the facts listed below. Do not add any fact that is not there.",
|
||
"- Never state a temperature, a rainfall amount, or any other measurement. "
|
||
"You have not been given any, and inventing one is a serious error.",
|
||
"- Never forecast, predict, or give advice.",
|
||
"- One sentence, at most 25 words. Plain prose — no markdown, no quotes, "
|
||
"no preamble, no trailing commentary.",
|
||
"",
|
||
"Facts:",
|
||
]
|
||
if facts.get("place"):
|
||
lines.append(f"Place: {facts['place']}")
|
||
lines.append(f"Date: {when}")
|
||
for m in facts["metrics"]:
|
||
if m["pct"] is None:
|
||
# The only percentile-less case in practice is a dry day, whose tier
|
||
# ("Dry") is already a plain-English statement of the fact.
|
||
lines.append(f"{m['label']}: {m['grade'].lower()} — no measurable rain")
|
||
continue
|
||
# "warmer/heavier than N% of <noun> at this time of year here" — the
|
||
# direction is stated, so the tier cannot be read backwards. See _METRICS.
|
||
verb = "heavier" if m["label"] == "Rain" else "warmer"
|
||
lines.append(f"{m['label']}: {verb} than {m['pct']}% of {m['noun']} "
|
||
f"at this time of year here (tier: {m['grade']})")
|
||
lines += ["", "Sentence:"]
|
||
return "\n".join(lines)
|
||
|
||
|
||
def _sanitize(text: str) -> str | None:
|
||
"""Normalize a raw completion, or None if it isn't fit to store.
|
||
|
||
Rejecting is always safe here: the caller stores nothing and the endpoint
|
||
serves null, which renders as an absent sentence rather than a wrong one."""
|
||
if not text:
|
||
return None
|
||
text = _THINK.sub(" ", text)
|
||
text = _LABEL.sub("", _LEADIN.sub("", text))
|
||
text = re.sub(r"\s+", " ", text).strip()
|
||
text = text.strip('"“”‘’\'').strip()
|
||
if len(text) < 20: # too short to be a real sentence
|
||
return None
|
||
if _BANNED.search(text): # a measurement or a forecast it was never given
|
||
return None
|
||
if len(text) > MAX_CHARS:
|
||
# Truncate at the last sentence end that fits; if there isn't one, the
|
||
# model rambled past the limit without punctuation — treat that as a
|
||
# failed generation rather than storing a fragment.
|
||
cut = max(text.rfind(". ", 0, MAX_CHARS), text.rfind("! ", 0, MAX_CHARS),
|
||
text.rfind("? ", 0, MAX_CHARS))
|
||
if cut < 20:
|
||
return None
|
||
text = text[:cut + 1]
|
||
return text
|
||
|
||
|
||
def generate(facts: dict, *, model: str | None = None, client: httpx.Client | None = None) -> str | None:
|
||
"""One narrative for one graded day, or None on any failure.
|
||
|
||
Fail-soft by design and by convention (the derived store's readers behave the
|
||
same way): an unreachable Ollama, a cold model that blows the timeout, a
|
||
refusal, or output that trips the sanitizer all return None. The batch job
|
||
logs it, skips the cell, and moves on — a missing narrative is a non-event,
|
||
a wrong one is not."""
|
||
prompt = build_prompt(facts)
|
||
body = {
|
||
"model": model or MODEL,
|
||
"prompt": prompt,
|
||
"stream": False,
|
||
# Reasoning buys nothing for a one-sentence rephrasing and costs the bulk
|
||
# of the tokens — decisive when the GPU is unavailable and this is running
|
||
# on CPU at single-digit tokens/sec.
|
||
"think": False,
|
||
"options": {
|
||
# Low but not zero: greedy decoding on a small instruct model tends to
|
||
# produce the same stock opener for every cell, which reads worse
|
||
# across a grid than mild variation.
|
||
"temperature": 0.4,
|
||
"top_p": 0.9,
|
||
# ~25 words plus slack; a hard stop is the cheapest defence against a
|
||
# model that decides to write an essay.
|
||
"num_predict": 96,
|
||
"stop": ["\n\n"],
|
||
},
|
||
}
|
||
try:
|
||
if client is not None:
|
||
r = client.post(f"{OLLAMA_URL}/api/generate", json=body, timeout=TIMEOUT)
|
||
else:
|
||
with httpx.Client(timeout=TIMEOUT) as c:
|
||
r = c.post(f"{OLLAMA_URL}/api/generate", json=body)
|
||
r.raise_for_status()
|
||
return _sanitize((r.json() or {}).get("response") or "")
|
||
except Exception: # noqa: BLE001 - see docstring: every failure is a skip
|
||
return None
|
||
|
||
|
||
def health() -> dict:
|
||
"""Whether the configured Ollama can serve the configured model, and where it
|
||
would run. Used by the batch job to fail fast with a useful message instead of
|
||
grinding through a thousand cells that will each time out.
|
||
|
||
``processor`` is the interesting field in practice: Ollama reports "100% CPU"
|
||
when no accelerator is present, which is the difference between a full-grid
|
||
pass taking minutes and taking most of a day."""
|
||
out = {"url": OLLAMA_URL, "model": MODEL, "reachable": False,
|
||
"model_present": False, "processor": None}
|
||
try:
|
||
with httpx.Client(timeout=10.0) as c:
|
||
tags = c.get(f"{OLLAMA_URL}/api/tags").json()
|
||
out["reachable"] = True
|
||
out["model_present"] = any(
|
||
m.get("name") == MODEL or m.get("model") == MODEL
|
||
for m in tags.get("models") or ()
|
||
)
|
||
for m in (c.get(f"{OLLAMA_URL}/api/ps").json() or {}).get("models") or ():
|
||
if m.get("name") == MODEL or m.get("model") == MODEL:
|
||
total = m.get("size") or 0
|
||
gpu = m.get("size_vram") or 0
|
||
out["processor"] = ("100% CPU" if not gpu
|
||
else "100% GPU" if gpu >= total
|
||
else f"{round(100 * gpu / total)}% GPU")
|
||
except Exception: # noqa: BLE001 - health is advisory; callers handle False
|
||
pass
|
||
return out
|