Faster search-engine indexing for the ~14k climate/city/record URLs:
- IndexNow (backend/indexnow.py): instantly notify Bing/DuckDuckGo/Yandex of new
or changed URLs. Per-host key resolved env → gitignored file → generated (mirrors
push.py), served at /{key}.txt, and echoed in submissions. `submit_all()` + a
`make indexnow` CLI push every indexable URL, batched under the 10k cap. Google
doesn't use IndexNow, so it stays on the sitemap.
- Sitemap: replace the per-request today() <lastmod> (which churns every fetch and
trains crawlers to ignore lastmod) with a stable content-build date; factor the
URL list into public_paths() shared with IndexNow so the two never drift.
- Search-console verification: render google-site-verification + msvalidate.01
<meta> tags from env into every page's <head> — the SEO pages via base.html.j2
and the static pages (incl. the homepage Google verifies) via _page().
- Docs: env example documents the three new vars; .gitignore ignores the key.
Tests: key-file route, stable single-date lastmod, IndexNow payload/batching, and
verification meta on both SEO and static pages.
116 lines
4.4 KiB
Python
116 lines
4.4 KiB
Python
"""Shared test setup: import path, offline guards, and synthetic weather data.
|
|
|
|
Everything here keeps the suite hermetic — no Open-Meteo, no Nominatim, no
|
|
GeoNames download, and no writes into the repo's data/ or logs/ folders.
|
|
"""
|
|
import datetime
|
|
import os
|
|
import sys
|
|
import tempfile
|
|
import threading
|
|
|
|
import numpy as np
|
|
import polars as pl
|
|
import pytest
|
|
|
|
BACKEND = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
sys.path.insert(0, BACKEND)
|
|
|
|
import places # noqa: E402
|
|
|
|
# app.py calls places.start_loading() at import; pretend it already ran so no
|
|
# background GeoNames download starts. Tests build their own tiny index.
|
|
places._load_started = True
|
|
|
|
_TMP = tempfile.mkdtemp(prefix="thermograph-tests-")
|
|
|
|
# Keep the authoritative accounts DB out of the repo's data/ folder, and don't
|
|
# start the background notifier thread during tests. Set before db is imported.
|
|
os.environ.setdefault("THERMOGRAPH_ACCOUNTS_DB", os.path.join(_TMP, "accounts.sqlite"))
|
|
os.environ.setdefault("THERMOGRAPH_ENABLE_NOTIFIER", "0")
|
|
# Pin the IndexNow key so it isn't generated + written into the repo's data/ dir.
|
|
os.environ.setdefault("THERMOGRAPH_INDEXNOW_KEY", "test0000indexnow0000key000000000")
|
|
|
|
import store # noqa: E402
|
|
|
|
# Keep derived-store writes out of the repo's data/ folder. Individual store
|
|
# tests re-point this per test; everything else shares one throwaway DB.
|
|
store.DB_PATH = os.path.join(_TMP, "store.sqlite")
|
|
|
|
import audit # noqa: E402
|
|
|
|
audit.AUDIT_DIR = os.path.join(_TMP, "logs", "audit")
|
|
audit.ERROR_DIR = os.path.join(_TMP, "logs", "errors")
|
|
|
|
|
|
def _years_before(d: datetime.date, years: int) -> datetime.date:
|
|
"""`d` shifted back `years` years, same month/day (Feb 29 → Feb 28)."""
|
|
try:
|
|
return d.replace(year=d.year - years)
|
|
except ValueError:
|
|
return d.replace(year=d.year - years, day=28)
|
|
|
|
|
|
def _daily(start: datetime.date, end: datetime.date) -> list[datetime.date]:
|
|
"""Inclusive daily date list from `start` to `end`."""
|
|
return [start + datetime.timedelta(days=i) for i in range((end - start).days + 1)]
|
|
|
|
|
|
def _with_doy(df: pl.DataFrame) -> pl.DataFrame:
|
|
return df.with_columns(pl.col("date").dt.ordinal_day().cast(pl.Int16).alias("doy"))
|
|
|
|
|
|
def make_history(years: int = 20, end: str | None = None, seed: int = 7) -> pl.DataFrame:
|
|
"""A plausible daily record (tmax/tmin/precip + doy), matching the columns
|
|
and dtypes climate.get_history returns for an older cache."""
|
|
end_ts = (datetime.date.fromisoformat(end) if end
|
|
else datetime.date.today() - datetime.timedelta(days=6))
|
|
dates = _daily(_years_before(end_ts, years), end_ts)
|
|
rng = np.random.default_rng(seed)
|
|
doy = np.array([d.timetuple().tm_yday for d in dates])
|
|
seasonal = 55 + 30 * np.sin((doy - 100) / 366.0 * 2 * np.pi)
|
|
tmax = seasonal + rng.normal(0, 8, len(dates))
|
|
tmin = tmax - 15 + rng.normal(0, 3, len(dates))
|
|
precip = np.where(rng.random(len(dates)) < 0.3, rng.gamma(1.5, 0.2, len(dates)), 0.0)
|
|
return _with_doy(pl.DataFrame({
|
|
"date": dates,
|
|
"tmax": np.round(tmax, 1),
|
|
"tmin": np.round(tmin, 1),
|
|
"precip": np.round(precip, 2),
|
|
}))
|
|
|
|
|
|
def make_recent(history: pl.DataFrame, future_days: int = 7, seed: int = 11) -> pl.DataFrame:
|
|
"""A recent+forecast bundle: from a couple of weeks before the archive's end
|
|
through `future_days` past today — the shape climate.get_recent_forecast returns."""
|
|
today = datetime.date.today()
|
|
start = history["date"].max() - datetime.timedelta(days=14)
|
|
dates = _daily(start, today + datetime.timedelta(days=future_days))
|
|
rng = np.random.default_rng(seed)
|
|
doy = np.array([d.timetuple().tm_yday for d in dates])
|
|
seasonal = 55 + 30 * np.sin((doy - 100) / 366.0 * 2 * np.pi)
|
|
tmax = seasonal + rng.normal(0, 8, len(dates))
|
|
return _with_doy(pl.DataFrame({
|
|
"date": dates,
|
|
"tmax": np.round(tmax, 1),
|
|
"tmin": np.round(tmax - 15, 1),
|
|
"precip": np.where(rng.random(len(dates)) < 0.3, 0.15, 0.0),
|
|
}))
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def history():
|
|
return make_history()
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def recent(history):
|
|
return make_recent(history)
|
|
|
|
|
|
@pytest.fixture
|
|
def tmp_store(tmp_path, monkeypatch):
|
|
"""store.py against a fresh database file, isolated per test."""
|
|
monkeypatch.setattr(store, "DB_PATH", str(tmp_path / "store.sqlite"))
|
|
monkeypatch.setattr(store, "_local", threading.local())
|
|
return store
|