diff --git a/app.py b/app.py
index d0b4092..d29c2df 100644
--- a/app.py
+++ b/app.py
@@ -596,6 +596,11 @@ def _page(name):
proto = request.headers.get("x-forwarded-proto") or request.url.scheme
host = request.headers.get("host") or request.url.netloc
html = html.replace("__ORIGIN__", f"{proto}://{host}{BASE}")
+ # Search-engine verification tags (same source as the SEO pages), so
+ # the homepage Google/Bing verify against carries them too.
+ verify = content.head_verify_html()
+ if verify:
+ html = html.replace("
", f"\n {verify}", 1)
etag = f'W/"{hashlib.sha1(html.encode()).hexdigest()[:20]}"'
if _not_modified(request, etag):
return Response(status_code=304, headers={"ETag": etag})
diff --git a/content.py b/content.py
index b010014..a555c82 100644
--- a/content.py
+++ b/content.py
@@ -6,6 +6,7 @@ with Jinja2 from the same builders the API uses, linking into the interactive to
Registered on the app (via register()) BEFORE the StaticFiles mount so they win.
"""
import datetime
+import functools
import hashlib
import json
import os
@@ -169,34 +170,67 @@ def robots_txt(request: Request) -> Response:
return PlainTextResponse(body)
+def _sitemap_entries() -> list[tuple[str, str, str]]:
+ """Every indexable page as (path, changefreq, priority). Paths are relative to
+ BASE. Shared by the sitemap and IndexNow so the two never drift."""
+ entries = [("/", "daily", "1.0")]
+ for p in ("/climate", "/about", "/glossary", "/calendar", "/compare", "/legend"):
+ entries.append((p, "weekly", "0.6"))
+ for slug in cities.all_slugs():
+ entries.append((f"/climate/{slug}", "daily", "0.8"))
+ entries.append((f"/climate/{slug}/records", "monthly", "0.5"))
+ for m in MONTHS:
+ entries.append((f"/climate/{slug}/{m}", "monthly", "0.5"))
+ return entries
+
+
+def public_paths() -> list[str]:
+ """BASE-relative paths of every indexable page — for IndexNow submission."""
+ return [p for p, _, _ in _sitemap_entries()]
+
+
+@functools.lru_cache(maxsize=1)
+def _content_lastmod() -> str:
+ """A stable 'content last built' date for — the newest mtime of the
+ city list and this module (both change on a content/code rebuild, and the
+ process restarts on deploy). Honest and stable, unlike a per-request today()
+ that churns every fetch and trains crawlers to ignore lastmod entirely."""
+ paths = [os.path.join(os.path.dirname(__file__), "cities.json"), __file__]
+ mtimes = [os.path.getmtime(p) for p in paths if os.path.exists(p)]
+ day = datetime.date.fromtimestamp(max(mtimes)) if mtimes else datetime.date.today()
+ return day.isoformat()
+
+
def sitemap_xml(request: Request) -> Response:
base_url = f"{origin(request)}{BASE}"
- today = datetime.date.today().isoformat()
- urls: list[tuple[str, str, str]] = [] # (loc, changefreq, priority)
-
- def add(path: str, changefreq: str, priority: str) -> None:
- urls.append((f"{base_url}{path}", changefreq, priority))
-
- add("/", "daily", "1.0")
- for p in ("/climate", "/about", "/glossary", "/calendar", "/compare", "/legend"):
- add(p, "weekly", "0.6")
- for slug in cities.all_slugs():
- add(f"/climate/{slug}", "daily", "0.8")
- add(f"/climate/{slug}/records", "monthly", "0.5")
- for m in MONTHS:
- add(f"/climate/{slug}/{m}", "monthly", "0.5")
-
+ lastmod = _content_lastmod()
parts = ['',
'']
- for loc, cf, pr in urls:
+ for path, cf, pr in _sitemap_entries():
parts.append(
- f"{loc}{today}"
+ f"{base_url}{path}{lastmod}"
f"{cf}{pr}"
)
parts.append("")
return Response("\n".join(parts), media_type="application/xml")
+def head_verify_html() -> Markup:
+ """Search-engine ownership-verification tags, from env (empty when
+ unset). Injected into every page's — Google verifies the homepage."""
+ metas = []
+ google = os.environ.get("THERMOGRAPH_GOOGLE_VERIFY", "").strip()
+ bing = os.environ.get("THERMOGRAPH_BING_VERIFY", "").strip()
+ if google:
+ metas.append(Markup('').format(google))
+ if bing:
+ metas.append(Markup('').format(bing))
+ return Markup("\n ").join(metas)
+
+
+_env.globals["head_verify"] = head_verify_html
+
+
# --- per-city climate pages ---------------------------------------------------
def _history_for(cell: dict):
"""Cached archive for a cell, fetching once if missing (self-heals like the
@@ -697,6 +731,19 @@ def register(app) -> None:
"""Attach all content routes. Call from app.py BEFORE the StaticFiles mount."""
app.add_api_route(f"{BASE}/robots.txt", robots_txt, methods=["GET"], include_in_schema=False)
app.add_api_route(f"{BASE}/sitemap.xml", sitemap_xml, methods=["GET"], include_in_schema=False)
+ # IndexNow ownership key, served as a text file at the site root (/{key}.txt)
+ # so Bing/DuckDuckGo/Yandex can verify our submissions. The key is fixed at
+ # startup, so registering its literal path here is safe.
+ import indexnow
+ _inkey = indexnow.key()
+
+ def _indexnow_key_file() -> Response:
+ return PlainTextResponse(_inkey + "\n")
+
+ app.add_api_route(
+ f"{BASE}/{_inkey}.txt", _indexnow_key_file,
+ methods=["GET", "HEAD"], include_in_schema=False,
+ )
app.add_api_route(f"{BASE}/about", about_page, methods=["GET", "HEAD"], include_in_schema=False)
app.add_api_route(f"{BASE}/glossary", glossary_index, methods=["GET", "HEAD"], include_in_schema=False)
app.add_api_route(f"{BASE}/glossary/{{term}}", glossary_term, methods=["GET", "HEAD"], include_in_schema=False)
diff --git a/indexnow.py b/indexnow.py
new file mode 100644
index 0000000..7f2d6e9
--- /dev/null
+++ b/indexnow.py
@@ -0,0 +1,116 @@
+"""IndexNow — instantly notify Bing, DuckDuckGo, Yandex, Seznam and friends of new
+or changed URLs so they (re)crawl within minutes instead of days.
+
+Note: Google does NOT use IndexNow. For Google, submit the sitemap in Search
+Console; this module covers everyone else (and DuckDuckGo, which is powered by
+Bing's index).
+
+A per-host key authorises submissions: it is served as a plain-text file at the
+site root (``/{key}.txt``, wired up in content.register) and echoed in every
+submission so the search engine can confirm we own the host. Key resolution order
+mirrors push.py's VAPID handling: env var → gitignored file → generated once and
+persisted.
+
+Usage (run after a content rebuild, or once to seed a fresh site):
+
+ cd backend && python indexnow.py https://thermograph.org
+ # or: make indexnow URL=https://thermograph.org
+"""
+import logging
+import os
+import secrets
+import threading
+from urllib.parse import urlparse
+
+import httpx
+
+log = logging.getLogger("thermograph.indexnow")
+
+_DATA_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "data"))
+_KEY_PATH = os.environ.get("THERMOGRAPH_INDEXNOW_KEY_FILE") or os.path.join(_DATA_DIR, "indexnow_key.txt")
+_ENDPOINT = "https://api.indexnow.org/indexnow" # a shared endpoint; it fans out to all participants
+_BATCH = 10000 # IndexNow's max URLs per request
+
+_lock = threading.Lock()
+_key = None
+
+
+def key() -> str:
+ """The IndexNow key (env → file → generate-and-persist), cached for the process."""
+ global _key
+ if _key:
+ return _key
+ with _lock:
+ if _key:
+ return _key
+ env = os.environ.get("THERMOGRAPH_INDEXNOW_KEY", "").strip()
+ if env:
+ _key = env
+ return _key
+ try:
+ with open(_KEY_PATH, encoding="utf-8") as f:
+ k = f.read().strip()
+ if k:
+ _key = k
+ return _key
+ except OSError:
+ pass
+ _key = secrets.token_hex(16) # 32 hex chars — within IndexNow's 8–128 range
+ try:
+ os.makedirs(_DATA_DIR, exist_ok=True)
+ with open(_KEY_PATH, "w", encoding="utf-8") as f:
+ f.write(_key)
+ os.chmod(_KEY_PATH, 0o600)
+ except OSError:
+ log.warning("could not persist IndexNow key to %s; using an in-memory key", _KEY_PATH)
+ return _key
+
+
+def submit(urls, host: str, key_location: str, scheme: str = "https", timeout: float = 30) -> dict:
+ """POST ``urls`` (deduped, in batches of 10k) to IndexNow. Every URL must be on
+ ``host``; ``key_location`` is the public URL of our key file. Returns a summary."""
+ urls = list(dict.fromkeys(urls))
+ if not urls:
+ return {"submitted": 0, "total": 0, "batches": 0}
+ k = key()
+ submitted, statuses = 0, []
+ for i in range(0, len(urls), _BATCH):
+ batch = urls[i:i + _BATCH]
+ payload = {"host": host, "key": k, "keyLocation": key_location, "urlList": batch}
+ try:
+ r = httpx.post(_ENDPOINT, json=payload, timeout=timeout,
+ headers={"Content-Type": "application/json; charset=utf-8"})
+ except httpx.HTTPError as e:
+ log.warning("IndexNow request failed: %s", e)
+ statuses.append("error")
+ continue
+ statuses.append(r.status_code)
+ if r.status_code in (200, 202):
+ submitted += len(batch)
+ else:
+ log.warning("IndexNow batch rejected: %s %s", r.status_code, r.text[:200])
+ return {"submitted": submitted, "total": len(urls),
+ "batches": len(statuses), "statuses": statuses, "key_location": key_location}
+
+
+def submit_all(site_base_url: str, **kw) -> dict:
+ """Submit every indexable page. ``site_base_url`` is the public origin + base
+ path, e.g. ``https://thermograph.org`` (root) or ``https://host/thermograph``."""
+ import content # lazy: avoids a content ↔ indexnow import cycle
+ site = site_base_url.rstrip("/")
+ parsed = urlparse(site)
+ urls = [site + path for path in content.public_paths()]
+ return submit(urls, host=parsed.hostname, scheme=parsed.scheme or "https",
+ key_location=f"{site}/{key()}.txt", **kw)
+
+
+if __name__ == "__main__":
+ import sys
+
+ arg = sys.argv[1].strip() if len(sys.argv) > 1 else ""
+ base = arg or os.environ.get("THERMOGRAPH_BASE_URL", "https://thermograph.org")
+ logging.basicConfig(level=logging.INFO)
+ print(f"IndexNow key: {key()} (served at {base.rstrip('/')}/{key()}.txt)")
+ result = submit_all(base)
+ print(f"Submitted {result['submitted']}/{result['total']} URLs "
+ f"in {result['batches']} batch(es); statuses={result['statuses']}")
diff --git a/templates/base.html.j2 b/templates/base.html.j2
index e7a5615..ae65e09 100644
--- a/templates/base.html.j2
+++ b/templates/base.html.j2
@@ -3,6 +3,7 @@
+ {{ head_verify() }}
{% block title %}Thermograph{% endblock %}
diff --git a/tests/conftest.py b/tests/conftest.py
index 01acaae..f92a4cf 100644
--- a/tests/conftest.py
+++ b/tests/conftest.py
@@ -28,6 +28,8 @@ _TMP = tempfile.mkdtemp(prefix="thermograph-tests-")
# start the background notifier thread during tests. Set before db is imported.
os.environ.setdefault("THERMOGRAPH_ACCOUNTS_DB", os.path.join(_TMP, "accounts.sqlite"))
os.environ.setdefault("THERMOGRAPH_ENABLE_NOTIFIER", "0")
+# Pin the IndexNow key so it isn't generated + written into the repo's data/ dir.
+os.environ.setdefault("THERMOGRAPH_INDEXNOW_KEY", "test0000indexnow0000key000000000")
import store # noqa: E402
diff --git a/tests/test_content.py b/tests/test_content.py
index e46d441..a2994ae 100644
--- a/tests/test_content.py
+++ b/tests/test_content.py
@@ -191,3 +191,51 @@ def test_seo_pages_load_unit_and_account_scripts(client):
b = client.get(f"{B}/climate/{SLUG}").text
for src in (f"{B}/units.js", f"{B}/account.js", f"{B}/climate.js"):
assert f'src="{src}"' in b
+
+
+def test_indexnow_key_file_served(client):
+ import indexnow
+ k = indexnow.key()
+ r = client.get(f"{B}/{k}.txt")
+ assert r.status_code == 200
+ assert r.text.strip() == k
+ assert r.headers["content-type"].startswith("text/plain")
+
+
+def test_sitemap_lastmod_is_stable(client):
+ import datetime
+ import re
+ body = client.get(f"{B}/sitemap.xml").text
+ mods = set(re.findall(r"([^<]+)", body))
+ assert len(mods) == 1 # one build date, not a per-request churn
+ datetime.date.fromisoformat(next(iter(mods))) # a valid ISO date
+
+
+def test_indexnow_submit_all_builds_payload(monkeypatch):
+ import indexnow
+
+ calls = []
+
+ class _Resp:
+ status_code = 200
+ text = "ok"
+
+ monkeypatch.setattr(indexnow.httpx, "post",
+ lambda url, json=None, **kw: (calls.append((url, json)), _Resp())[1])
+ res = indexnow.submit_all("https://thermograph.org")
+ assert res["submitted"] == res["total"] > 100
+ first = calls[0][1]
+ assert first["host"] == "thermograph.org"
+ assert first["keyLocation"] == f"https://thermograph.org/{indexnow.key()}.txt"
+ assert first["urlList"][0] == "https://thermograph.org/"
+ assert all(len(c[1]["urlList"]) <= 10000 for c in calls) # batched under the IndexNow cap
+
+
+def test_search_verification_meta(client, monkeypatch):
+ monkeypatch.setenv("THERMOGRAPH_GOOGLE_VERIFY", "gtok123")
+ monkeypatch.setenv("THERMOGRAPH_BING_VERIFY", "btok456")
+ seo = client.get(f"{B}/climate/{SLUG}").text # Jinja SEO page
+ home = client.get(f"{B}/").text # static homepage via _page()
+ for b in (seo, home):
+ assert '' in b
+ assert '' in b