From 23615a1085fdf063fc4b5fd8f707a0ec9e4352f2 Mon Sep 17 00:00:00 2001 From: Emi Griffith Date: Thu, 16 Jul 2026 11:08:09 -0700 Subject: [PATCH] Add IndexNow, stable sitemap lastmod, and search-verification meta (#127) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Faster search-engine indexing for the ~14k climate/city/record URLs: - IndexNow (backend/indexnow.py): instantly notify Bing/DuckDuckGo/Yandex of new or changed URLs. Per-host key resolved env → gitignored file → generated (mirrors push.py), served at /{key}.txt, and echoed in submissions. `submit_all()` + a `make indexnow` CLI push every indexable URL, batched under the 10k cap. Google doesn't use IndexNow, so it stays on the sitemap. - Sitemap: replace the per-request today() (which churns every fetch and trains crawlers to ignore lastmod) with a stable content-build date; factor the URL list into public_paths() shared with IndexNow so the two never drift. - Search-console verification: render google-site-verification + msvalidate.01 tags from env into every page's — the SEO pages via base.html.j2 and the static pages (incl. the homepage Google verifies) via _page(). - Docs: env example documents the three new vars; .gitignore ignores the key. Tests: key-file route, stable single-date lastmod, IndexNow payload/batching, and verification meta on both SEO and static pages. --- app.py | 5 ++ content.py | 81 ++++++++++++++++++++++------ indexnow.py | 116 +++++++++++++++++++++++++++++++++++++++++ templates/base.html.j2 | 1 + tests/conftest.py | 2 + tests/test_content.py | 48 +++++++++++++++++ 6 files changed, 236 insertions(+), 17 deletions(-) create mode 100644 indexnow.py diff --git a/app.py b/app.py index d0b4092..d29c2df 100644 --- a/app.py +++ b/app.py @@ -596,6 +596,11 @@ def _page(name): proto = request.headers.get("x-forwarded-proto") or request.url.scheme host = request.headers.get("host") or request.url.netloc html = html.replace("__ORIGIN__", f"{proto}://{host}{BASE}") + # Search-engine verification tags (same source as the SEO pages), so + # the homepage Google/Bing verify against carries them too. + verify = content.head_verify_html() + if verify: + html = html.replace("", f"\n {verify}", 1) etag = f'W/"{hashlib.sha1(html.encode()).hexdigest()[:20]}"' if _not_modified(request, etag): return Response(status_code=304, headers={"ETag": etag}) diff --git a/content.py b/content.py index b010014..a555c82 100644 --- a/content.py +++ b/content.py @@ -6,6 +6,7 @@ with Jinja2 from the same builders the API uses, linking into the interactive to Registered on the app (via register()) BEFORE the StaticFiles mount so they win. """ import datetime +import functools import hashlib import json import os @@ -169,34 +170,67 @@ def robots_txt(request: Request) -> Response: return PlainTextResponse(body) +def _sitemap_entries() -> list[tuple[str, str, str]]: + """Every indexable page as (path, changefreq, priority). Paths are relative to + BASE. Shared by the sitemap and IndexNow so the two never drift.""" + entries = [("/", "daily", "1.0")] + for p in ("/climate", "/about", "/glossary", "/calendar", "/compare", "/legend"): + entries.append((p, "weekly", "0.6")) + for slug in cities.all_slugs(): + entries.append((f"/climate/{slug}", "daily", "0.8")) + entries.append((f"/climate/{slug}/records", "monthly", "0.5")) + for m in MONTHS: + entries.append((f"/climate/{slug}/{m}", "monthly", "0.5")) + return entries + + +def public_paths() -> list[str]: + """BASE-relative paths of every indexable page — for IndexNow submission.""" + return [p for p, _, _ in _sitemap_entries()] + + +@functools.lru_cache(maxsize=1) +def _content_lastmod() -> str: + """A stable 'content last built' date for — the newest mtime of the + city list and this module (both change on a content/code rebuild, and the + process restarts on deploy). Honest and stable, unlike a per-request today() + that churns every fetch and trains crawlers to ignore lastmod entirely.""" + paths = [os.path.join(os.path.dirname(__file__), "cities.json"), __file__] + mtimes = [os.path.getmtime(p) for p in paths if os.path.exists(p)] + day = datetime.date.fromtimestamp(max(mtimes)) if mtimes else datetime.date.today() + return day.isoformat() + + def sitemap_xml(request: Request) -> Response: base_url = f"{origin(request)}{BASE}" - today = datetime.date.today().isoformat() - urls: list[tuple[str, str, str]] = [] # (loc, changefreq, priority) - - def add(path: str, changefreq: str, priority: str) -> None: - urls.append((f"{base_url}{path}", changefreq, priority)) - - add("/", "daily", "1.0") - for p in ("/climate", "/about", "/glossary", "/calendar", "/compare", "/legend"): - add(p, "weekly", "0.6") - for slug in cities.all_slugs(): - add(f"/climate/{slug}", "daily", "0.8") - add(f"/climate/{slug}/records", "monthly", "0.5") - for m in MONTHS: - add(f"/climate/{slug}/{m}", "monthly", "0.5") - + lastmod = _content_lastmod() parts = ['', ''] - for loc, cf, pr in urls: + for path, cf, pr in _sitemap_entries(): parts.append( - f"{loc}{today}" + f"{base_url}{path}{lastmod}" f"{cf}{pr}" ) parts.append("") return Response("\n".join(parts), media_type="application/xml") +def head_verify_html() -> Markup: + """Search-engine ownership-verification tags, from env (empty when + unset). Injected into every page's — Google verifies the homepage.""" + metas = [] + google = os.environ.get("THERMOGRAPH_GOOGLE_VERIFY", "").strip() + bing = os.environ.get("THERMOGRAPH_BING_VERIFY", "").strip() + if google: + metas.append(Markup('').format(google)) + if bing: + metas.append(Markup('').format(bing)) + return Markup("\n ").join(metas) + + +_env.globals["head_verify"] = head_verify_html + + # --- per-city climate pages --------------------------------------------------- def _history_for(cell: dict): """Cached archive for a cell, fetching once if missing (self-heals like the @@ -697,6 +731,19 @@ def register(app) -> None: """Attach all content routes. Call from app.py BEFORE the StaticFiles mount.""" app.add_api_route(f"{BASE}/robots.txt", robots_txt, methods=["GET"], include_in_schema=False) app.add_api_route(f"{BASE}/sitemap.xml", sitemap_xml, methods=["GET"], include_in_schema=False) + # IndexNow ownership key, served as a text file at the site root (/{key}.txt) + # so Bing/DuckDuckGo/Yandex can verify our submissions. The key is fixed at + # startup, so registering its literal path here is safe. + import indexnow + _inkey = indexnow.key() + + def _indexnow_key_file() -> Response: + return PlainTextResponse(_inkey + "\n") + + app.add_api_route( + f"{BASE}/{_inkey}.txt", _indexnow_key_file, + methods=["GET", "HEAD"], include_in_schema=False, + ) app.add_api_route(f"{BASE}/about", about_page, methods=["GET", "HEAD"], include_in_schema=False) app.add_api_route(f"{BASE}/glossary", glossary_index, methods=["GET", "HEAD"], include_in_schema=False) app.add_api_route(f"{BASE}/glossary/{{term}}", glossary_term, methods=["GET", "HEAD"], include_in_schema=False) diff --git a/indexnow.py b/indexnow.py new file mode 100644 index 0000000..7f2d6e9 --- /dev/null +++ b/indexnow.py @@ -0,0 +1,116 @@ +"""IndexNow — instantly notify Bing, DuckDuckGo, Yandex, Seznam and friends of new +or changed URLs so they (re)crawl within minutes instead of days. + +Note: Google does NOT use IndexNow. For Google, submit the sitemap in Search +Console; this module covers everyone else (and DuckDuckGo, which is powered by +Bing's index). + +A per-host key authorises submissions: it is served as a plain-text file at the +site root (``/{key}.txt``, wired up in content.register) and echoed in every +submission so the search engine can confirm we own the host. Key resolution order +mirrors push.py's VAPID handling: env var → gitignored file → generated once and +persisted. + +Usage (run after a content rebuild, or once to seed a fresh site): + + cd backend && python indexnow.py https://thermograph.org + # or: make indexnow URL=https://thermograph.org +""" +import logging +import os +import secrets +import threading +from urllib.parse import urlparse + +import httpx + +log = logging.getLogger("thermograph.indexnow") + +_DATA_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "data")) +_KEY_PATH = os.environ.get("THERMOGRAPH_INDEXNOW_KEY_FILE") or os.path.join(_DATA_DIR, "indexnow_key.txt") +_ENDPOINT = "https://api.indexnow.org/indexnow" # a shared endpoint; it fans out to all participants +_BATCH = 10000 # IndexNow's max URLs per request + +_lock = threading.Lock() +_key = None + + +def key() -> str: + """The IndexNow key (env → file → generate-and-persist), cached for the process.""" + global _key + if _key: + return _key + with _lock: + if _key: + return _key + env = os.environ.get("THERMOGRAPH_INDEXNOW_KEY", "").strip() + if env: + _key = env + return _key + try: + with open(_KEY_PATH, encoding="utf-8") as f: + k = f.read().strip() + if k: + _key = k + return _key + except OSError: + pass + _key = secrets.token_hex(16) # 32 hex chars — within IndexNow's 8–128 range + try: + os.makedirs(_DATA_DIR, exist_ok=True) + with open(_KEY_PATH, "w", encoding="utf-8") as f: + f.write(_key) + os.chmod(_KEY_PATH, 0o600) + except OSError: + log.warning("could not persist IndexNow key to %s; using an in-memory key", _KEY_PATH) + return _key + + +def submit(urls, host: str, key_location: str, scheme: str = "https", timeout: float = 30) -> dict: + """POST ``urls`` (deduped, in batches of 10k) to IndexNow. Every URL must be on + ``host``; ``key_location`` is the public URL of our key file. Returns a summary.""" + urls = list(dict.fromkeys(urls)) + if not urls: + return {"submitted": 0, "total": 0, "batches": 0} + k = key() + submitted, statuses = 0, [] + for i in range(0, len(urls), _BATCH): + batch = urls[i:i + _BATCH] + payload = {"host": host, "key": k, "keyLocation": key_location, "urlList": batch} + try: + r = httpx.post(_ENDPOINT, json=payload, timeout=timeout, + headers={"Content-Type": "application/json; charset=utf-8"}) + except httpx.HTTPError as e: + log.warning("IndexNow request failed: %s", e) + statuses.append("error") + continue + statuses.append(r.status_code) + if r.status_code in (200, 202): + submitted += len(batch) + else: + log.warning("IndexNow batch rejected: %s %s", r.status_code, r.text[:200]) + return {"submitted": submitted, "total": len(urls), + "batches": len(statuses), "statuses": statuses, "key_location": key_location} + + +def submit_all(site_base_url: str, **kw) -> dict: + """Submit every indexable page. ``site_base_url`` is the public origin + base + path, e.g. ``https://thermograph.org`` (root) or ``https://host/thermograph``.""" + import content # lazy: avoids a content ↔ indexnow import cycle + site = site_base_url.rstrip("/") + parsed = urlparse(site) + urls = [site + path for path in content.public_paths()] + return submit(urls, host=parsed.hostname, scheme=parsed.scheme or "https", + key_location=f"{site}/{key()}.txt", **kw) + + +if __name__ == "__main__": + import sys + + arg = sys.argv[1].strip() if len(sys.argv) > 1 else "" + base = arg or os.environ.get("THERMOGRAPH_BASE_URL", "https://thermograph.org") + logging.basicConfig(level=logging.INFO) + print(f"IndexNow key: {key()} (served at {base.rstrip('/')}/{key()}.txt)") + result = submit_all(base) + print(f"Submitted {result['submitted']}/{result['total']} URLs " + f"in {result['batches']} batch(es); statuses={result['statuses']}") diff --git a/templates/base.html.j2 b/templates/base.html.j2 index e7a5615..ae65e09 100644 --- a/templates/base.html.j2 +++ b/templates/base.html.j2 @@ -3,6 +3,7 @@ + {{ head_verify() }} {% block title %}Thermograph{% endblock %} diff --git a/tests/conftest.py b/tests/conftest.py index 01acaae..f92a4cf 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -28,6 +28,8 @@ _TMP = tempfile.mkdtemp(prefix="thermograph-tests-") # start the background notifier thread during tests. Set before db is imported. os.environ.setdefault("THERMOGRAPH_ACCOUNTS_DB", os.path.join(_TMP, "accounts.sqlite")) os.environ.setdefault("THERMOGRAPH_ENABLE_NOTIFIER", "0") +# Pin the IndexNow key so it isn't generated + written into the repo's data/ dir. +os.environ.setdefault("THERMOGRAPH_INDEXNOW_KEY", "test0000indexnow0000key000000000") import store # noqa: E402 diff --git a/tests/test_content.py b/tests/test_content.py index e46d441..a2994ae 100644 --- a/tests/test_content.py +++ b/tests/test_content.py @@ -191,3 +191,51 @@ def test_seo_pages_load_unit_and_account_scripts(client): b = client.get(f"{B}/climate/{SLUG}").text for src in (f"{B}/units.js", f"{B}/account.js", f"{B}/climate.js"): assert f'src="{src}"' in b + + +def test_indexnow_key_file_served(client): + import indexnow + k = indexnow.key() + r = client.get(f"{B}/{k}.txt") + assert r.status_code == 200 + assert r.text.strip() == k + assert r.headers["content-type"].startswith("text/plain") + + +def test_sitemap_lastmod_is_stable(client): + import datetime + import re + body = client.get(f"{B}/sitemap.xml").text + mods = set(re.findall(r"([^<]+)", body)) + assert len(mods) == 1 # one build date, not a per-request churn + datetime.date.fromisoformat(next(iter(mods))) # a valid ISO date + + +def test_indexnow_submit_all_builds_payload(monkeypatch): + import indexnow + + calls = [] + + class _Resp: + status_code = 200 + text = "ok" + + monkeypatch.setattr(indexnow.httpx, "post", + lambda url, json=None, **kw: (calls.append((url, json)), _Resp())[1]) + res = indexnow.submit_all("https://thermograph.org") + assert res["submitted"] == res["total"] > 100 + first = calls[0][1] + assert first["host"] == "thermograph.org" + assert first["keyLocation"] == f"https://thermograph.org/{indexnow.key()}.txt" + assert first["urlList"][0] == "https://thermograph.org/" + assert all(len(c[1]["urlList"]) <= 10000 for c in calls) # batched under the IndexNow cap + + +def test_search_verification_meta(client, monkeypatch): + monkeypatch.setenv("THERMOGRAPH_GOOGLE_VERIFY", "gtok123") + monkeypatch.setenv("THERMOGRAPH_BING_VERIFY", "btok456") + seo = client.get(f"{B}/climate/{SLUG}").text # Jinja SEO page + home = client.get(f"{B}/").text # static homepage via _page() + for b in (seo, home): + assert '' in b + assert '' in b