thermograph/backend/tests/api/test_internal_routes.py
emi 2d3f37c474
All checks were successful
Sync infra to hosts / sync-beta (push) Successful in 8s
Sync infra to hosts / sync-prod (push) Successful in 7s
secrets-guard / encrypted (push) Successful in 8s
shell-lint / shellcheck (push) Successful in 10s
Build + push backend image (Forgejo registry) / build-push (push) Successful in 1m14s
Deploy backend to beta VPS / deploy (push) Successful in 2m4s
daemon: move the Discord gateway and scheduler out of the web process into Go (#21)
The gateway bot and APScheduler were long-lived stateful I/O loops running
inside the async web app under a leader election. They move into a single Go
binary that owns ONLY that I/O -- websocket, RESUME, heartbeat, backoff, timers.

It owns no grading logic. Anything needing data calls back over a new
internal-only surface (/internal/discord/grade, /internal/jobs/*). Grading
depends on polars and the parquet cache; reimplementing it in Go would let the
bot's grades drift from the API's. The grade route returns gateway-ready JSON
and Go relays the bytes verbatim.

The binary ships in the backend image and runs as a second compose service off
the same tag, so the two ends of the /internal/* contract can never skew.
deploy.sh rolls daemon alongside backend -- without that the service would never
be created, since a single-service deploy uses --no-deps. It also probes the
image first and skips the daemon when rolling a tag that predates the binary:
infra tracks main while image tags are env-staged, so a host can legitimately be
asked to roll an older backend image, and creating the service anyway would
leave a container crash-looping on a missing binary.

replicas: 1 with order: stop-first replaces the leader election -- Discord
permits one gateway connection per bot token.

THERMOGRAPH_INTERNAL_TOKEN is optional: both ends derive it from
THERMOGRAPH_AUTH_SECRET via HMAC under a domain-separation label, so this needs
no new vault entry. The derivation is pinned to a shared cross-language test
vector asserted on both sides, so drift fails CI instead of 401ing every call.
Fail closed when neither secret is set.

Improvements over the Python: a close intended for RESUME uses 4000 rather than
1000 (Discord invalidates a session closed 1000, so the old default defeated its
own resume); MESSAGE_CREATE runs on a bounded worker pool; and a malformed HELLO
returns an error rather than a clean reconnect, which would otherwise reset
backoff and hot-loop against the gateway.

365 Python tests pass; Go build/vet/test -race clean; shellcheck 0 findings.
2026-07-23 22:49:54 +00:00

134 lines
5.6 KiB
Python

"""The internal control surface the thermograph-daemon Go binary calls back
into (api/internal_routes.py): shared-secret gating (fail closed when the token
was never provisioned), the gateway-ready /grade reply with the interactions-only
ephemeral flag dropped, the two job triggers, and the one-in-flight-run guard.
The underlying job/grade functions are stubbed — hermetic, no real network."""
import pytest
from fastapi.testclient import TestClient
from api import internal_routes
from notifications import discord_interactions as di
from web import app as appmod
TOKEN = "test-internal-token"
HDR = {"X-Thermograph-Internal-Token": TOKEN}
@pytest.fixture
def client(monkeypatch):
monkeypatch.setenv("THERMOGRAPH_INTERNAL_TOKEN", TOKEN)
return TestClient(appmod.app)
@pytest.fixture
def grade_stub(monkeypatch):
"""Reply payload for the shared grade builder, mirroring the real shape
(embeds + an ephemeral flag the gateway path must drop)."""
calls = []
monkeypatch.setattr(
di, "_grade_message",
lambda query: calls.append(query) or {"embeds": [{"title": f"grade:{query}"}],
"flags": 64})
return calls
# --- auth: fail closed, then reject before accept ----------------------------
def test_router_is_disabled_when_no_token_is_provisioned(monkeypatch):
"""No THERMOGRAPH_INTERNAL_TOKEN => the surface doesn't exist (404), even
for a caller presenting a header — never fall open to no-auth."""
monkeypatch.delenv("THERMOGRAPH_INTERNAL_TOKEN", raising=False)
client = TestClient(appmod.app)
assert client.post("/internal/discord/grade", json={"query": "x"},
headers=HDR).status_code == 404
assert client.post("/internal/jobs/warm-cities", json={}, headers=HDR).status_code == 404
assert client.post("/internal/jobs/indexnow", json={}, headers=HDR).status_code == 404
def test_missing_header_is_rejected(client):
assert client.post("/internal/jobs/indexnow", json={}).status_code == 401
def test_wrong_token_is_rejected(client):
r = client.post("/internal/jobs/indexnow", json={},
headers={"X-Thermograph-Internal-Token": "not-the-token"})
assert r.status_code == 401
def test_correct_token_is_accepted(client, monkeypatch):
import indexnow
monkeypatch.setattr(indexnow, "submit_if_changed", lambda: None)
assert client.post("/internal/jobs/indexnow", json={}, headers=HDR).status_code == 200
# --- /internal/discord/grade -------------------------------------------------
def test_grade_returns_the_shared_builder_message_without_ephemeral_flag(client, grade_stub):
r = client.post("/internal/discord/grade", json={"query": "Phoenix"}, headers=HDR)
assert r.status_code == 200
# Gateway messages can't be ephemeral — the flag must be gone, everything
# else exactly as _grade_message built it.
assert r.json() == {"embeds": [{"title": "grade:Phoenix"}]}
assert grade_stub == ["Phoenix"]
def test_grade_requires_a_query(client, grade_stub):
assert client.post("/internal/discord/grade", json={}, headers=HDR).status_code == 422
assert grade_stub == []
# --- job triggers ------------------------------------------------------------
def test_warm_cities_invokes_warm_cities_main(client, monkeypatch):
import warm_cities
calls = []
monkeypatch.setattr(warm_cities, "main", lambda: calls.append(1))
r = client.post("/internal/jobs/warm-cities", json={}, headers=HDR)
assert r.status_code == 200
assert r.json() == {"ok": True}
assert calls == [1]
def test_indexnow_invokes_submit_if_changed(client, monkeypatch):
import indexnow
calls = []
monkeypatch.setattr(indexnow, "submit_if_changed", lambda: calls.append(1))
r = client.post("/internal/jobs/indexnow", json={}, headers=HDR)
assert r.status_code == 200
assert r.json() == {"ok": True}
assert calls == [1]
# --- one-in-flight-run guard -------------------------------------------------
def test_concurrent_job_invocation_answers_409(client, monkeypatch):
"""A second trigger while a run is in flight must not stack — warm_cities
spends the Open-Meteo quota, so a stacked run double-spends it. Holding the
job's lock stands in for an in-flight run."""
import indexnow
monkeypatch.setattr(indexnow, "submit_if_changed", lambda: None)
lock = internal_routes._JOB_LOCKS["warm-cities"]
assert lock.acquire(blocking=False)
try:
assert client.post("/internal/jobs/warm-cities", json={}, headers=HDR).status_code == 409
# The guard is per job: an in-flight warm run doesn't block indexnow.
assert client.post("/internal/jobs/indexnow", json={}, headers=HDR).status_code == 200
finally:
lock.release()
# Released => the next trigger runs again.
import warm_cities
monkeypatch.setattr(warm_cities, "main", lambda: None)
assert client.post("/internal/jobs/warm-cities", json={}, headers=HDR).status_code == 200
def test_job_lock_is_released_after_a_failing_run(monkeypatch):
"""A crashed run must not wedge the job forever: the 500 surfaces, the lock
is released, and the next trigger runs."""
monkeypatch.setenv("THERMOGRAPH_INTERNAL_TOKEN", TOKEN)
client = TestClient(appmod.app, raise_server_exceptions=False)
import indexnow
monkeypatch.setattr(indexnow, "submit_if_changed",
lambda: (_ for _ in ()).throw(RuntimeError("boom")))
assert client.post("/internal/jobs/indexnow", json={}, headers=HDR).status_code == 500
monkeypatch.setattr(indexnow, "submit_if_changed", lambda: None)
assert client.post("/internal/jobs/indexnow", json={}, headers=HDR).status_code == 200