thermograph/backend/tests/accounts/test_db_engines.py
emi 0617a3b067
All checks were successful
Sync infra to hosts / sync-beta (push) Has been skipped
Build + push images (Forgejo registry) / build-push (frontend) (push) Successful in 9s
Sync infra to hosts / sync-prod (push) Has been skipped
Sync infra to hosts / sync-dev (push) Successful in 12s
Deploy / deploy (frontend) (push) Successful in 19s
secrets-guard / encrypted (push) Successful in 18s
shell-lint / shellcheck (push) Successful in 16s
Build + push images (Forgejo registry) / build-push (backend) (push) Successful in 55s
Deploy / deploy (backend) (push) Successful in 1m22s
PR build (required check) / changes (pull_request) Successful in 8s
secrets-guard / encrypted (pull_request) Successful in 6s
PR build (required check) / build-frontend (pull_request) Has been skipped
shell-lint / shellcheck (pull_request) Successful in 10s
PR build (required check) / validate-observability (pull_request) Has been skipped
PR build (required check) / build-backend (pull_request) Successful in 55s
PR build (required check) / gate (pull_request) Successful in 2s
db: raise max_connections to 200, cap work_mem at 64MB, trim async pool overflow (#124)
2026-07-26 19:48:50 +00:00

113 lines
5.3 KiB
Python

"""The accounts DB engine layer: dialect selection and the read/read-write split.
These run on the SQLite fallback (no THERMOGRAPH_DATABASE_URL in the test env);
the Postgres path (asyncpg RO/RW engines) is exercised by the docker-compose stack,
not the unit suite, so tests stay Postgres-free and fast."""
import asyncio
from sqlalchemy import select
from accounts import db
# The server-side ceiling these pools spend from, and the deployment shape that
# spends it. Kept here so a pool change that stops fitting is a failing test
# rather than a 2am "sorry, too many clients already". Mirrors max_connections in
# infra/deploy/db/init/20-tuning.sh. ONE Postgres instance serves prod and beta.
MAX_CONNECTIONS = 200
PROD_WEB_WORKERS = 4 # uvicorn workers per prod web replica
BETA_WEB_WORKERS = 2
ASYNC_ENGINES_PER_PROCESS = 2 # RW + RO, both built from _ASYNC_POOL_KWARGS
def _async_ceiling_per_process() -> int:
"""Connections one backend process can hold from its async pools at once."""
k = db._ASYNC_POOL_KWARGS
return ASYNC_ENGINES_PER_PROCESS * (k["pool_size"] + k["max_overflow"])
def _sync_ceiling() -> int:
"""The notifier's sync pool. One process holds it cluster-wide (leader flock),
so it is NOT multiplied by workers or replicas."""
k = db._SYNC_POOL_KWARGS
return k["pool_size"] + k["max_overflow"]
def test_defaults_to_sqlite_when_no_database_url():
assert db.IS_POSTGRES is False
# Both engines exist; on SQLite the RO engine is an alias of the RW one.
assert db.async_engine is not None
assert db.async_engine_ro is db.async_engine
def test_read_and_write_session_makers_are_distinct_dependencies():
# Distinct FastAPI deps so pure-read endpoints can opt into the RO engine.
assert db.get_async_session is not db.get_read_session
assert callable(db.get_async_session) and callable(db.get_read_session)
def test_sync_url_derivation():
assert db._sync_url("postgresql+asyncpg://u:p@h:5432/d") == "postgresql+psycopg://u:p@h:5432/d"
assert db._sync_url("sqlite+aiosqlite:////tmp/x.sqlite") == "sqlite:////tmp/x.sqlite"
def test_postgres_pool_sizing_and_sync_timeouts():
"""The Postgres engine-construction kwargs, defined at module scope so they're
testable without a real Postgres connection (the engines built from them are
only actually constructed when IS_POSTGRES, exercised by the docker-compose
stack -- see the module docstring). pool_size=1/max_overflow=1 (the original
setting) let a 3rd concurrent request per worker wait out pool_timeout for a
slot, so the steady-state pool needs real headroom; the sync engine (the
notifier's, driven off-event-loop) must bound connect + statement time so a
wedged Postgres can't hang it forever under its leader flock."""
assert db._ASYNC_POOL_KWARGS["pool_size"] >= 5
# The overflow is the part nobody budgets for -- it is multiplied by two async
# engines x uvicorn workers x replicas, against a ceiling shared with beta.
# Prod saturated max_connections on 2026-07-26 with this at 5. Keep it well
# under pool_size; the budget test below is the real constraint.
assert db._ASYNC_POOL_KWARGS["max_overflow"] <= 2
assert db._SYNC_POOL_KWARGS["pool_size"] < db._ASYNC_POOL_KWARGS["pool_size"]
connect_args = db._SYNC_POOL_KWARGS["connect_args"]
assert connect_args["connect_timeout"] > 0
assert "statement_timeout" in connect_args["options"]
# Build a real (unconnected -- SQLAlchemy engines are lazy) engine from these
# exact kwargs to confirm they're accepted by the psycopg dialect.
import sqlalchemy
engine = sqlalchemy.create_engine("postgresql+psycopg://u:p@h:5432/d", **db._SYNC_POOL_KWARGS)
assert engine.pool.size() == db._SYNC_POOL_KWARGS["pool_size"]
engine.dispose()
def test_pool_budget_fits_under_max_connections():
"""At today's replica counts the whole estate's configured worst case must fit
inside the server's max_connections, with margin -- prod web + prod worker +
beta web + beta worker all draw on ONE Postgres instance. This is the test
that would have failed before 2026-07-26, when the configured worst case was
170 against a ceiling of 100.
Note this counts *configured maxima*: every pool at full overflow at the same
instant. Prod web autoscaling to its max of 3 replicas is deliberately NOT in
this sum -- that peak still does not fit, and closing it needs a connection
pooler or a lower autoscale ceiling rather than a bigger number here."""
per_process = _async_ceiling_per_process()
prod = per_process * PROD_WEB_WORKERS + (per_process + _sync_ceiling())
beta = per_process * BETA_WEB_WORKERS + (per_process + _sync_ceiling())
assert prod + beta <= MAX_CONNECTIONS * 0.8, (
f"configured worst case {prod + beta} leaves too little room under "
f"max_connections={MAX_CONNECTIONS}"
)
def test_read_session_yields_a_working_session():
# On SQLite the read session is fully usable (no read-only enforcement); this
# just confirms the dependency wiring resolves to a live AsyncSession.
db.Base.metadata.create_all(db.sync_engine)
async def _use():
async for session in db.get_read_session():
return (await session.execute(select(1))).scalar_one()
assert asyncio.run(_use()) == 1