2026-07-20 14:33:09 +00:00
|
|
|
#!/bin/bash
|
|
|
|
|
# Postgres memory / performance tuning, scaled to the container's DB_MEMORY budget so
|
2026-07-26 06:56:38 +00:00
|
|
|
# the same init works for any host with no hardcoding. There is ONE shared TimescaleDB
|
|
|
|
|
# instance now (prod and beta are separate databases/roles on it, not separate
|
|
|
|
|
# containers), sized from prod's vault values (16g on the 48 GB box) -- beta no longer
|
|
|
|
|
# gets a second database or a second tuning pass. Dev keeps its own, separate container
|
|
|
|
|
# (8g). Runs once on a fresh data volume from /docker-entrypoint-initdb.d, after
|
Move the climate record from parquet to TimescaleDB hypertables (#227)
Replace the per-cell parquet cache with TimescaleDB hypertables as the
production backend for the raw daily climate record, and drop pg_duckdb.
Parquet stays the backend whenever THERMOGRAPH_DATABASE_URL is not a
Postgres URL (dev, tests, offline tooling), the same dialect switch the
accounts DB and derived store already use, so CI stays Postgres-free.
- data/climate_store.py: psycopg + polars bridge over climate_history
(a hypertable), climate_recent, and climate_sync (per-cell freshness).
Reads via pl.read_database, writes via COPY + ON CONFLICT upsert;
fail-soft to a cache miss so a DB hiccup degrades to upstream refetch.
- data/climate.py: route every cache/mtime touchpoint through a backend
dispatch. recent_stamp becomes int(recent_synced_at) on Postgres; the
stale-serve path still avoids bumping it, so derived-payload tokens
invalidate on exactly the same events as before.
- alembic 0002: CREATE EXTENSION timescaledb plus the hypertable schema
(compression policy on year-old chunks), guarded to no-op off Postgres.
- migrate_cache_to_pg.py (make migrate-cache): idempotent backfill of the
parquet cache into the hypertables, preserving file mtimes as sync
timestamps so recent_stamp is unchanged across cutover.
- db image -> stock timescale/timescaledb:latest-pg18; drop the custom
pg_duckdb Dockerfile, the read-only /parquet mount, and the duckdb
tuning GUC. Docs updated for the new backend and cutover.
Co-authored-by: Claude <noreply@anthropic.com>
2026-07-20 20:15:55 +00:00
|
|
|
# 10-timescaledb.sql enables timescaledb. Settings are written via ALTER SYSTEM
|
|
|
|
|
# (persisted to postgresql.auto.conf, which the timescaledb image's own
|
|
|
|
|
# timescaledb-tune postgresql.conf defers to); the container's post-init restart brings
|
|
|
|
|
# restart-only settings (shared_buffers, …) into effect. NB: never ALTER SYSTEM SET
|
|
|
|
|
# shared_preload_libraries here — that would land in auto.conf and override the image's
|
|
|
|
|
# `timescaledb` preload. Init scripts do NOT re-run on an existing volume — to
|
2026-07-20 14:33:09 +00:00
|
|
|
# re-tune later, set DB_MEMORY and run this by hand, then restart:
|
|
|
|
|
# docker compose exec -e DB_MEMORY=16g db bash /docker-entrypoint-initdb.d/20-tuning.sh
|
|
|
|
|
# docker compose restart db
|
|
|
|
|
set -euo pipefail
|
|
|
|
|
|
|
|
|
|
# Parse DB_MEMORY ("16g" / "8192m" / plain MB) into whole MB; default + floor at 8 GB.
|
|
|
|
|
budget="${DB_MEMORY:-8g}"
|
|
|
|
|
num="${budget//[!0-9]/}"
|
|
|
|
|
num="${num:-8}"
|
|
|
|
|
unit="$(printf '%s' "$budget" | tr -dc '[:alpha:]' | tr '[:upper:]' '[:lower:]')"
|
|
|
|
|
case "$unit" in
|
|
|
|
|
g | gb) mb=$((num * 1024)) ;;
|
|
|
|
|
m | mb | "") mb="$num" ;;
|
|
|
|
|
*) mb=8192 ;;
|
|
|
|
|
esac
|
|
|
|
|
if [ "$mb" -lt 1024 ]; then mb=8192; fi
|
|
|
|
|
|
|
|
|
|
# Derive settings from the budget. The ratios reproduce the historical 8 GB tuning
|
Move the climate record from parquet to TimescaleDB hypertables (#227)
Replace the per-cell parquet cache with TimescaleDB hypertables as the
production backend for the raw daily climate record, and drop pg_duckdb.
Parquet stays the backend whenever THERMOGRAPH_DATABASE_URL is not a
Postgres URL (dev, tests, offline tooling), the same dialect switch the
accounts DB and derived store already use, so CI stays Postgres-free.
- data/climate_store.py: psycopg + polars bridge over climate_history
(a hypertable), climate_recent, and climate_sync (per-cell freshness).
Reads via pl.read_database, writes via COPY + ON CONFLICT upsert;
fail-soft to a cache miss so a DB hiccup degrades to upstream refetch.
- data/climate.py: route every cache/mtime touchpoint through a backend
dispatch. recent_stamp becomes int(recent_synced_at) on Postgres; the
stale-serve path still avoids bumping it, so derived-payload tokens
invalidate on exactly the same events as before.
- alembic 0002: CREATE EXTENSION timescaledb plus the hypertable schema
(compression policy on year-old chunks), guarded to no-op off Postgres.
- migrate_cache_to_pg.py (make migrate-cache): idempotent backfill of the
parquet cache into the hypertables, preserving file mtimes as sync
timestamps so recent_stamp is unchanged across cutover.
- db image -> stock timescale/timescaledb:latest-pg18; drop the custom
pg_duckdb Dockerfile, the read-only /parquet mount, and the duckdb
tuning GUC. Docs updated for the new backend and cutover.
Co-authored-by: Claude <noreply@anthropic.com>
2026-07-20 20:15:55 +00:00
|
|
|
# (shared_buffers 2 GB, effective_cache_size 6 GB, work_mem 64 MB,
|
2026-07-20 14:33:09 +00:00
|
|
|
# maintenance_work_mem 512 MB) and scale linearly on a bigger box.
|
|
|
|
|
shared_buffers=$((mb / 4)) # 25% — the shared page cache
|
|
|
|
|
effective_cache=$((mb * 3 / 4)) # 75% — planner's view of total cache (PG + OS)
|
|
|
|
|
work_mem=$((mb / 128)) # ~64 MB at 8 GB (per-operation; kept modest)
|
|
|
|
|
if [ "$work_mem" -lt 16 ]; then work_mem=16; fi
|
|
|
|
|
maint_mem=$((mb / 16)) # 512 MB at 8 GB — index builds / VACUUM
|
|
|
|
|
|
|
|
|
|
echo "[tuning] DB_MEMORY=${budget} -> ${mb}MB: shared_buffers=${shared_buffers}MB" \
|
|
|
|
|
"effective_cache_size=${effective_cache}MB work_mem=${work_mem}MB" \
|
Move the climate record from parquet to TimescaleDB hypertables (#227)
Replace the per-cell parquet cache with TimescaleDB hypertables as the
production backend for the raw daily climate record, and drop pg_duckdb.
Parquet stays the backend whenever THERMOGRAPH_DATABASE_URL is not a
Postgres URL (dev, tests, offline tooling), the same dialect switch the
accounts DB and derived store already use, so CI stays Postgres-free.
- data/climate_store.py: psycopg + polars bridge over climate_history
(a hypertable), climate_recent, and climate_sync (per-cell freshness).
Reads via pl.read_database, writes via COPY + ON CONFLICT upsert;
fail-soft to a cache miss so a DB hiccup degrades to upstream refetch.
- data/climate.py: route every cache/mtime touchpoint through a backend
dispatch. recent_stamp becomes int(recent_synced_at) on Postgres; the
stale-serve path still avoids bumping it, so derived-payload tokens
invalidate on exactly the same events as before.
- alembic 0002: CREATE EXTENSION timescaledb plus the hypertable schema
(compression policy on year-old chunks), guarded to no-op off Postgres.
- migrate_cache_to_pg.py (make migrate-cache): idempotent backfill of the
parquet cache into the hypertables, preserving file mtimes as sync
timestamps so recent_stamp is unchanged across cutover.
- db image -> stock timescale/timescaledb:latest-pg18; drop the custom
pg_duckdb Dockerfile, the read-only /parquet mount, and the duckdb
tuning GUC. Docs updated for the new backend and cutover.
Co-authored-by: Claude <noreply@anthropic.com>
2026-07-20 20:15:55 +00:00
|
|
|
"maintenance_work_mem=${maint_mem}MB"
|
2026-07-20 14:33:09 +00:00
|
|
|
|
|
|
|
|
psql -v ON_ERROR_STOP=1 --username "$POSTGRES_USER" --dbname "$POSTGRES_DB" <<SQL
|
|
|
|
|
-- Caching: the shared page cache, and the planner's view of total cache (PG + OS).
|
|
|
|
|
ALTER SYSTEM SET shared_buffers = '${shared_buffers}MB';
|
|
|
|
|
ALTER SYSTEM SET effective_cache_size = '${effective_cache}MB';
|
|
|
|
|
|
|
|
|
|
-- Processing: per-operation sort/hash memory, and maintenance (index builds, VACUUM).
|
|
|
|
|
ALTER SYSTEM SET work_mem = '${work_mem}MB';
|
|
|
|
|
ALTER SYSTEM SET maintenance_work_mem = '${maint_mem}MB';
|
|
|
|
|
|
|
|
|
|
-- Write throughput: fewer, larger checkpoints.
|
|
|
|
|
ALTER SYSTEM SET wal_buffers = '16MB';
|
|
|
|
|
ALTER SYSTEM SET min_wal_size = '1GB';
|
|
|
|
|
ALTER SYSTEM SET max_wal_size = '4GB';
|
|
|
|
|
ALTER SYSTEM SET checkpoint_completion_target = 0.9;
|
|
|
|
|
|
|
|
|
|
-- SSD-friendly planner + IO concurrency.
|
|
|
|
|
ALTER SYSTEM SET random_page_cost = 1.1;
|
|
|
|
|
ALTER SYSTEM SET effective_io_concurrency = 200;
|
|
|
|
|
|
|
|
|
|
-- Room for parallel scans/aggregates on the bigger analytic queries.
|
|
|
|
|
ALTER SYSTEM SET max_parallel_workers_per_gather = 2;
|
|
|
|
|
SQL
|