thermograph/infra/ops/dbq.sh
emi d138f00a20
Some checks failed
Sync infra to hosts / sync-beta (push) Has been skipped
Sync infra to hosts / sync-prod (push) Has been skipped
Sync infra to hosts / sync-dev (push) Failing after 6s
secrets-guard / encrypted (push) Successful in 6s
shell-lint / shellcheck (push) Successful in 13s
Validate observability stack / validate (push) Successful in 17s
PR build (required check) / changes (pull_request) Successful in 6s
secrets-guard / encrypted (pull_request) Successful in 5s
PR build (required check) / build-backend (pull_request) Has been skipped
shell-lint / shellcheck (pull_request) Successful in 6s
PR build (required check) / build-frontend (pull_request) Has been skipped
PR build (required check) / validate-observability (pull_request) Successful in 18s
PR build (required check) / gate (pull_request) Successful in 2s
infra: split the estate into vps1/vps2 — beta joins prod, dev gets a home (#103)
2026-07-26 06:56:38 +00:00

105 lines
4 KiB
Bash
Executable file

#!/usr/bin/env bash
# dbq -- run READ-ONLY SQL against any Thermograph Postgres (LAN dev / beta / prod).
#
# infra/ops/dbq.sh dev "select count(*) from climate_history"
# infra/ops/dbq.sh prod -tA "select max(date) from climate_history"
# infra/ops/dbq.sh beta -c '\dt'
# echo "select 1" | infra/ops/dbq.sh prod -f -
#
# Why exec-into-the-container instead of a connection string:
# none of the databases are exposed over TCP. Dev's listens only on its own
# private docker network on vps1; prod and beta share ONE instance on prod's
# Swarm *overlay* (10.0.2.0/24) on vps2, which the host itself cannot route
# to, so `ssh -L` is impossible for either. Publishing 5432 would mean new
# firewall + Swarm endpoint changes on production. Running psql *inside* the
# db container works identically in all three environments with no ports, no
# tunnels and no infra changes -- over SSH to vps1 for dev, to vps2 for
# beta/prod.
#
# Host, SSH target and container filter all come from env-topology.sh, never
# a hardcoded table -- beta and prod are two Swarm stacks on the SAME box
# (vps2) now, and beta has no db container of its own: it reaches prod's
# `thermograph_db` task, the one shared TimescaleDB instance.
#
# Safety: every environment connects as a READ-ONLY role, never as the role the
# app itself uses -- a stray INSERT/DDL fails with "permission denied" even
# against prod. The role is the environment's own owning role with an `_ro`
# suffix, which resolves to the long-standing `thermograph_ro` for prod and dev
# and to `thermograph_beta_ro` for beta.
#
# That suffix convention is why beta needed no special case. It would have been
# easy to give beta one: `CONNECT` is revoked from `PUBLIC` on
# `thermograph_beta`, so prod's `thermograph_ro` cannot reach it, and the
# tempting shortcut was to let beta queries connect as `thermograph_beta` --
# the role that OWNS beta's database. That would have quietly made ad-hoc
# queries read-write on beta alone. deploy/db/provision-env-db.sh provisions the
# `_ro` role for every environment instead.
#
# Any extra arguments are passed straight through to psql, so -tA, -c, -f, -x,
# --csv etc. all work. Stdin is forwarded, so `-f -` reads piped SQL.
set -euo pipefail
SELF_DIR=$(cd "$(dirname "$0")" && pwd)
# shellcheck source=infra/deploy/env-topology.sh
. "$SELF_DIR/../deploy/env-topology.sh"
KEYFILE="${THERMOGRAPH_AGENT_KEY:-$HOME/.ssh/thermograph_agent_ed25519}"
usage() {
sed -n '2,9p' "$0" | sed 's/^# \{0,1\}//' >&2
echo "environments: dev | beta | prod" >&2
exit 2
}
env_name="${1:-}"
[ -n "$env_name" ] || usage
shift
case "$env_name" in
dev|beta|prod) ;;
-h|--help) usage ;;
*) echo "dbq: unknown environment '$env_name' (want: dev|beta|prod)" >&2; exit 2 ;;
esac
thermograph_topology "$env_name"
ssh_target="$TG_SSH_TARGET"
# Container filter: a Swarm task name (shared across prod and beta -- beta has
# no db task of its own) or a compose container name, per the environment's
# deploy mode.
if [ "$TG_DEPLOY_MODE" = stack ]; then
filter="$TG_DB_SERVICE"
else
filter="${TG_COMPOSE_PROJECT}-${TG_DB_SERVICE}"
fi
DB_NAME="${THERMOGRAPH_DB_NAME:-$TG_DB_NAME}"
if [ -n "${THERMOGRAPH_DB_QUERY_USER:-}" ]; then
DB_USER="$THERMOGRAPH_DB_QUERY_USER"
else
# thermograph_ro for prod/dev, thermograph_beta_ro for beta -- provisioned by
# deploy/db/provision-env-db.sh. Never the owning role.
DB_USER="${TG_DB_USER}_ro"
fi
[ $# -gt 0 ] || usage
# Quote the psql arguments so they survive the remote shell intact.
psql_args=$(printf '%q ' "$@")
# The container name is resolved on the target host at call time: prod's Swarm
# task name changes on every redeploy, so it can never be hardcoded.
remote=$(cat <<REMOTE
cid=\$(docker ps -qf name=${filter} | head -1)
[ -n "\$cid" ] || { echo "dbq: no running db container matching '${filter}'" >&2; exit 1; }
exec docker exec -i "\$cid" psql -U ${DB_USER} -d ${DB_NAME} -v ON_ERROR_STOP=1 ${psql_args}
REMOTE
)
if [ -z "$ssh_target" ]; then
exec bash -c "$remote"
else
exec ssh -i "$KEYFILE" -o StrictHostKeyChecking=no -o ConnectTimeout=15 \
"$ssh_target" "$remote"
fi