thermograph/deploy/deploy.sh

135 lines
6 KiB
Bash
Executable file

#!/usr/bin/env bash
# Pull the latest code and roll the docker-compose stack (backend + frontend +
# PostgreSQL). Run on the VPS — the Forgejo Actions workflow invokes this over
# SSH, and you can run it by hand too.
#
# ssh deploy@vps '/opt/thermograph/deploy/deploy.sh'
set -euo pipefail
APP_DIR="${APP_DIR:-/opt/thermograph}"
BRANCH="${BRANCH:-main}"
HEALTH_PORT="${HEALTH_PORT:-8137}"
cd "$APP_DIR"
# Secrets (POSTGRES_PASSWORD, VAPID keys, AUTH_SECRET, ...) drive compose
# interpolation and are also loaded into the backend container via env_file.
#
# When this host is configured for SOPS (an age key + /etc/thermograph/secrets-env),
# first render /etc/thermograph.env from the committed encrypted source of truth
# (deploy/secrets/*.yaml) so a key rotation is just an edit+commit+deploy. The guard
# on the helper's existence keeps the very deploy that INTRODUCES this file safe: on
# the first pass the checkout may predate it (it arrives with the git reset below,
# after which deploy.sh re-execs), so a missing helper simply falls back to the
# existing /etc/thermograph.env. Then source it so a by-hand run interpolates the
# same as the systemd unit does. See deploy/render-secrets.sh + deploy/secrets/.
if [ -f "$APP_DIR/deploy/render-secrets.sh" ]; then
# shellcheck source=deploy/render-secrets.sh
. "$APP_DIR/deploy/render-secrets.sh"
render_thermograph_secrets "$APP_DIR"
fi
set -a; . /etc/thermograph.env 2>/dev/null || true; set +a
# Pre-warm the ~750 city-page archives so /climate pages serve from cache and a
# search-engine crawl never bursts the archive API quota. Detached inside the
# backend container (compose exec -d), idempotent (skips already-cached cells),
# so it never blocks the deploy or health check and is cheap on every deploy
# after the first full warm.
warm_city_archives() {
echo "==> Warming city-page archives in the background (backend:/app/logs/warm-cities.log)"
docker compose exec -d backend sh -c \
'python warm_cities.py --pace 2 >> /app/logs/warm-cities.log 2>&1' || true
}
# Notify IndexNow (Bing / DuckDuckGo / Yandex) of the site's URLs, but only when
# the set of pages actually changed (a new/removed city) — code-only deploys skip.
# Best-effort: never fails the deploy.
ping_indexnow() {
echo "==> Pinging IndexNow (only if the URL set changed)"
local base="${THERMOGRAPH_BASE_URL:-https://thermograph.org}"
docker compose exec -T backend python indexnow.py --if-changed "$base" \
|| echo "!! IndexNow ping failed (non-fatal)" >&2
}
echo "==> Fetching $BRANCH"
git fetch --prune origin "$BRANCH"
git reset --hard "origin/$BRANCH"
# Re-exec: git reset --hard just rewrote this very file's bytes on disk while
# it's still running. bash reads a script via buffered, byte-offset I/O, so
# anything AFTER this point in the OLD execution can read from the wrong
# offset once the file's size/content changed underneath it -- a classic
# self-modifying-script footgun. Confirmed live: after this PR added ~15
# lines above, one deploy ran with the OLD "Building images" log lines even
# though `git status` showed the checkout correctly at the NEW commit --
# the file changed under a running interpreter, not the checkout. Restart
# fresh from the now-updated file so everything after this line is
# guaranteed self-consistent. Guarded so the second invocation doesn't
# fetch+reset+re-exec forever.
if [ -z "${DEPLOY_SH_REEXECED:-}" ]; then
export DEPLOY_SH_REEXECED=1
exec "$0" "$@"
fi
# Registry-pull cutover (repo-split Stage 6): pull the tag build-push.yml
# already pushed for this exact commit instead of building in place. Mirrors
# build-push.yml's own tag computation (sha-<12 hex>) so this always resolves
# to "whatever HEAD of this branch built to" -- no separate promotion step.
REGISTRY_HOST="${REGISTRY_HOST:-git.thermograph.org}"
IMAGE_PATH="${IMAGE_PATH:-emi/thermograph/app}"
export IMAGE_TAG="sha-$(git rev-parse --short=12 HEAD)"
export REGISTRY_HOST IMAGE_PATH
echo "==> Logging in to the registry ($REGISTRY_HOST)"
echo "$REGISTRY_TOKEN" | docker login "$REGISTRY_HOST" --username emi --password-stdin
echo "==> Pulling images ($IMAGE_TAG)"
# Retry: build-push.yml (triggered by the same push) has no ordering
# guarantee against this deploy -- Forgejo/GitHub Actions `needs:` only
# works between jobs in ONE workflow file, not across two workflows
# triggered by the same event. Confirmed live: a real deploy raced ahead of
# the push and failed with "not found". A bounded retry (~5 min) comfortably
# covers a normal build; a genuine problem (bad tag, registry down) still
# fails loudly after that, just a bit slower to surface.
pull_ok=0
for i in $(seq 1 30); do
if docker compose pull backend frontend; then
pull_ok=1
break
fi
echo " pull attempt $i/30 failed (image may not be pushed yet); retrying in 10s..." >&2
sleep 10
done
if [ "$pull_ok" != 1 ]; then
echo "!! docker compose pull failed after 30 attempts" >&2
exit 1
fi
# Schema migrations run inside the backend container's entrypoint (alembic
# upgrade head) before uvicorn starts, so there's no separate migrate step or
# service restart here — compose owns the process model. frontend has no
# migrations (stateless). --remove-orphans matters whenever the compose
# service topology itself changes (e.g. the old single `app` service ->
# `backend`+`frontend`, repo-split Stage 4): compose only manages containers
# for services CURRENTLY defined in the file, so a plain `up -d` leaves a
# renamed-away service's old container running and squatting its port
# forever -- confirmed live, this is exactly what blocked beta's first
# dual-service deploy ("port is already allocated").
echo "==> Starting stack"
docker compose up -d --remove-orphans
echo "==> Health check"
url="http://127.0.0.1:${HEALTH_PORT}/"
for i in $(seq 1 30); do
if curl -fsS -o /dev/null "$url"; then
echo "==> OK: $url is serving"
warm_city_archives
ping_indexnow
exit 0
fi
sleep 1
done
echo "!! Health check failed for $url" >&2
docker compose ps || true
docker compose logs --tail=50 backend || true
docker compose logs --tail=50 frontend || true
exit 1