name: Ops cron (backup + IndexNow) # Scheduled operational jobs that don't belong in the app's own worker-tier # scheduler (notifications/scheduler.py, Track A chunk 5) because they're # infra/ops concerns rather than app-domain background work -- see the job # classification table in # thermograph-docs/architecture/repo-topology-and-infrastructure.md ยง7. # # Both jobs SSH into the prod host and run inside the already-running compose # stack (docker compose exec), the same way deploy.sh already runs its own # post-deploy IndexNow ping -- no new network exposure, no separate dependency # install. Runs on the `docker` label (the always-on Swarm-hosted runner # deploy/forgejo/ stood up). # # Targets PROD via the PROD_SSH_* secrets (prod = 169.58.46.181, `agent` user, # in the docker group so no sudo needed; /etc/thermograph.env is agent-readable). # These are the same secrets deploy-prod.yml uses for the release->prod deploy -- # NOT the SSH_* secrets, which point at BETA (deploy.yml's `main`->beta path). An # earlier revision reused SSH_* here, so the "prod" backup was silently dumping # beta; prod itself had no backup at all. The prod database is the one that must # be backed up, so both jobs use PROD_SSH_*. # Monorepo port: the compose file now lives under infra/ of the host's # /opt/thermograph monorepo checkout, so every docker-compose invocation cd's # into /opt/thermograph/infra (compose also pins `name: thermograph` so the # project name no longer depends on the directory). Everything else unchanged. on: schedule: # 03:00 UTC daily -- a low-traffic window for both jobs. - cron: '0 3 * * *' workflow_dispatch: {} jobs: backup: name: pg_dump backup runs-on: docker # Guards against the schedule and a manual workflow_dispatch landing # close together (or two manual triggers): queue rather than cancel, so # a workflow_dispatch never aborts a pg_dump mid-write and leaves a # truncated .dump file -- same cancel-in-progress:false rationale as # deploy.yml/deploy-dev.yml's own SSH-script jobs. concurrency: group: ops-backup cancel-in-progress: false steps: - name: Dump the prod database over SSH uses: https://github.com/appleboy/ssh-action@v1.2.0 with: host: ${{ secrets.PROD_SSH_HOST }} username: ${{ secrets.PROD_SSH_USER }} key: ${{ secrets.PROD_SSH_KEY }} port: ${{ secrets.PROD_SSH_PORT }} script: | set -euo pipefail cd /opt/thermograph/infra # Source the env so `docker compose` can interpolate POSTGRES_PASSWORD; # without it compose fails to parse docker-compose.yml, the redirect # still creates the target, and the job leaves a 0-byte .dump and exits # non-zero (exactly how this backup was silently failing). Mirrors the # IndexNow job below, which already sources it. set -a; . /etc/thermograph.env 2>/dev/null || true; set +a backup_dir="$HOME/thermograph-backups" mkdir -p "$backup_dir" stamp="$(date -u +%Y%m%dT%H%M%SZ)" out="$backup_dir/thermograph-$stamp.dump" # The db may run under plain compose OR as a Swarm stack task # (prod post-cutover); resolve the container either way so the # backup survives the deploy-mode switch. dbc=$(docker ps -q --filter "label=com.docker.swarm.service.name=thermograph_db" | head -1) [ -z "$dbc" ] && dbc=$(cd /opt/thermograph/infra && docker compose ps -q db 2>/dev/null | head -1) [ -n "$dbc" ] || { echo "!! no db container found (compose or stack)"; exit 1; } # Write to a .partial and rename on success so a mid-dump failure can # never leave a truncated file that looks like a good backup. docker exec "$dbc" pg_dump -U thermograph -d thermograph \ --format=custom > "$out.partial" mv "$out.partial" "$out" echo "wrote $out ($(du -h "$out" | cut -f1))" # The dumps are the disaster-recovery copy, not a versioned # archive -- keep the last 14 days and let the rest age out. find "$backup_dir" -name 'thermograph-*.dump' -mtime +14 -delete find "$backup_dir" -name 'thermograph-*.dump.partial' -mtime +1 -delete indexnow: name: IndexNow ping runs-on: docker # Same overlap guard as the backup job above (schedule vs. manual # dispatch); --if-changed already makes a second ping a cheap no-op, but # queueing avoids two SSH sessions racing on the host regardless. concurrency: group: ops-indexnow cancel-in-progress: false steps: - name: Ping IndexNow if the URL set changed uses: https://github.com/appleboy/ssh-action@v1.2.0 with: host: ${{ secrets.PROD_SSH_HOST }} username: ${{ secrets.PROD_SSH_USER }} key: ${{ secrets.PROD_SSH_KEY }} port: ${{ secrets.PROD_SSH_PORT }} script: | set -euo pipefail cd /opt/thermograph/infra set -a; . /etc/thermograph.env 2>/dev/null || true; set +a bec=$(docker ps -q --filter "label=com.docker.swarm.service.name=thermograph_web" | head -1) [ -z "$bec" ] && bec=$(cd /opt/thermograph/infra && docker compose ps -q backend 2>/dev/null | head -1) [ -n "$bec" ] || { echo "!! no backend/web container found"; exit 1; } docker exec "$bec" python indexnow.py --if-changed \ "${THERMOGRAPH_BASE_URL:-https://thermograph.org}"