thermograph/infra/deploy/stack/autoscale.sh
Emi Griffith ae1d9bb534 Subtree-merge thermograph-infra (origin/main) into infra/
git-subtree-dir: infra
git-subtree-mainline: d6df04eab2
git-subtree-split: 99b4b3f78d
2026-07-22 22:01:11 -07:00

93 lines
3.1 KiB
Bash
Executable file

#!/bin/sh
# Autoscaler for the stack's `web` service: scale replicas between
# MIN_REPLICAS and MAX_REPLICAS on sustained per-task CPU.
#
# Runs as a Swarm service on the manager with the docker socket mounted (see
# thermograph-stack.yml). Every web task is placed on this node today, so
# node-local `docker stats` sees them all — when a second app node exists,
# this needs a per-node reader or a metrics-based signal instead; that's the
# documented upgrade path, not a today problem.
#
# Semantics (deliberately boring):
# - Sample avg CPU% per web task every POLL_SECONDS (docker stats CPUPerc:
# 100 = one full host core).
# - UP_SAMPLES consecutive samples above SCALE_UP_CPU -> scale +1.
# - DOWN_SAMPLES consecutive samples below SCALE_DOWN_CPU -> scale -1.
# (Down is ~7x slower than up on defaults: flap-averse by construction.)
# - COOLDOWN_SECONDS after any change: samples are ignored entirely.
# - Clamped to [MIN_REPLICAS, MAX_REPLICAS]; scaling waits for convergence
# (--detach=false), so a stuck rollout blocks further changes rather than
# stacking them.
set -eu
STACK_NAME="${STACK_NAME:-thermograph}"
SERVICE="${STACK_NAME}_web"
MIN="${MIN_REPLICAS:-1}"
MAX="${MAX_REPLICAS:-3}"
UP_AT="${SCALE_UP_CPU:-220}"
DOWN_AT="${SCALE_DOWN_CPU:-60}"
POLL="${POLL_SECONDS:-15}"
UP_N="${UP_SAMPLES:-3}"
DOWN_N="${DOWN_SAMPLES:-20}"
COOLDOWN="${COOLDOWN_SECONDS:-180}"
up_hits=0
down_hits=0
last_change=0
log() { echo "[autoscale] $(date -u +%H:%M:%S) $*"; }
replicas() {
docker service inspect "$SERVICE" \
--format '{{.Spec.Mode.Replicated.Replicas}}' 2>/dev/null || echo ""
}
avg_cpu() {
# Mean CPUPerc across this node's web tasks, as an integer percent.
docker stats --no-stream --format '{{.Name}} {{.CPUPerc}}' 2>/dev/null \
| awk -v svc="$SERVICE" '
index($1, svc".") == 1 {
gsub(/%/, "", $2); sum += $2; n++
}
END { if (n > 0) printf "%d", sum / n; else print "" }'
}
log "watching $SERVICE: min=$MIN max=$MAX up>@${UP_AT}%x${UP_N} down<@${DOWN_AT}%x${DOWN_N} poll=${POLL}s cooldown=${COOLDOWN}s"
while :; do
sleep "$POLL"
now=$(date +%s)
if [ $((now - last_change)) -lt "$COOLDOWN" ]; then
continue
fi
cur=$(replicas)
[ -n "$cur" ] || { log "service $SERVICE not found; waiting"; continue; }
cpu=$(avg_cpu)
[ -n "$cpu" ] || continue # no running tasks visible this sample
if [ "$cpu" -gt "$UP_AT" ]; then
up_hits=$((up_hits + 1)); down_hits=0
elif [ "$cpu" -lt "$DOWN_AT" ]; then
down_hits=$((down_hits + 1)); up_hits=0
else
up_hits=0; down_hits=0
fi
if [ "$up_hits" -ge "$UP_N" ] && [ "$cur" -lt "$MAX" ]; then
target=$((cur + 1))
log "avg cpu ${cpu}% > ${UP_AT}% x${UP_N}: scaling $cur -> $target"
if docker service scale --detach=false "$SERVICE=$target"; then
last_change=$(date +%s)
fi
up_hits=0; down_hits=0
elif [ "$down_hits" -ge "$DOWN_N" ] && [ "$cur" -gt "$MIN" ]; then
target=$((cur - 1))
log "avg cpu ${cpu}% < ${DOWN_AT}% x${DOWN_N}: scaling $cur -> $target"
if docker service scale --detach=false "$SERVICE=$target"; then
last_change=$(date +%s)
fi
up_hits=0; down_hits=0
fi
done