diff --git a/.forgejo/workflows/observability-validate.yml b/.forgejo/workflows/observability-validate.yml index 7b6fcc0..157a3d9 100644 --- a/.forgejo/workflows/observability-validate.yml +++ b/.forgejo/workflows/observability-validate.yml @@ -12,10 +12,13 @@ name: Validate observability stack # path-filtered to the domain, and workflow_call lets pr-build.yml reuse this # as the domain's PR check under its single `gate` required status. # -# Not validated here: the Alloy config (observability/alloy/config.alloy, an -# HCL-like format, needs the `alloy` binary) and docker-compose *schema* -# (unknown-key checks, which would need the docker CLI). Add either if the -# churn warrants the extra tooling. +# The Alloy config (observability/alloy/config.alloy, an HCL-like format) IS +# validated below -- the static `alloy` binary is downloaded straight from its +# GitHub release (pinned to the same v1.9.1 the fleet runs; see +# observability/alloy/docker-compose.agent.yml), no docker CLI needed. +# +# Not validated here: docker-compose *schema* (unknown-key checks, which would +# need the docker CLI). Add it if the churn warrants the extra tooling. on: workflow_call: {} @@ -70,3 +73,21 @@ jobs: print("INVALID YAML", f, "-", e); bad = True sys.exit(1 if bad else 0) PY + + - name: Alloy config parses and validates (components, relabel/process wiring) + run: | + set -euo pipefail + apt-get update -qq && apt-get install -y -qq unzip >/dev/null + curl -sSL -o /tmp/alloy.zip \ + https://github.com/grafana/alloy/releases/download/v1.9.1/alloy-linux-amd64.zip + unzip -q /tmp/alloy.zip -d /tmp/alloybin + chmod +x /tmp/alloybin/alloy-linux-amd64 + + # `alloy validate` builds the real component graph (catches bad field + # names, dangling forward_to/receiver refs, malformed relabel/process + # stages) but doesn't recognize the top-level `livedebugging {}` singleton + # block as a component -- it only knows named components, even though + # `alloy run` loads that block fine. Known gap in the `validate` + # subcommand, not a config error, so strip that one line before checking. + grep -v '^livedebugging ' observability/alloy/config.alloy > /tmp/config-for-validate.alloy + /tmp/alloybin/alloy-linux-amd64 validate /tmp/config-for-validate.alloy diff --git a/infra/deploy/Caddyfile b/infra/deploy/Caddyfile index 0d6a31a..6baf615 100644 --- a/infra/deploy/Caddyfile +++ b/infra/deploy/Caddyfile @@ -31,10 +31,12 @@ thermograph.org { # Active health check on the same cheap /healthz route each container's own # HEALTHCHECK uses (Dockerfile) — so a deploy that's still restarting/booting # never gets proxied into (a reload alone has no gate, hop-1 runbook hazard #10). + # 15s (was 5s): plenty responsive for a process that only restarts on a deploy, + # and a quarter of the polling load. handle @backend_paths { reverse_proxy 127.0.0.1:8137 { health_uri /healthz - health_interval 5s + health_interval 15s health_timeout 3s health_status 2xx } @@ -43,14 +45,35 @@ thermograph.org { handle { reverse_proxy 127.0.0.1:8080 { health_uri /healthz - health_interval 5s + health_interval 15s health_timeout 3s health_status 2xx } } + # Access-log hygiene: the default JSON encoder serializes full request headers, + # the TLS block, and response headers on every line (measured ~1,133B/line) -- + # strip those with the `filter` format encoder. Also strip the query string from + # the logged URI: Caddy's default logger records request.uri *including* the + # query string, so every `?q=` a visitor typed sat in Loki next to + # their client IP for the full 30-day retention -- a real privacy leak, not just + # noise. Bot/crawler skipping stays out of here: `log_skip` needs Caddy >= 2.7 + # and an upgrade is out of scope, so that's handled downstream in Alloy's + # loki.process "caddy" stage instead (see observability/alloy/config.alloy). log { - output file /var/log/caddy/thermograph.log + output file /var/log/caddy/thermograph.log { + roll_size 20MiB + roll_keep 5 + } + format filter { + wrap json + fields { + request>headers delete + request>tls delete + resp_headers delete + request>uri regexp \?.* "" + } + } } } diff --git a/infra/deploy/stack/deploy-stack.sh b/infra/deploy/stack/deploy-stack.sh index 15d8a99..c3badc2 100755 --- a/infra/deploy/stack/deploy-stack.sh +++ b/infra/deploy/stack/deploy-stack.sh @@ -166,13 +166,19 @@ else echo "==> Rolling web + worker + lake to $BACKEND_IMAGE" docker service update --with-registry-auth --detach=false --image "$BACKEND_IMAGE" "${STACK_NAME}_web" docker service update --with-registry-auth --detach=false --image "$BACKEND_IMAGE" "${STACK_NAME}_worker" - # lake ships in the same image; a stack file predating it has no service - # yet — the next SERVICE=all stack deploy creates it, so don't fail here. - if docker service inspect "${STACK_NAME}_lake" >/dev/null 2>&1; then - docker service update --with-registry-auth --detach=false --image "$BACKEND_IMAGE" "${STACK_NAME}_lake" - else - echo " (no ${STACK_NAME}_lake service yet; created on the next full stack deploy)" - fi + # lake and daemon ship in the same image; a stack file predating either + # has no service yet — the next SERVICE=all stack deploy creates it, so + # don't fail here. The daemon especially must roll with web: they share + # the /internal/* contract, and a version skew between them is exactly + # what pinning one BACKEND_IMAGE_TAG exists to prevent (seen live: the + # first post-creation backend roll left the daemon a release behind). + for extra in lake daemon; do + if docker service inspect "${STACK_NAME}_${extra}" >/dev/null 2>&1; then + docker service update --with-registry-auth --detach=false --image "$BACKEND_IMAGE" "${STACK_NAME}_${extra}" + else + echo " (no ${STACK_NAME}_${extra} service yet; created on the next full stack deploy)" + fi + done ;; frontend) echo "==> Rolling frontend to $FRONTEND_IMAGE" diff --git a/infra/terraform/modules/thermograph-host/templates/Caddyfile.tftpl b/infra/terraform/modules/thermograph-host/templates/Caddyfile.tftpl index 340bdb7..95cd448 100644 --- a/infra/terraform/modules/thermograph-host/templates/Caddyfile.tftpl +++ b/infra/terraform/modules/thermograph-host/templates/Caddyfile.tftpl @@ -12,6 +12,16 @@ # # NOTE: the repo's deploy/Caddyfile additionally serves the emigriffith.dev portfolio # and legacy redirects; those are host-specific and intentionally not templated here. +# +# Access-log hygiene: the default JSON encoder serializes full request headers, the +# TLS block, and response headers on every line (measured ~1,133B/line on prod, +# ~922B on beta) -- strip those with the `filter` format encoder. Also strip the +# query string from the logged URI: Caddy's default logger records request.uri +# *including* the query string, so every `?q=` a visitor typed sat in +# Loki next to their client IP for the full 30-day retention -- a real privacy +# leak, not just noise. Bot/crawler skipping stays out of here: `log_skip` needs +# Caddy >= 2.7 and an upgrade is out of scope, so that's handled downstream in +# Alloy's loki.process "caddy" stage instead (see observability/alloy/config.alloy). ${domain} { encode zstd gzip @@ -21,10 +31,12 @@ ${domain} { # HEALTHCHECK uses (Dockerfile) — so a `docker compose up -d --build` deploy # that's still restarting/booting never gets proxied into (a reload alone has # no gate, hop-1 runbook hazard #10). health_uri is relative to the upstream. + # 15s (was 5s): plenty responsive for an active health check against a process + # that only restarts on a deploy, and a quarter of the polling load. handle @backend_paths { reverse_proxy 127.0.0.1:${port} { health_uri /healthz - health_interval 5s + health_interval 15s health_timeout 3s health_status 2xx } @@ -33,13 +45,25 @@ ${domain} { handle { reverse_proxy 127.0.0.1:${frontend_port} { health_uri /healthz - health_interval 5s + health_interval 15s health_timeout 3s health_status 2xx } } log { - output file /var/log/caddy/thermograph.log + output file /var/log/caddy/thermograph.log { + roll_size 20MiB + roll_keep 5 + } + format filter { + wrap json + fields { + request>headers delete + request>tls delete + resp_headers delete + request>uri regexp \?.* "" + } + } } } diff --git a/observability/alloy/config.alloy b/observability/alloy/config.alloy index 7ddbc29..3279da2 100644 --- a/observability/alloy/config.alloy +++ b/observability/alloy/config.alloy @@ -15,12 +15,22 @@ livedebugging { enabled = false } // --- 1. All Docker container logs ------------------------------------------------ +// refresh_interval defaults to 60s; Swarm task churn reshuffles the target set on +// roughly that cadence, which restarts tailers ~every 90s and was costing ~11% of +// Alloy's own CPU in tailer restarts alone. 5m is still fast enough to pick up a +// real deploy without paying that churn cost. discovery.docker "containers" { - host = "unix:///var/run/docker.sock" + host = "unix:///var/run/docker.sock" + refresh_interval = "5m" } // Turn Docker metadata into tidy labels: `container` (short name) and `service` -// (the compose service, e.g. app/db). Drop Alloy's own container to avoid a loop. +// (the compose service, e.g. app/db). Drop noise/duplicate containers so Loki never +// ingests them: Alloy itself (loop), the autoscaler, throwaway `thermograph-test_*` +// stacks, the loopback LB bridge (`thermograph-lb`, pure plumbing, nothing to debug +// from its logs), and the app's own worker (`thermograph_worker`'s stdout is 100% +// `/healthz` poll noise — the app's `access/*.jsonl` under source #3 is a strict +// superset of anything useful it logs). discovery.relabel "containers" { targets = discovery.docker.containers.targets @@ -35,27 +45,55 @@ discovery.relabel "containers" { } rule { source_labels = ["container"] - regex = ".*alloy.*" + regex = "(alloy|autoscaler|thermograph-test_.*|thermograph-lb|thermograph_worker).*" action = "drop" } } +// Pass RAW targets here (not discovery.relabel.containers.output) alongside +// relabel_rules: loki.source.docker applies relabel_rules itself, once, using the +// __meta_docker_* metadata it still holds at collection time. Passing the +// already-relabelled output *and* relabel_rules ran the same rules twice per log +// entry, and made the drop rules above a no-op on the second pass since the +// __meta_docker_* labels are already gone from the pre-relabelled output. loki.source.docker "containers" { host = "unix:///var/run/docker.sock" - targets = discovery.relabel.containers.output + targets = discovery.docker.containers.targets forward_to = [loki.write.central.receiver] relabel_rules = discovery.relabel.containers.rules labels = { job = "docker" } } // --- 2. Caddy host access logs --------------------------------------------------- +// sync_period (glob rescan) defaults to 10s; 1m is plenty for a log file that only +// appears/rotates on the order of hours. local.file_match "caddy" { path_targets = [{ __path__ = "/var/log/caddy/*.log", job = "caddy" }] + sync_period = "1m" } loki.source.file "caddy" { targets = local.file_match.caddy.targets + forward_to = [loki.process.caddy.receiver] + + // PollingFileWatcher defaults (250ms/250ms) stat every tailed file 4x/second + // forever. Caddy's access log doesn't need sub-second latency into Loki. + file_watch { + min_poll_frequency = "2s" + max_poll_frequency = "10s" + } +} + +// Drop well-known crawler/bot traffic before it hits Loki. Caddy itself can't do +// this cheaply (log_skip needs Caddy >= 2.7; both hosts run older Caddy, and an +// upgrade is out of scope here), so filter it at the shipper instead. +loki.process "caddy" { forward_to = [loki.write.central.receiver] + + stage.drop { + expression = "(?i)(semrushbot|claudebot|ahrefsbot|yandexbot|bytespider|mj12bot|petalbot)" + drop_counter_reason = "crawler" + } } // --- 3. App structured JSON logs (errors / access / audit) ----------------------- @@ -63,11 +101,17 @@ loki.source.file "caddy" { // compose). Lift `level`/`tag`/`phase` out of the JSON so they're queryable. local.file_match "app_jsonl" { path_targets = [{ __path__ = "/applogs/**/*.jsonl", job = "app-json" }] + sync_period = "1m" } loki.source.file "app_jsonl" { targets = local.file_match.app_jsonl.targets forward_to = [loki.process.app_jsonl.receiver] + + file_watch { + min_poll_frequency = "2s" + max_poll_frequency = "10s" + } } loki.process "app_jsonl" { diff --git a/observability/loki/config.yml b/observability/loki/config.yml index d17a37f..b2dfb46 100644 --- a/observability/loki/config.yml +++ b/observability/loki/config.yml @@ -32,6 +32,17 @@ schema_config: prefix: index_ period: 24h +# 30 low-rate streams almost always hit chunk_idle_period (30m default) long before +# they'd ever fill chunk_target_size, so chunks were flushed 1.98% full on average +# (972 near-empty chunks for 30MB of actual log data). Give idle streams far longer +# to accumulate before an idle flush, and cap age/size so a chunk still can't grow +# unbounded. +ingester: + chunk_idle_period: 2h + max_chunk_age: 12h + chunk_target_size: 1572864 + chunk_encoding: snappy + limits_config: # A hobby fleet's volume is tiny; keep 30 days and cap ingestion generously. retention_period: 720h @@ -40,6 +51,15 @@ limits_config: max_query_series: 5000 allow_structured_metadata: true volume_enabled: true + # No explicit ingestion/stream limits meant Loki fell back to its (much stricter) + # built-in defaults, which is why 25 pushes came back HTTP 429 with nothing in + # this file explaining why. These are sized for a three-node hobby fleet, not the + # multi-tenant defaults. + ingestion_rate_mb: 8 + ingestion_burst_size_mb: 16 + per_stream_rate_limit: 3MB + per_stream_rate_limit_burst: 10MB + max_global_streams_per_user: 1000 compactor: working_directory: /loki/compactor