// Grafana Alloy — the log collector that runs on EVERY node (prod, beta, LAN // dev). It gathers three sources and ships them to the central Loki on beta: // // 1. Every Docker container's stdout/stderr (app, db, and on beta forgejo) // via the Docker socket. // 2. Caddy's host access logs (/var/log/caddy/*.log) — the reverse proxy runs // on the host, not in a container, so its logs aren't in Docker. // 3. The app's structured JSON logs (errors/access/audit *.jsonl) from the // `applogs` Docker volume, parsed so `level`/`tag`/`phase` become labels. // // Per-node settings come from the environment (see docker-compose.agent.yml): // ALLOY_NODE — this node's name label (prod | beta | dev) // LOKI_URL — where to push (http://10.10.0.2:3100/loki/api/v1/push over wg0) livedebugging { enabled = false } // --- 1. All Docker container logs ------------------------------------------------ // refresh_interval defaults to 60s; Swarm task churn reshuffles the target set on // roughly that cadence, which restarts tailers ~every 90s and was costing ~11% of // Alloy's own CPU in tailer restarts alone. 5m is still fast enough to pick up a // real deploy without paying that churn cost. discovery.docker "containers" { host = "unix:///var/run/docker.sock" refresh_interval = "5m" } // Turn Docker metadata into tidy labels: `container` (short name) and `service` // (the compose service, e.g. app/db). Drop noise/duplicate containers so Loki never // ingests them: Alloy itself (loop), the autoscaler, throwaway `thermograph-test_*` // stacks, the loopback LB bridge (`thermograph-lb`, pure plumbing, nothing to debug // from its logs), and the app's own worker (`thermograph_worker`'s stdout is 100% // `/healthz` poll noise — the app's `access/*.jsonl` under source #3 is a strict // superset of anything useful it logs). discovery.relabel "containers" { targets = discovery.docker.containers.targets rule { source_labels = ["__meta_docker_container_name"] regex = "/(.*)" target_label = "container" } rule { source_labels = ["__meta_docker_container_label_com_docker_compose_service"] target_label = "service" } rule { source_labels = ["container"] regex = "(alloy|autoscaler|thermograph-test_.*|thermograph-lb|thermograph_worker).*" action = "drop" } } // Pass RAW targets here (not discovery.relabel.containers.output) alongside // relabel_rules: loki.source.docker applies relabel_rules itself, once, using the // __meta_docker_* metadata it still holds at collection time. Passing the // already-relabelled output *and* relabel_rules ran the same rules twice per log // entry, and made the drop rules above a no-op on the second pass since the // __meta_docker_* labels are already gone from the pre-relabelled output. loki.source.docker "containers" { host = "unix:///var/run/docker.sock" targets = discovery.docker.containers.targets forward_to = [loki.write.central.receiver] relabel_rules = discovery.relabel.containers.rules labels = { job = "docker" } } // --- 2. Caddy host access logs --------------------------------------------------- // sync_period (glob rescan) defaults to 10s; 1m is plenty for a log file that only // appears/rotates on the order of hours. local.file_match "caddy" { path_targets = [{ __path__ = "/var/log/caddy/*.log", job = "caddy" }] sync_period = "1m" } loki.source.file "caddy" { targets = local.file_match.caddy.targets forward_to = [loki.process.caddy.receiver] // PollingFileWatcher defaults (250ms/250ms) stat every tailed file 4x/second // forever. Caddy's access log doesn't need sub-second latency into Loki. file_watch { min_poll_frequency = "2s" max_poll_frequency = "10s" } } // Drop well-known crawler/bot traffic before it hits Loki. Caddy itself can't do // this cheaply (log_skip needs Caddy >= 2.7; both hosts run older Caddy, and an // upgrade is out of scope here), so filter it at the shipper instead. loki.process "caddy" { forward_to = [loki.write.central.receiver] stage.drop { expression = "(?i)(semrushbot|claudebot|ahrefsbot|yandexbot|bytespider|mj12bot|petalbot)" drop_counter_reason = "crawler" } } // --- 3. App structured JSON logs (errors / access / audit) ----------------------- // Mounted read-only from the app's `applogs` volume at /applogs (see the agent // compose). Lift `level`/`tag`/`phase` out of the JSON so they're queryable. local.file_match "app_jsonl" { path_targets = [{ __path__ = "/applogs/**/*.jsonl", job = "app-json" }] sync_period = "1m" } loki.source.file "app_jsonl" { targets = local.file_match.app_jsonl.targets forward_to = [loki.process.app_jsonl.receiver] file_watch { min_poll_frequency = "2s" max_poll_frequency = "10s" } } loki.process "app_jsonl" { forward_to = [loki.write.central.receiver] stage.json { expressions = { level = "level", tag = "tag", phase = "phase", status = "status" } } // A record with no explicit level: an error-folder line is an error, else info. stage.static_labels { values = { source = "app" } } stage.labels { values = { level = "", tag = "", phase = "" } } } // --- Ship to the central Loki over the WireGuard mesh ---------------------------- loki.write "central" { endpoint { url = sys.env("LOKI_URL") } // Every line from this node is stamped with its node name, so one Grafana // view can slice prod vs beta vs dev. external_labels = { host = sys.env("ALLOY_NODE") } }