forgejo: restart the db service on a clean exit, not just on failure #128

Merged
admin_emi merged 1 commit from forgejo-db-restart-policy into dev 2026-07-30 18:09:21 +00:00
Showing only changes of commit 8a2c838663 - Show all commits

View file

@ -53,7 +53,19 @@ services:
cpus: "${FORGEJO_DB_CPUS:-1}"
memory: ${FORGEJO_DB_MEMORY:-1g}
restart_policy:
condition: on-failure
# `any`, NOT `on-failure`. This took Forgejo down for 27 hours on 2026-07-29:
# Postgres hit an invalid data-directory lock file ("could not open file
# postmaster.pid ... performing immediate shutdown") and exited **0**. A clean
# exit is not a failure, so Swarm considered the task Complete, dropped the
# service to 0/1 replicas, and never rescheduled it. Forgejo itself stayed Up
# and served its homepage while every repo page, the API and all CI returned
# 500 with `dial tcp: lookup db ... no such host`.
#
# `on-failure` is the wrong policy for any always-on stateful service: it
# cannot distinguish "finished successfully" from "shut itself down and should
# be restarted", and Postgres does the latter with status 0 on several paths.
condition: any
delay: 5s
forgejo:
image: codeberg.org/forgejo/forgejo:9-rootless