From b6e1dc0362bd3d9926d239d5d46cc95b960deba9 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 29 Sep 2026 12:36:04 +0200 Subject: [PATCH] fix(immich): size postgres probes for HDD stalls Postmaster was SIGKILLed in a loop: 70s fsync stalls on the loaded rotational disk outlasted the 5-minute startup budget and the 60s liveness tolerance, and every kill bought another full WAL replay. Startup budget 15min, liveness 5x60s. Already applied live with kubectl; this keeps git in sync. --- immich/k8s/postgres.yaml | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/immich/k8s/postgres.yaml b/immich/k8s/postgres.yaml index fe8d603..6075238 100644 --- a/immich/k8s/postgres.yaml +++ b/immich/k8s/postgres.yaml @@ -90,10 +90,16 @@ spec: # rotational disk on a loaded single node, and pg_isready can take # seconds during WAL recovery. A 1s timeout kills the container # mid-recovery and restarts the spiral. + # + # Budgets are sized for HDD stalls, not for a healthy disk: fsync of + # a single file was observed taking 70s under node IO pressure, so + # the startup budget is 15 minutes and liveness tolerates 5 minutes + # of unresponsiveness. Killing a stalled-but-healthy postmaster only + # buys another full WAL replay, which is more IO, not less. startupProbe: exec: command: ["sh", "-c", "pg_isready -U immich -d immich"] - failureThreshold: 60 + failureThreshold: 180 periodSeconds: 5 timeoutSeconds: 5 readinessProbe: @@ -105,8 +111,9 @@ spec: exec: command: ["sh", "-c", "pg_isready -U immich -d immich"] initialDelaySeconds: 30 - periodSeconds: 20 - timeoutSeconds: 5 + periodSeconds: 60 + timeoutSeconds: 10 + failureThreshold: 5 # The image template sets shared_buffers to 512MB, and the vchord and # vectors workers are Rust binaries with a real RSS footprint on top # of postmaster, checkpointer and friends. 1Gi was enough to start