From a90fb19ed4eff8f74b54b0c095bf1aba5b959fa1 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 29 Sep 2026 14:47:53 +0200 Subject: [PATCH] fix(traefik): size probes for HDD stalls Single replica is the whole ingress; liveness kills at 2s timeouts took every public service down in a loop. Same stall-sized budgets as postgres/metallb. Applied live via helm (pinned 41.5.0, values from git). --- traefik/k8s/traefik-values.yaml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/traefik/k8s/traefik-values.yaml b/traefik/k8s/traefik-values.yaml index d4f02af..c5baaf4 100644 --- a/traefik/k8s/traefik-values.yaml +++ b/traefik/k8s/traefik-values.yaml @@ -46,6 +46,22 @@ resources: cpu: "1000m" memory: "1Gi" +# Single replica is the whole ingress: a SIGKILL here takes every public +# service down. Budgets are sized for HDD stalls on a loaded node, not for a +# healthy disk — same treatment as immich postgres and metallb speaker. +readinessProbe: + failureThreshold: 6 + initialDelaySeconds: 10 + periodSeconds: 10 + successThreshold: 1 + timeoutSeconds: 5 +livenessProbe: + failureThreshold: 6 + initialDelaySeconds: 30 + periodSeconds: 30 + successThreshold: 1 + timeoutSeconds: 5 + providers: kubernetesIngress: enabled: true