From 16aaeb60c1bb50233559d5c7f40b58fa7eddca7a Mon Sep 17 00:00:00 2001 From: mr-forust Date: Mon, 28 Sep 2026 10:14:52 +0200 Subject: [PATCH] fix(k8s): set requests and limits on the pods that shipped with neither Eighteen containers had no memory limit at all, so nothing on the node could bound them. Three of the values files even claimed to set resources: Helm does not complain about a key it does not recognise, so the block sat there looking like a limit while the pod ran unbounded. alloy is the one that mattered. The chart reads `alloy.resources`; the file had `controller.resources`, so the DaemonSet that tails every pod log on the node shipped with nothing at all. `kubeStateMetrics` is the same trap in a different shape -- that is the condition key, the values live under `kube-state-metrics` -- and `configReloader` in the alloy chart sits at the top level rather than under `alloy`. Each one is verified by rendering the chart and reading the resources back off the containers, because a values key that is ignored looks exactly like one that works. reloader turned out to be set and still wrong: 64Mi request against a measured p95 of 73M, so the pod ran permanently above its own request and stayed a standing eviction candidate. That is the pod that restarts every other pod, so it is the last one that should be evicted. Raised to 96Mi. Requests are set at p95 throughout, grafana, playwright and alloy included. Left at the values first proposed they would have sat below their own p95 and queued for eviction ahead of everything smaller. CPU limits are deliberately absent: the node is I/O bound at 5% CPU, and CFS throttling would turn disk wait into runnable-throttled, which is the failure mode that took the node down. The prometheus and alertmanager configReloader sidecars are left open: chart 86.2.3 does not template the key, so reaching those two containers needs a postRenderer. Verified: all four charts render with the resources landing on the intended containers, and 16/16 local gates pass. --- edu_master/k8s/playwright.yaml | 9 ++++ errorpages/k8s/error-pages.yaml | 7 +++ loki/k8s/alloy-values.yaml | 22 +++++++-- loki/k8s/loki-values.yaml | 10 ++++ prometheus-stack/k8s/grafana-values.yaml | 60 ++++++++++++++++++++++++ reloader/k8s/reloader-values.yaml | 7 ++- searxng/k8s/valkey.yaml | 7 +++ 7 files changed, 116 insertions(+), 6 deletions(-) diff --git a/edu_master/k8s/playwright.yaml b/edu_master/k8s/playwright.yaml index 6d0a4e3..ebf71d8 100644 --- a/edu_master/k8s/playwright.yaml +++ b/edu_master/k8s/playwright.yaml @@ -20,6 +20,15 @@ spec: # renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker image: mcr.microsoft.com/playwright:v1.56.0-jammy imagePullPolicy: IfNotPresent + # p95 412M, max 478M over 7 days, no limit before. Request is set at p95 + # so the pod is not an eviction candidate; the limit stays above 2x the + # request because browser page lifetimes are unpredictable. + resources: + requests: + cpu: "200m" + memory: "416Mi" + limits: + memory: "1Gi" command: - npx - -y diff --git a/errorpages/k8s/error-pages.yaml b/errorpages/k8s/error-pages.yaml index 0f7b993..4d1845d 100644 --- a/errorpages/k8s/error-pages.yaml +++ b/errorpages/k8s/error-pages.yaml @@ -28,6 +28,13 @@ spec: containers: - name: error-pages image: gcr.forust.xyz/forust/error-pages:prod + # p95 6M, max 10M, no limit before. + resources: + requests: + cpu: "10m" + memory: "32Mi" + limits: + memory: "128Mi" ports: - containerPort: 80 readinessProbe: diff --git a/loki/k8s/alloy-values.yaml b/loki/k8s/alloy-values.yaml index 2f88adc..7e8307b 100644 --- a/loki/k8s/alloy-values.yaml +++ b/loki/k8s/alloy-values.yaml @@ -7,18 +7,32 @@ controller: type: daemonset + +# config-reloader sidecar: p95 33M, max 43M. The chart keeps it at the top level, +# not under `alloy:`. +configReloader: resources: requests: - memory: "128Mi" - cpu: "50m" + memory: "32Mi" + cpu: "10m" limits: - memory: "512Mi" - cpu: "500m" + memory: "128Mi" image: tag: "v1.19.2" alloy: + # p95 275M, max 287M. Alloy tails every pod log and ships it to Loki, so it sits + # on the same IronWolf read path the node is I/O bound on. Request is set at p95. + # The chart key is `alloy.resources`. `controller.resources` is ignored silently, + # which is why this pod shipped with no limits at all. + resources: + requests: + memory: "288Mi" + cpu: "50m" + limits: + memory: "512Mi" + configMap: create: true content: | diff --git a/loki/k8s/loki-values.yaml b/loki/k8s/loki-values.yaml index d22c4f3..7b5a058 100644 --- a/loki/k8s/loki-values.yaml +++ b/loki/k8s/loki-values.yaml @@ -39,6 +39,16 @@ loki: local: directory: /var/loki/rules +# p95 84M, max 85M for the rules sidecar that shares the singleBinary pod. +# The chart exposes it as `sidecar.resources`, shared with any other sidecar. +sidecar: + resources: + requests: + memory: "96Mi" + cpu: "10m" + limits: + memory: "192Mi" + singleBinary: replicas: 1 persistence: diff --git a/prometheus-stack/k8s/grafana-values.yaml b/prometheus-stack/k8s/grafana-values.yaml index 8feddb9..b7a00e8 100644 --- a/prometheus-stack/k8s/grafana-values.yaml +++ b/prometheus-stack/k8s/grafana-values.yaml @@ -26,6 +26,25 @@ grafana: service: port: 80 + # p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the + # scheduler sees reality; the limit is a manual exception above the 1.3x max + # formula, because a single query burst reached 1046M. + resources: + requests: + memory: "416Mi" + cpu: 100m + limits: + memory: "1536Mi" + + # One block covers both the dashboards and datasources sidecars (p95 91M / 80M). + sidecar: + resources: + requests: + memory: "96Mi" + cpu: 10m + limits: + memory: "192Mi" + additionalDataSources: - name: Loki type: loki @@ -55,6 +74,14 @@ prometheus: alertmanager: alertmanagerSpec: configSecret: alertmanager-config + # p95 66M, max 68M. Silences and notification state live here, so the request + # stays above p95 to keep the pod out of the eviction candidates. + resources: + requests: + memory: "96Mi" + cpu: 10m + limits: + memory: "192Mi" storage: volumeClaimTemplate: spec: @@ -66,6 +93,39 @@ alertmanager: requests: storage: 20Gi +# p95 103M, max 107M, and it grows with the number of cluster objects. +kube-state-metrics: + # Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics` + # condition key. Setting resources under `kubeStateMetrics:` is silently ignored. + resources: + requests: + memory: "128Mi" + cpu: 50m + limits: + memory: "256Mi" + +# p95 39M, max 40M. One per node, so it scales with node count. +prometheus-node-exporter: + resources: + requests: + memory: "32Mi" + cpu: 20m + limits: + memory: "128Mi" + +# p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster. +prometheusOperator: + resources: + requests: + memory: "96Mi" + cpu: 50m + limits: + memory: "192Mi" + +# config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not +# template `prometheusSpec.configReloader.resources`, so there is no values key +# for them. They keep shipping with requests only. + defaultRules: disabled: CPUThrottlingHigh: true diff --git a/reloader/k8s/reloader-values.yaml b/reloader/k8s/reloader-values.yaml index 656e939..f26caea 100644 --- a/reloader/k8s/reloader-values.yaml +++ b/reloader/k8s/reloader-values.yaml @@ -11,10 +11,13 @@ reloader: replicas: 1 # The chart defaults to no requests or limits, so the pod is evictable under node # pressure and the restarts go with it. + # Memory was raised from 64Mi: measured p95 over 7 days is 73M, so the pod was + # running above its own request and sitting in the eviction candidates. This pod + # is the one that restarts every other pod, so it must not be evicted. resources: requests: cpu: "10m" - memory: "64Mi" + memory: "96Mi" limits: cpu: "100m" - memory: "128Mi" + memory: "192Mi" diff --git a/searxng/k8s/valkey.yaml b/searxng/k8s/valkey.yaml index 3ec5b9d..08f9933 100644 --- a/searxng/k8s/valkey.yaml +++ b/searxng/k8s/valkey.yaml @@ -30,6 +30,13 @@ spec: containers: - name: valkey image: docker.io/valkey/valkey:9.1.2-alpine + # p95 15M, max 17M, no limit before. Matches the netbox valkey pod. + resources: + requests: + cpu: "50m" + memory: "64Mi" + limits: + memory: "256Mi" command: - valkey-server - --save