grafana: grafana.ini: server: domain: grafana.forust.xyz root_url: https://grafana.forust.xyz auth.anonymous: enabled: false users: allow_sign_up: false admin: existingSecret: grafana-admin userKey: admin-user passwordKey: admin-password persistence: enabled: true # Matches live PVC (10Gi/local-path). Retain migration is a separate task # with data migration (see storage-audit doc); do NOT change SC/size here # without migrating, helm upgrade fails on immutable PVC fields. storageClassName: local-path size: 10Gi ingress: enabled: false service: port: 80 # p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the # scheduler sees reality; the limit is a manual exception above the 1.3x max # formula, because a single query burst reached 1046M. resources: requests: memory: "416Mi" cpu: 100m limits: memory: "1536Mi" # One block covers both the dashboards and datasources sidecars (p95 91M / 80M). sidecar: resources: requests: memory: "96Mi" cpu: 10m limits: memory: "192Mi" additionalDataSources: - name: Loki type: loki url: http://loki-gateway.prometheus.svc.cluster.local access: proxy prometheus: prometheusSpec: retention: 60d retentionSize: 32GB storageSpec: volumeClaimTemplate: spec: # Matches live PVC, see note on grafana.persistence above. storageClassName: "local-path" accessModes: - ReadWriteOnce resources: requests: storage: 40Gi resources: requests: memory: "768Mi" cpu: 200m limits: memory: "2560Mi" alertmanager: alertmanagerSpec: configSecret: alertmanager-config # p95 66M, max 68M. Silences and notification state live here, so the request # stays above p95 to keep the pod out of the eviction candidates. resources: requests: memory: "96Mi" cpu: 10m limits: memory: "192Mi" storage: volumeClaimTemplate: spec: # Matches live PVC (20Gi), see note on grafana.persistence above. storageClassName: local-path accessModes: - ReadWriteOnce resources: requests: storage: 20Gi # p95 103M, max 107M, and it grows with the number of cluster objects. kube-state-metrics: # Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics` # condition key. Setting resources under `kubeStateMetrics:` is silently ignored. resources: requests: memory: "128Mi" cpu: 50m limits: memory: "256Mi" # p95 39M, max 40M. One per node, so it scales with node count. prometheus-node-exporter: resources: requests: memory: "32Mi" cpu: 20m limits: memory: "128Mi" # p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster. prometheusOperator: resources: requests: memory: "96Mi" cpu: 50m limits: memory: "192Mi" # config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not # template `prometheusSpec.configReloader.resources`, so there is no values key # for them. They keep shipping with requests only. defaultRules: disabled: CPUThrottlingHigh: true KubeControllerManagerDown: true KubeSchedulerDown: true KubeEtcdDown: true KubeEtcdHighCommitDurations: true # k0s runs controller-manager/scheduler/etcd internally, not as pods with # component=kube-controller-manager/kube-scheduler/k8s-app=kube-etcd labels. # Their Services get no endpoints, so the targets are permanently down. # kube-proxy and kubelet have endpoints on k0s, keep them enabled. kubeControllerManager: enabled: false kubeScheduler: enabled: false kubeEtcd: enabled: false