Files
homelab/prometheus-stack/k8s/grafana-values.yaml
T
forust 7fdeacffb8
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 13s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 10s
ci / lint-prettier (push) Successful in 14s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 7s
ci / build (push) Successful in 19s
feat(monitoring): stand down Prometheus server during VM trial
vmagent scrapes and remote-writes to VictoriaMetrics, so the Prometheus server scales to 0. Encoded as prometheusSpec.replicas in values instead of a kubectl patch, so helm keeps owning spec.replicas and Helm 4 server-side apply stops conflicting with the kubectl-patch field manager.
2026-10-06 19:29:30 +02:00

158 lines
4.3 KiB
YAML

grafana:
grafana.ini:
server:
domain: grafana.forust.xyz
root_url: https://grafana.forust.xyz
auth.anonymous:
enabled: false
users:
allow_sign_up: false
admin:
existingSecret: grafana-admin
userKey: admin-user
passwordKey: admin-password
persistence:
enabled: true
# Matches live PVC (10Gi/local-path). Retain migration is a separate task
# with data migration (see storage-audit doc); do NOT change SC/size here
# without migrating, helm upgrade fails on immutable PVC fields.
storageClassName: local-path
size: 10Gi
ingress:
enabled: false
service:
port: 80
# p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the
# scheduler sees reality; the limit is a manual exception above the 1.3x max
# formula, because a single query burst reached 1046M.
resources:
requests:
memory: "416Mi"
cpu: 100m
limits:
memory: "1536Mi"
# One block covers both the dashboards and datasources sidecars (p95 91M / 80M).
sidecar:
datasources:
defaultDatasourceEnabled: false
resources:
requests:
memory: "96Mi"
cpu: 10m
limits:
memory: "192Mi"
additionalDataSources:
- name: Loki
type: loki
url: http://loki-gateway.prometheus.svc.cluster.local
access: proxy
- name: VictoriaMetrics
type: prometheus
url: http://victoria-metrics.prometheus.svc.cluster.local:8428
access: proxy
isDefault: true
prometheus:
prometheusSpec:
# VM trial: vmagent scrapes and remote-writes to VictoriaMetrics, so the
# Prometheus server itself stands down. Encoded here (not a kubectl patch)
# so helm keeps owning spec.replicas and upgrades do not conflict on it.
replicas: 0
retention: 60d
retentionSize: 32GB
storageSpec:
volumeClaimTemplate:
spec:
# Matches live PVC, see note on grafana.persistence above.
storageClassName: "local-path"
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 40Gi
resources:
requests:
memory: "768Mi"
cpu: 200m
limits:
memory: "2560Mi"
alertmanager:
alertmanagerSpec:
configSecret: alertmanager-config
# p95 66M, max 68M. Silences and notification state live here, so the request
# stays above p95 to keep the pod out of the eviction candidates.
resources:
requests:
memory: "96Mi"
cpu: 10m
limits:
memory: "192Mi"
storage:
volumeClaimTemplate:
spec:
# Matches live PVC (20Gi), see note on grafana.persistence above.
storageClassName: local-path
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 20Gi
# p95 103M, max 107M, and it grows with the number of cluster objects.
kube-state-metrics:
# Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics`
# condition key. Setting resources under `kubeStateMetrics:` is silently ignored.
resources:
requests:
memory: "128Mi"
cpu: 50m
limits:
memory: "256Mi"
# p95 39M, max 40M. One per node, so it scales with node count.
prometheus-node-exporter:
resources:
requests:
memory: "32Mi"
cpu: 20m
limits:
memory: "128Mi"
# p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster.
prometheusOperator:
resources:
requests:
memory: "96Mi"
cpu: 50m
limits:
memory: "192Mi"
# config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not
# template `prometheusSpec.configReloader.resources`, so there is no values key
# for them. They keep shipping with requests only.
defaultRules:
disabled:
CPUThrottlingHigh: true
KubeControllerManagerDown: true
KubeSchedulerDown: true
KubeEtcdDown: true
KubeEtcdHighCommitDurations: true
# k0s runs controller-manager/scheduler/etcd internally, not as pods with
# component=kube-controller-manager/kube-scheduler/k8s-app=kube-etcd labels.
# Their Services get no endpoints, so the targets are permanently down.
# kube-proxy and kubelet have endpoints on k0s, keep them enabled.
kubeControllerManager:
enabled: false
kubeScheduler:
enabled: false
kubeEtcd:
enabled: false