Files
homelab/prometheus-stack/k8s/alerts.yaml
T
forust 74adf38d63 feat(alerts): cover OOM kills, restart loops and evictions
The OOMKilled container behind the immich crash loop was invisible: PodCrashLooping only fires once kubelet has already given up and started the backoff.
2026-09-28 17:01:36 +02:00

114 lines
5.0 KiB
YAML

apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: homelab-infrastructure
namespace: prometheus
labels:
release: prometheus-stack
spec:
groups:
- name: homelab.infrastructure
rules:
- alert: TargetDown
expr: up == 0
for: 10m
labels:
severity: warning
annotations:
summary: "Prometheus target is down"
description: "{{ $labels.job }} target {{ $labels.instance }} has been down for more than 10 minutes."
- alert: PodCrashLooping
expr: max_over_time(kube_pod_container_status_waiting_reason{reason="CrashLoopBackOff"}[10m]) >= 1
for: 10m
labels:
severity: warning
annotations:
summary: "Pod is crash looping"
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} is in CrashLoopBackOff."
- alert: ContainerOOMKilled
expr: max_over_time(kube_pod_container_status_terminated_reason{reason="OOMKilled"}[15m]) >= 1
for: 5m
labels:
severity: warning
annotations:
summary: "Container was OOMKilled"
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} was killed for exceeding its memory limit. Raise the limit or reduce the workload."
- alert: ContainerRestartingTooOften
expr: max by (namespace, pod, container) (increase(kube_pod_container_status_restarts_total[30m])) > 3
for: 5m
labels:
severity: warning
annotations:
summary: "Container restarting too often"
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} restarted {{ $value }} times in the last 30 minutes."
- alert: PodEvicted
expr: max_over_time(kube_pod_status_reason{reason="Evicted"}[15m]) >= 1
for: 5m
labels:
severity: warning
annotations:
summary: "Pod was evicted"
description: "Pod {{ $labels.namespace }}/{{ $labels.pod }} was evicted, usually for node disk or memory pressure."
- alert: PersistentVolumeClaimFillingUp
expr: kubelet_volume_stats_available_bytes / kubelet_volume_stats_capacity_bytes < 0.15
for: 15m
labels:
severity: warning
annotations:
summary: "PVC has less than 15% free space"
description: "{{ $labels.namespace }}/{{ $labels.persistentvolumeclaim }} has less than 15% free space."
- alert: CPUThrottlingHigh
expr: |
sum without (id, metrics_path, name, image, endpoint, job, node) (
topk by (cluster, namespace, pod, container, instance) (1,
increase(container_cpu_cfs_throttled_periods_total{container!="", job="kubelet", metrics_path="/metrics/cadvisor", }[5m])
)
)
/ on (cluster, namespace, pod, container, instance) group_left
sum without (id, metrics_path, name, image, endpoint, job, node) (
topk by (cluster, namespace, pod, container, instance) (1,
increase(container_cpu_cfs_periods_total{job="kubelet", metrics_path="/metrics/cadvisor", }[5m])
)
)
> ( 50 / 100 )
for: 30m
labels:
severity: info
annotations:
summary: "Processes experience elevated CPU throttling"
description: "{{ $value | humanizePercentage }} throttling of CPU in namespace {{ $labels.namespace }} for container {{ $labels.container }} in pod {{ $labels.pod }} on cluster {{ $labels.cluster }}."
runbook_url: "https://runbooks.prometheus-operator.dev/runbooks/kubernetes/cputhrottlinghigh"
- alert: NodeMemoryPressure
expr: 100 * (1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) > 90
for: 15m
labels:
severity: warning
annotations:
summary: "Node memory pressure"
description: "Node {{ $labels.instance }} has used more than 90% of memory for 15 minutes."
- alert: LokiDown
expr: kube_statefulset_status_replicas_unavailable{statefulset="loki"} > 0 or kube_deployment_status_replicas_unavailable{deployment="loki-gateway"} > 0
for: 10m
labels:
severity: warning
annotations:
summary: "Loki is down"
description: "Loki in namespace {{ $labels.namespace }} has unavailable replicas for more than 10 minutes. Logs are not queryable."
- alert: AlloyDown
expr: kube_daemonset_status_number_unavailable{daemonset="alloy"} > 0
for: 10m
labels:
severity: warning
annotations:
summary: "Alloy is down"
description: "Alloy DaemonSet in namespace {{ $labels.namespace }} has {{ $value }} unavailable pods for more than 10 minutes. Pod logs are not being shipped to Loki."