feat(prometheus): add alerting rules and alertmanager config
This commit is contained in:
1 parent
9a806724af
commit
360a6fc5dc
4 files changed
+178
-4
No files matched your search
@@ -0,0 +1,68 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: homelab-infrastructure
|
||||
namespace: prometheus
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
spec:
|
||||
groups:
|
||||
- name: homelab.infrastructure
|
||||
rules:
|
||||
- alert: TargetDown
|
||||
expr: up == 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Prometheus target is down"
|
||||
description: "{{ $labels.job }} target {{ $labels.instance }} has been down for more than 10 minutes."
|
||||
|
||||
- alert: PodCrashLooping
|
||||
expr: max_over_time(kube_pod_container_status_waiting_reason{reason="CrashLoopBackOff"}[10m]) >= 1
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Pod is crash looping"
|
||||
description: "Container {{ $labels.container }} in {{ $labels.namespace }}/{{ $labels.pod }} is in CrashLoopBackOff."
|
||||
|
||||
- alert: PersistentVolumeClaimFillingUp
|
||||
expr: kubelet_volume_stats_available_bytes / kubelet_volume_stats_capacity_bytes < 0.15
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "PVC has less than 15% free space"
|
||||
description: "{{ $labels.namespace }}/{{ $labels.persistentvolumeclaim }} has less than 15% free space."
|
||||
|
||||
- alert: CPUThrottlingHigh
|
||||
expr: |
|
||||
sum without (id, metrics_path, name, image, endpoint, job, node) (
|
||||
topk by (cluster, namespace, pod, container, instance) (1,
|
||||
increase(container_cpu_cfs_throttled_periods_total{container!="", job="kubelet", metrics_path="/metrics/cadvisor", }[5m])
|
||||
)
|
||||
)
|
||||
/ on (cluster, namespace, pod, container, instance) group_left
|
||||
sum without (id, metrics_path, name, image, endpoint, job, node) (
|
||||
topk by (cluster, namespace, pod, container, instance) (1,
|
||||
increase(container_cpu_cfs_periods_total{job="kubelet", metrics_path="/metrics/cadvisor", }[5m])
|
||||
)
|
||||
)
|
||||
> ( 50 / 100 )
|
||||
for: 30m
|
||||
labels:
|
||||
severity: info
|
||||
annotations:
|
||||
summary: "Processes experience elevated CPU throttling"
|
||||
description: "{{ $value | humanizePercentage }} throttling of CPU in namespace {{ $labels.namespace }} for container {{ $labels.container }} in pod {{ $labels.pod }} on cluster {{ $labels.cluster }}."
|
||||
runbook_url: "https://runbooks.prometheus-operator.dev/runbooks/kubernetes/cputhrottlinghigh"
|
||||
|
||||
- alert: NodeMemoryPressure
|
||||
expr: 100 * (1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes) > 90
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "Node memory pressure"
|
||||
description: "Node {{ $labels.instance }} has used more than 90% of memory for 15 minutes."
|
||||
Reference in new issue
Block a user