Files
homelab/prometheus-stack/k8s/traefik-alerts.yaml
T
forust a564dd8b67
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 17s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 12s
ci / build (push) Successful in 10s
fix(alerts): exclude netbird streams from traefik latency alert
SignalExchange ConnectStream holds 60s gRPC streams by design; at night they exceed 5% of samples and pin P95 to the 5.0s bucket ceiling, flapping the alert. Also fixes the stale xui exclusion pattern, which matched no real service label.
2026-09-29 08:12:07 +02:00

61 lines
2.4 KiB
YAML

apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: traefik
namespace: prometheus
labels:
release: prometheus-stack
spec:
groups:
- name: traefik
rules:
- alert: TraefikDown
expr: absent(up{job="traefik"})
for: 10m
labels:
severity: critical
annotations:
summary: "Traefik metrics are unavailable"
description: "Prometheus has no Traefik target."
- alert: TraefikConfigReloadFailed
expr: traefik_config_last_reload_success == 0
for: 5m
labels:
severity: warning
annotations:
summary: "Traefik configuration reload failed"
description: "Traefik failed to apply its last configuration reload. Check Traefik logs and the dynamic config sources."
- alert: TraefikServiceHigh5xxRate
expr: |
sum(rate(traefik_service_requests_total{code=~"5.."}[5m])) by (service)
/ sum(rate(traefik_service_requests_total[5m])) by (service) * 100 > 5
and sum(rate(traefik_service_requests_total[5m])) by (service) > 0
for: 5m
labels:
severity: warning
annotations:
summary: "High 5xx error rate for service {{ $labels.service }}"
description: "Service {{ $labels.service }} is returning 5xx errors for more than 5% of requests over the last 5 minutes (current: {{ $value | humanizePercentage }})."
- alert: TraefikServiceHighLatency
expr: |
histogram_quantile(0.95,
sum(rate(traefik_service_request_duration_seconds_bucket{service!~"xui-service|netbird-server-service"}[5m])) by (le, service)) > 2
for: 5m
labels:
severity: warning
annotations:
summary: "High latency for service {{ $labels.service }}"
description: "P95 latency of {{ $labels.service }} exceeded 2 seconds over the last 5 minutes (current: {{ $value | humanizeDuration }})."
- alert: TraefikCertExpiringSoon
expr: min(traefik_tls_certs_not_after) - time() < 14 * 24 * 60 * 60
for: 10m
labels:
severity: warning
annotations:
summary: "Traefik TLS certificate expires soon"
description: "A TLS certificate managed by Traefik expires in less than 14 days (in {{ $value | humanizeDuration }})."