chore(deploy): rework k8s pipeline, monitoring and postgres 17
Deploy workflow uses git-tracked manifests, DISABLED flag and kustomize overlays; add webinar-checker metrics with ServiceMonitor and alerts; upgrade shared postgres to 17 with statuspage DB and probes/resources.
This commit is contained in:
1 parent
8b2cf29771
commit
46c7e99b1d
19 files changed
+629
-227
No files matched your search
@@ -0,0 +1,77 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: edu-master-webinar
|
||||
namespace: edu-master
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
spec:
|
||||
groups:
|
||||
- name: edu_master.webinar
|
||||
rules:
|
||||
# No successful webinar check for 5m (~2-3 missed 2-min checks).
|
||||
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
|
||||
- alert: WebinarCheckerNoSuccessfulCheck
|
||||
expr: |
|
||||
(time() - webinar_check_last_success_timestamp_seconds > 300)
|
||||
and (webinar_check_last_run_timestamp_seconds > 0)
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker has no successful check for 5m"
|
||||
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
|
||||
|
||||
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
|
||||
- alert: WebinarCheckerConsecutiveFailures
|
||||
expr: |
|
||||
webinar_check_consecutive_failures >= 3
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker failing consecutively"
|
||||
description: "edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace=\"edu-master\", container=\"webinar-checker\"})."
|
||||
|
||||
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
|
||||
- alert: WebinarCheckerScrapeDown
|
||||
expr: |
|
||||
absent(webinar_check_last_run_timestamp_seconds) == 1
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker metrics missing"
|
||||
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
|
||||
|
||||
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
|
||||
- alert: EduPhpsessidMissing
|
||||
expr: |
|
||||
edu_phpsessid_present == 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "EDU_PHPSESSID missing"
|
||||
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
|
||||
|
||||
# Hard deps: checker and playwright deployments unavailable.
|
||||
- alert: WebinarCheckerDeploymentDown
|
||||
expr: |
|
||||
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker deployment unavailable"
|
||||
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
|
||||
|
||||
- alert: PlaywrightServiceDown
|
||||
expr: |
|
||||
kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Playwright service unavailable"
|
||||
description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it."
|
||||
Reference in new issue
Block a user