From 46c7e99b1da23bd63613729f44aa27085300abf3 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 22 Sep 2026 23:40:56 +0200 Subject: [PATCH] chore(deploy): rework k8s pipeline, monitoring and postgres 17 Deploy workflow uses git-tracked manifests, DISABLED flag and kustomize overlays; add webinar-checker metrics with ServiceMonitor and alerts; upgrade shared postgres to 17 with statuspage DB and probes/resources. --- .gitea/workflows/ci.yaml | 20 -- .gitea/workflows/deploy.yaml | 214 ++++++++++--- edu_master/PLAYWRIGHT_VERSION | 1 + edu_master/compose.yaml | 2 +- edu_master/k8s/alerts.yaml | 77 +++++ edu_master/k8s/playwright.yaml | 3 +- edu_master/k8s/secrets.yaml.example | 2 + edu_master/k8s/service.yaml | 15 + edu_master/k8s/servicemonitor.yaml | 16 + edu_master/k8s/webinar-checker.yaml | 4 + edu_master/webinar-checker/Dockerfile | 7 +- edu_master/webinar-checker/checker.py | 417 ++++++++++++++++--------- gitea/k8s/gitea.yaml | 2 +- postgres/.env.example | 1 + postgres/README.md | 27 +- postgres/initdb/01-create-databases.sh | 3 +- postgres/k8s/postgres.yaml | 12 + postgres/shared-compose.yaml | 2 +- renovate.json | 31 +- 19 files changed, 629 insertions(+), 227 deletions(-) create mode 100644 edu_master/PLAYWRIGHT_VERSION create mode 100644 edu_master/k8s/alerts.yaml create mode 100644 edu_master/k8s/service.yaml create mode 100644 edu_master/k8s/servicemonitor.yaml diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index d017a83..3b3dac1 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -347,23 +347,3 @@ jobs: ;; esac done - - deploy-userbot-panel: - needs: build - if: github.ref_name == 'main' && contains(needs.build.outputs.services, 'userbot') - runs-on: [self-hosted, linux, arch, homelab, prod] - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Apply and roll out userbot panel - shell: bash - run: | - kubectl apply -f userbot/k8s/base/panel.yaml - kubectl get secret userbot-common-secrets -n default -o json \ - | jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \ - | kubectl apply -f - - # Keep legacy deployments (forust/anna) in sync with manifests; they have no replicas field, so apply leaves scaling to the user manager only. - kubectl apply -f userbot/k8s/base/userbots.yaml - kubectl rollout restart deployment/userbot-panel -n userbot - kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 35f0e6f..b19c8d0 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -19,9 +19,6 @@ jobs: DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} - # Set APPLY_PRUNE=true to enable kubectl apply --prune. Requires every - # manifest to carry label app.kubernetes.io/managed-by=homelab-deploy, - # otherwise previously applied resources get deleted on the next run. APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} run: | set -euo pipefail @@ -57,59 +54,170 @@ jobs: fi git -C "$repo" fetch origin main - git -C "$repo" reset --hard origin/main - # Runtime selection: a service is k8s-managed when $SERVICE/k8s/active - # exists. Otherwise it is compose-managed, and only k8s/routing/* - # manifests (external Services / EndpointSlices / ServersTransport / - # Ingresses that route to docker backends) are applied. - # migrate: touch SERVICE/k8s/active (+ move routing files up) - # rollback: rm SERVICE/k8s/active + echo "== Workstation state ==" + echo " local: $(git -C "$repo" rev-parse --short HEAD)" + echo " remote: $(git -C "$repo" rev-parse --short origin/main)" + + if [ -n "$(git -C "$repo" status --porcelain --untracked-files=no)" ]; then + echo "ERROR: workstation has local tracked modifications, refusing reset:" + git -C "$repo" status --porcelain --untracked-files=no + git -C "$repo" diff --stat + exit 1 + fi + + git -C "$repo" reset --hard origin/main + cd "$repo" + + is_disabled() { + local target="$1" + if [ -f "$target" ]; then + target="$(dirname "$target")" + fi + while true; do + if [ -f "$target/DISABLED" ]; then + return 0 + fi + if [ "$target" = "$repo" ]; then + break + fi + target="$(dirname "$target")" + case "$target" in + "$repo"/*) ;; + *) break ;; + esac + done + return 1 + } + collect_k8s() { - find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \ - ! -path '*/routing/*' ! -path '*/overlays/*' \ - ! -name 'kustomization.y*ml' ! -name '*.example.y*ml' \ - ! -name '*values.y*ml' ! -name 'patch-*.y*ml' \ + git ls-files -- "$1" \ + | grep -E '\.ya?ml$' \ + | grep -Ev '/routing/|/overlays/' \ + | grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \ + | grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \ | sort } collect_k8s_inactive() { - find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \ - \( -name 'namespace.y*ml' -o -path '*/routing/*' \) \ - ! -path '*/overlays/*' ! -name '*.example.y*ml' \ - | sort + collect_k8s "$1" \ + | grep -E '(^|/)namespace\.ya?ml$|/routing/' } - mapfile -t compose_stacks < <( - find "$repo" -type f \( -name 'compose.yaml' -o -name 'compose.yml' \) | sort + kustomize_overlay() { + if [ -f "$1/overlays/prod/kustomization.yaml" ]; then + echo "$1/overlays/prod" + elif [ -f "$1/base/kustomization.yaml" ]; then + echo "$1/base" + fi + } + + mapfile -t k8s_dirs < <( + git ls-files '*.yaml' '*.yml' \ + | grep -E '(^|/)k8s/' \ + | sed -E 's#((^|.*/)k8s)/.*#\1#' \ + | sort -u ) - mapfile -t k8s_manifests < <( - for kd in $(find "$repo" -type d -name k8s ! -path '*/.git/*' | sort); do - if [ -f "$kd/active" ]; then - collect_k8s "$kd" + k8s_manifests=() + kustomize_apps=() + for kd_rel in "${k8s_dirs[@]}"; do + kd="$repo/$kd_rel" + if is_disabled "$kd"; then + echo "skip (DISABLED): $kd_rel" + continue + fi + if [ -f "$kd/active" ]; then + overlay="$(kustomize_overlay "$kd" || true)" + if [ -n "${overlay:-}" ]; then + echo "kustomize app: ${overlay#$repo/}" + kustomize_apps+=("$overlay") else - collect_k8s_inactive "$kd" + while IFS= read -r f; do + [ -n "$f" ] && k8s_manifests+=("$repo/$f") + done < <(collect_k8s "$kd_rel" || true) fi - done + else + while IFS= read -r f; do + [ -n "$f" ] && k8s_manifests+=("$repo/$f") + done < <(collect_k8s_inactive "$kd_rel" || true) + fi + done + + mapfile -t compose_rel < <( + git ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort ) + compose_stacks=() + for cf_rel in "${compose_rel[@]}"; do + cf="$repo/$cf_rel" + if is_disabled "$cf"; then + echo "skip (DISABLED): $cf_rel" + continue + fi + if [ -f "$(dirname "$cf")/k8s/active" ]; then + echo "skip (k8s-managed): $cf_rel" + continue + fi + compose_stacks+=("$cf") + done + echo "== Validate compose stacks ==" for cf in "${compose_stacks[@]}"; do - dir=$(dirname "$cf") - if [ -f "$dir/k8s/active" ]; then - echo " skip (k8s-managed): $dir" - continue - fi echo " config: $cf" docker compose -f "$cf" config --quiet done - echo "== Validate k8s manifests (kubectl dry-run) ==" + echo "== Validate k8s manifests (kubectl dry-run=client) ==" for m in "${k8s_manifests[@]}"; do echo " apply --dry-run=client $m" kubectl apply --dry-run=client -f "$m" >/dev/null done + for k in "${kustomize_apps[@]}"; do + echo " apply -k --dry-run=client $k" + kubectl apply -k "$k" --dry-run=client >/dev/null + done + + echo "== Validate k8s manifests (kubectl dry-run=server) ==" + for m in "${k8s_manifests[@]}"; do + echo " apply --dry-run=server $m" + kubectl apply --dry-run=server -f "$m" >/dev/null + done + for k in "${kustomize_apps[@]}"; do + echo " apply -k --dry-run=server $k" + kubectl apply -k "$k" --dry-run=server >/dev/null + done + + echo "== Checking referenced Secrets exist ==" + echo " (deploy never applies *secret*.yaml; create missing ones from the laptop)" + ref_secrets=() + if [ "${#k8s_manifests[@]}" -gt 0 ]; then + while IFS= read -r s; do + [ -n "$s" ] && ref_secrets+=("$s") + done < <( + { + grep -h -A1 -E 'secretRef:|secretKeyRef:' "${k8s_manifests[@]}" 2>/dev/null || true + grep -h -E 'secretName:' "${k8s_manifests[@]}" 2>/dev/null || true + } | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true + ) + fi + missing_secrets=() + all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)" + for s in "${ref_secrets[@]}"; do + if printf '%s\n' "$all_secrets" | grep -qx "$s"; then + echo " ok: $s" + else + echo " MISSING: $s" + missing_secrets+=("$s") + fi + done + if [ "${#missing_secrets[@]}" -gt 0 ]; then + echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:" + printf ' - %s\n' "${missing_secrets[@]}" + echo "Create them manually from the laptop, e.g.:" + echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example" + exit 1 + fi echo "== Applying Kubernetes manifests ==" ns_files=() @@ -130,15 +238,19 @@ jobs: echo " namespaces first: ${ns_files[*]}" kubectl apply -f "${ns_files[@]}" fi - if [ -f "$repo/prometheus-stack/k8s/active" ]; then + if [ -f "$repo/prometheus-stack/k8s/active" ] && ! is_disabled "$repo/prometheus-stack/k8s"; then + if [ ! -f "$repo/prometheus-stack/k8s/grafana-values.yaml" ]; then + echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first." + exit 1 + fi echo "== Upgrading kube-prometheus-stack ==" helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \ --namespace prometheus \ --version 86.2.3 \ --values "$repo/prometheus-stack/k8s/grafana-values.yaml" \ - --wait + --wait --timeout 10m fi - if [ -f "$repo/loki/k8s/active" ]; then + if [ -f "$repo/loki/k8s/active" ] && ! is_disabled "$repo/loki/k8s"; then echo "== Upgrading loki/alloy ==" helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true helm repo update grafana >/dev/null 2>&1 || true @@ -146,12 +258,12 @@ jobs: --version 7.3.0 \ --namespace prometheus \ --values "$repo/loki/k8s/loki-values.yaml" \ - --wait + --wait --timeout 10m helm upgrade --install alloy grafana/alloy \ --version 1.12.1 \ --namespace prometheus \ --values "$repo/loki/k8s/alloy-values.yaml" \ - --wait + --wait --timeout 10m fi if [ "${#other_files[@]}" -gt 0 ]; then @@ -159,14 +271,30 @@ jobs: kubectl apply "${prune_opts[@]}" -f "${other_files[@]}" fi + for k in "${kustomize_apps[@]}"; do + echo "== Applying kustomize app: ${k#$repo/} ==" + kubectl apply -k "$k" + done + + if [ -f "$repo/userbot/k8s/active" ] && ! is_disabled "$repo/userbot"; then + echo "== userbot panel hook ==" + if kubectl get secret userbot-common-secrets -n userbot >/dev/null 2>&1; then + echo " userbot-common-secrets already present in userbot ns, not touching" + elif kubectl get secret userbot-common-secrets -n default >/dev/null 2>&1; then + echo " bootstrapping userbot-common-secrets into userbot ns" + kubectl get secret userbot-common-secrets -n default -o json \ + | jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \ + | kubectl apply -f - + else + echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop" + fi + kubectl rollout restart deployment/userbot-panel -n userbot + kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s + fi + echo "== Redeploying docker compose stacks ==" for cf in "${compose_stacks[@]}"; do - dir=$(dirname "$cf") - if [ -f "$dir/k8s/active" ]; then - echo " skip (k8s-managed): $dir" - continue - fi - echo " compose: $dir" + echo " compose: $cf" if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then docker compose -f "$cf" build docker compose -f "$cf" push diff --git a/edu_master/PLAYWRIGHT_VERSION b/edu_master/PLAYWRIGHT_VERSION new file mode 100644 index 0000000..3ebf789 --- /dev/null +++ b/edu_master/PLAYWRIGHT_VERSION @@ -0,0 +1 @@ +1.56.0 diff --git a/edu_master/compose.yaml b/edu_master/compose.yaml index 5a919d3..0c5a495 100644 --- a/edu_master/compose.yaml +++ b/edu_master/compose.yaml @@ -11,7 +11,7 @@ services: retries: 5 playwright-service: - image: mcr.microsoft.com/playwright:v1.63.0-jammy + image: mcr.microsoft.com/playwright:v1.56.0-jammy restart: unless-stopped command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws diff --git a/edu_master/k8s/alerts.yaml b/edu_master/k8s/alerts.yaml new file mode 100644 index 0000000..c79b6e6 --- /dev/null +++ b/edu_master/k8s/alerts.yaml @@ -0,0 +1,77 @@ +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + name: edu-master-webinar + namespace: edu-master + labels: + release: prometheus-stack +spec: + groups: + - name: edu_master.webinar + rules: + # No successful webinar check for 5m (~2-3 missed 2-min checks). + # Catches: playwright hangs/timeouts, version skew, site changes, hung job. + - alert: WebinarCheckerNoSuccessfulCheck + expr: | + (time() - webinar_check_last_success_timestamp_seconds > 300) + and (webinar_check_last_run_timestamp_seconds > 0) + for: 2m + labels: + severity: critical + annotations: + summary: "Webinar checker has no successful check for 5m" + description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent." + + # Fast path: 3 consecutive failures (~6+ min at 2-min interval). + - alert: WebinarCheckerConsecutiveFailures + expr: | + webinar_check_consecutive_failures >= 3 + for: 5m + labels: + severity: critical + annotations: + summary: "Webinar checker failing consecutively" + description: "edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace=\"edu-master\", container=\"webinar-checker\"})." + + # Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken. + - alert: WebinarCheckerScrapeDown + expr: | + absent(webinar_check_last_run_timestamp_seconds) == 1 + for: 10m + labels: + severity: critical + annotations: + summary: "Webinar checker metrics missing" + description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved." + + # EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped. + - alert: EduPhpsessidMissing + expr: | + edu_phpsessid_present == 0 + for: 10m + labels: + severity: critical + annotations: + summary: "EDU_PHPSESSID missing" + description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials." + + # Hard deps: checker and playwright deployments unavailable. + - alert: WebinarCheckerDeploymentDown + expr: | + kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0 + for: 10m + labels: + severity: critical + annotations: + summary: "Webinar checker deployment unavailable" + description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m." + + - alert: PlaywrightServiceDown + expr: | + kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0 + for: 10m + labels: + severity: critical + annotations: + summary: "Playwright service unavailable" + description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it." diff --git a/edu_master/k8s/playwright.yaml b/edu_master/k8s/playwright.yaml index 669530d..6d0a4e3 100644 --- a/edu_master/k8s/playwright.yaml +++ b/edu_master/k8s/playwright.yaml @@ -17,7 +17,8 @@ spec: spec: containers: - name: playwright - image: mcr.microsoft.com/playwright:v1.63.0-jammy + # renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker + image: mcr.microsoft.com/playwright:v1.56.0-jammy imagePullPolicy: IfNotPresent command: - npx diff --git a/edu_master/k8s/secrets.yaml.example b/edu_master/k8s/secrets.yaml.example index ac30818..7f1cf9b 100644 --- a/edu_master/k8s/secrets.yaml.example +++ b/edu_master/k8s/secrets.yaml.example @@ -21,6 +21,8 @@ stringData: WEBINAR_TELEGRAM_TOKEN: "" WEBINAR_ADMIN_ID: "" WEBINAR_CHECK_INTERVAL: "60" + # Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml) + METRICS_PORT: "8000" # Database REDIS_HOST: "redis" REDIS_PORT: "6379" diff --git a/edu_master/k8s/service.yaml b/edu_master/k8s/service.yaml new file mode 100644 index 0000000..a20026e --- /dev/null +++ b/edu_master/k8s/service.yaml @@ -0,0 +1,15 @@ +apiVersion: v1 +kind: Service +metadata: + name: webinar-checker + namespace: edu-master + labels: + app: edu-master-webinar-checker +spec: + selector: + app: edu-master-webinar-checker + ports: + - name: metrics + port: 8000 + targetPort: metrics + protocol: TCP diff --git a/edu_master/k8s/servicemonitor.yaml b/edu_master/k8s/servicemonitor.yaml new file mode 100644 index 0000000..f57abd3 --- /dev/null +++ b/edu_master/k8s/servicemonitor.yaml @@ -0,0 +1,16 @@ +apiVersion: monitoring.coreos.com/v1 +kind: ServiceMonitor +metadata: + name: webinar-checker + namespace: edu-master + labels: + release: prometheus-stack +spec: + selector: + matchLabels: + app: edu-master-webinar-checker + endpoints: + - port: metrics + path: /metrics + interval: 30s + scrapeTimeout: 10s diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml index cd8fd9a..58fcf7b 100644 --- a/edu_master/k8s/webinar-checker.yaml +++ b/edu_master/k8s/webinar-checker.yaml @@ -47,6 +47,10 @@ spec: - name: webinar-checker image: gcr.forust.xyz/forust/webinar-checker:latest imagePullPolicy: Always + ports: + - name: metrics + containerPort: 8000 + protocol: TCP envFrom: - secretRef: name: edu-master-secrets diff --git a/edu_master/webinar-checker/Dockerfile b/edu_master/webinar-checker/Dockerfile index 87af4a3..3f8d5fa 100644 --- a/edu_master/webinar-checker/Dockerfile +++ b/edu_master/webinar-checker/Dockerfile @@ -2,8 +2,11 @@ FROM python:3.11-slim WORKDIR /app -# Install dependencies -RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==1.56.0 redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10" +# renovate: datasource=pypi depName=playwright versioning=pep440 +ARG PLAYWRIGHT_VERSION=1.56.0 + +# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers +RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10" COPY checker.py . diff --git a/edu_master/webinar-checker/checker.py b/edu_master/webinar-checker/checker.py index dd2db88..ae3de7a 100644 --- a/edu_master/webinar-checker/checker.py +++ b/edu_master/webinar-checker/checker.py @@ -1,12 +1,15 @@ +import asyncio import contextlib import json import logging import os import re import tempfile +import threading import time from datetime import datetime, timedelta from html import escape +from http.server import BaseHTTPRequestHandler, HTTPServer import redis from playwright.async_api import async_playwright @@ -48,6 +51,106 @@ USER_AGENT = _env( ) WEBINAR_TELEGRAM_TOKEN = _env('WEBINAR_TELEGRAM_TOKEN') ADMIN_ID = int(_env('WEBINAR_ADMIN_ID', '0')) +METRICS_PORT = int(_env('METRICS_PORT', '8000')) + +# --- Prometheus metrics (stdlib only, no extra deps) --- +# Scraped by prometheus-stack via ServiceMonitor (edu_master/k8s/servicemonitor.yaml). +# Critical alerts in edu_master/k8s/alerts.yaml fire to Telegram via Alertmanager. +_METRICS_LOCK = threading.Lock() +_METRICS = { + 'last_run': 0.0, # Unix ts of last check start + 'last_success': 0.0, # Unix ts of last successful check + 'last_duration': 0.0, # Duration of last check in seconds + 'success_total': 0, + 'failure_total': 0, + 'consecutive_failures': 0, + 'phpsessid_present': 1, # 1 if EDU_PHPSESSID found in redis, else 0 +} + + +def _metric_check_start(): + with _METRICS_LOCK: + _METRICS['last_run'] = time.time() + + +def _metric_check_ok(duration: float): + now = time.time() + with _METRICS_LOCK: + _METRICS['last_success'] = now + _METRICS['last_duration'] = duration + _METRICS['success_total'] += 1 + _METRICS['consecutive_failures'] = 0 + _METRICS['phpsessid_present'] = 1 + + +def _metric_check_fail(duration: float, phpsessid_missing: bool = False): + with _METRICS_LOCK: + _METRICS['last_duration'] = duration + _METRICS['failure_total'] += 1 + _METRICS['consecutive_failures'] += 1 + _METRICS['phpsessid_present'] = 0 if phpsessid_missing else 1 + + +def _metrics_render() -> bytes: + with _METRICS_LOCK: + m = dict(_METRICS) + lines = [ + '# HELP webinar_check_last_run_timestamp_seconds Unix timestamp of last webinar check start.', + '# TYPE webinar_check_last_run_timestamp_seconds gauge', + f'webinar_check_last_run_timestamp_seconds {m["last_run"]}', + '# HELP webinar_check_last_success_timestamp_seconds Unix timestamp of last successful webinar check.', + '# TYPE webinar_check_last_success_timestamp_seconds gauge', + f'webinar_check_last_success_timestamp_seconds {m["last_success"]}', + '# HELP webinar_check_last_duration_seconds Duration of last webinar check in seconds.', + '# TYPE webinar_check_last_duration_seconds gauge', + f'webinar_check_last_duration_seconds {m["last_duration"]}', + '# HELP webinar_check_success_total Total successful webinar checks.', + '# TYPE webinar_check_success_total counter', + f'webinar_check_success_total {m["success_total"]}', + '# HELP webinar_check_failure_total Total failed webinar checks (timeout, playwright error, page error).', + '# TYPE webinar_check_failure_total counter', + f'webinar_check_failure_total {m["failure_total"]}', + '# HELP webinar_check_consecutive_failures Consecutive failed webinar checks (reset on success).', + '# TYPE webinar_check_consecutive_failures gauge', + f'webinar_check_consecutive_failures {m["consecutive_failures"]}', + '# HELP edu_phpsessid_present 1 if EDU_PHPSESSID exists in redis, 0 otherwise.', + '# TYPE edu_phpsessid_present gauge', + f'edu_phpsessid_present {m["phpsessid_present"]}', + ] + return ('\n'.join(lines) + '\n').encode() + + +class _MetricsHandler(BaseHTTPRequestHandler): + def do_GET(self): + if self.path == '/metrics': + body = _metrics_render() + self.send_response(200) + self.send_header('Content-Type', 'text/plain; version=0.0.4') + self.send_header('Content-Length', str(len(body))) + self.end_headers() + self.wfile.write(body) + elif self.path in ('/healthz', '/health'): + body = b'ok\n' + self.send_response(200) + self.send_header('Content-Type', 'text/plain') + self.send_header('Content-Length', str(len(body))) + self.end_headers() + self.wfile.write(body) + else: + self.send_response(404) + self.end_headers() + + def log_message(self, *args): + pass # keep bot logs clean + + +def start_metrics_server(port: int = METRICS_PORT): + server = HTTPServer(('0.0.0.0', port), _MetricsHandler) # noqa: S104 - k8s ServiceMonitor scrapes pod IP + thread = threading.Thread(target=server.serve_forever, name='metrics-server', daemon=True) + thread.start() + logger.info(f'Metrics server listening on :{port}/metrics') + return server + # Redis Keys KEY_WHITELIST = 'bot:whitelist' @@ -597,59 +700,66 @@ async def _collect_event_times(page) -> dict: async def fetch_diary_data(phpsessid: str) -> dict | None: logger.info('Fetching diary data via Playwright...') try: - async with async_playwright() as p: - browser = await p.chromium.connect(PLAYWRIGHT_WS) - try: - context_browser = await browser.new_context(user_agent=USER_AGENT) - await context_browser.add_cookies( - [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] - ) - page = await context_browser.new_page() - + async with asyncio.timeout(60): + async with async_playwright() as p: + browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15) try: - await page.goto(DIARY_URL, wait_until='domcontentloaded') - await page.wait_for_selector('table.calendar', timeout=10000) - await page.wait_for_timeout(1500) - - table_html = await page.evaluate(""" - () => { - const t = document.querySelector('table.calendar'); - return t ? t.outerHTML : null; - } - """) - if not table_html: - logger.error('table.calendar not found in DOM') - return None - - # Debug: save HTML for troubleshooting - with contextlib.suppress(Exception), open('/tmp/diary_debug.html', 'w', encoding='utf-8') as f: # noqa: S108 - f.write(table_html) - - month_text, days = _parse_calendar_html(table_html) - - # Read event times by opening each event's AJAX popup. - times_by_id = await _collect_event_times(page) - if times_by_id: - for day_data in days.values(): - for ev in day_data.get('events', []): - eid = ev.get('id') - if eid and eid in times_by_id: - ev['time'] = times_by_id[eid] - - logger.info( - f'Diary parsed: month={month_text!r}, days_with_events={sum(1 for d in days.values() if d["events"])}/{len(days)}' + context_browser = await browser.new_context(user_agent=USER_AGENT) + await context_browser.add_cookies( + [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] ) + page = await context_browser.new_page() - return {'monthFullText': month_text, 'days': days} + try: + await asyncio.wait_for(page.goto(DIARY_URL, wait_until='domcontentloaded'), timeout=30) + await page.wait_for_selector('table.calendar', timeout=10000) + await page.wait_for_timeout(1500) - except Exception as e: - logger.error(f'Error parsing diary: {e}') - return None + table_html = await page.evaluate(""" + () => { + const t = document.querySelector('table.calendar'); + return t ? t.outerHTML : null; + } + """) + if not table_html: + logger.error('table.calendar not found in DOM') + return None + + # Debug: save HTML for troubleshooting + with contextlib.suppress(Exception), open('/tmp/diary_debug.html', 'w', encoding='utf-8') as f: # noqa: S108 + f.write(table_html) + + month_text, days = _parse_calendar_html(table_html) + + # Read event times by opening each event's AJAX popup. + times_by_id = await _collect_event_times(page) + if times_by_id: + for day_data in days.values(): + for ev in day_data.get('events', []): + eid = ev.get('id') + if eid and eid in times_by_id: + ev['time'] = times_by_id[eid] + + logger.info( + f'Diary parsed: month={month_text!r}, days_with_events={sum(1 for d in days.values() if d["events"])}/{len(days)}' + ) + + return {'monthFullText': month_text, 'days': days} + + except Exception as e: + logger.error(f'Error parsing diary: {e}') + return None + finally: + with contextlib.suppress(Exception): + await asyncio.wait_for(page.close(), timeout=5) + with contextlib.suppress(Exception): + await asyncio.wait_for(context_browser.close(), timeout=5) finally: - await page.close() - await context_browser.close() - finally: - await browser.close() + with contextlib.suppress(Exception): + await asyncio.wait_for(browser.close(), timeout=5) + except TimeoutError: + logger.error('Diary fetch timed out (60s)') + return None except Exception as e: logger.error(f'Playwright error in diary fetch: {e}') return None @@ -931,48 +1041,55 @@ def _parse_schedule_html(table_html: str) -> dict: async def fetch_schedule_data(phpsessid: str) -> dict | None: logger.info('Fetching schedule data via Playwright...') try: - async with async_playwright() as p: - browser = await p.chromium.connect(PLAYWRIGHT_WS) - try: - context_browser = await browser.new_context(user_agent=USER_AGENT) - await context_browser.add_cookies( - [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] - ) - page = await context_browser.new_page() - + async with asyncio.timeout(60): + async with async_playwright() as p: + browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15) try: - await page.goto(SCHEDULE_URL, wait_until='domcontentloaded') - await page.wait_for_selector('table.schedule-table', timeout=10000) - await page.wait_for_timeout(1500) + context_browser = await browser.new_context(user_agent=USER_AGENT) + await context_browser.add_cookies( + [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] + ) + page = await context_browser.new_page() - table_html = await page.evaluate(""" - () => { - const t = document.querySelector('table.schedule-table'); - return t ? t.outerHTML : null; - } - """) - if not table_html: - logger.error('table.schedule-table not found in DOM') + try: + await asyncio.wait_for(page.goto(SCHEDULE_URL, wait_until='domcontentloaded'), timeout=30) + await page.wait_for_selector('table.schedule-table', timeout=10000) + await page.wait_for_timeout(1500) + + table_html = await page.evaluate(""" + () => { + const t = document.querySelector('table.schedule-table'); + return t ? t.outerHTML : null; + } + """) + if not table_html: + logger.error('table.schedule-table not found in DOM') + return None + + debug_path = os.path.join(tempfile.gettempdir(), 'schedule_debug.html') + with contextlib.suppress(Exception), open(debug_path, 'w', encoding='utf-8') as f: + f.write(table_html) + + data = _parse_schedule_html(table_html) + logger.info(list(data['weekdays'].keys())) + logger.info(f'Schedule parsed: {len(data["weekdays"])} days, classes={data["classes"]}') + + return data + + except Exception as e: + logger.error(f'Error parsing schedule: {e}') return None - - debug_path = os.path.join(tempfile.gettempdir(), 'schedule_debug.html') - with contextlib.suppress(Exception), open(debug_path, 'w', encoding='utf-8') as f: - f.write(table_html) - - data = _parse_schedule_html(table_html) - logger.info(list(data['weekdays'].keys())) - logger.info(f'Schedule parsed: {len(data["weekdays"])} days, classes={data["classes"]}') - - return data - - except Exception as e: - logger.error(f'Error parsing schedule: {e}') - return None + finally: + with contextlib.suppress(Exception): + await asyncio.wait_for(page.close(), timeout=5) + with contextlib.suppress(Exception): + await asyncio.wait_for(context_browser.close(), timeout=5) finally: - await page.close() - await context_browser.close() - finally: - await browser.close() + with contextlib.suppress(Exception): + await asyncio.wait_for(browser.close(), timeout=5) + except TimeoutError: + logger.error('Schedule fetch timed out (60s)') + return None except Exception as e: logger.error(f'Playwright error in schedule fetch: {e}') return None @@ -1485,10 +1602,13 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE): int: Number of webinars found, or None if check failed """ logger.info('Running webinar check...') + _t0 = time.time() + _metric_check_start() phpsessid = redis_client.get(KEY_PHPSESSID) if not phpsessid: logger.warning('PHPSESSID missing. Skipping check.') + _metric_check_fail(time.time() - _t0, phpsessid_missing=True) # --- DEBUG LOGGING --- try: with open('phpsessid_missing.log', 'a') as f: @@ -1502,78 +1622,88 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE): content = '' try: - async with async_playwright() as p: - # Connect to remote Playwright service - browser = await p.chromium.connect(PLAYWRIGHT_WS) - - try: - # Create browser context with user agent - context_browser = await browser.new_context(user_agent=USER_AGENT) - - # Add PHPSESSID cookie - await context_browser.add_cookies( - [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] - ) - - # Create new page - page = await context_browser.new_page() + async with asyncio.timeout(90): + async with async_playwright() as p: + # Connect to remote Playwright service + browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15) try: - # Navigate to webinar page - await page.goto(WEBINAR_URL, wait_until='domcontentloaded') + # Create browser context with user agent + context_browser = await browser.new_context(user_agent=USER_AGENT) - # Wait for the table to load - await page.wait_for_selector('#meetings table', timeout=10000) - await page.wait_for_timeout(2000) + # Add PHPSESSID cookie + await context_browser.add_cookies( + [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] + ) - # Get page content - content = await page.content() + # Create new page + page = await context_browser.new_page() - # Check if "no webinar" message is present - if NO_WEBINAR_MARKER not in content: - logger.info('!!! WEBINAR FOUND !!!') + try: + # Navigate to webinar page + await asyncio.wait_for(page.goto(WEBINAR_URL, wait_until='domcontentloaded'), timeout=30) - # Extract webinar details from table rows - rows = page.locator('#meetings table tbody tr') - count = await rows.count() + # Wait for the table to load + await page.wait_for_selector('#meetings table', timeout=10000) + await page.wait_for_timeout(2000) - for i in range(count): - row = rows.nth(i) - text = await row.inner_text() + # Get page content + content = await page.content() - if NO_WEBINAR_MARKER not in text: - # Extract name (topic) from first column - name_elem = row.locator('td').nth(0) - name = await name_elem.inner_text() - name = name.strip() + # Check if "no webinar" message is present + if NO_WEBINAR_MARKER not in content: + logger.info('!!! WEBINAR FOUND !!!') - # Extract join URL from fourth column - url_elem = row.locator('td').nth(3).locator('a[href*="/webinar/join/"]').first - url = await url_elem.get_attribute('href') + # Extract webinar details from table rows + rows = page.locator('#meetings table tbody tr') + count = await rows.count() - if name and url: - current_webinars.append({'name': name, 'url': url, 'text': text.strip()}) - logger.info(f'Found webinar: {name} -> {url}') - else: - logger.info('No webinars found (expected message present)') + for i in range(count): + row = rows.nth(i) + text = await row.inner_text() - except Exception as e: - logger.error(f'Error checking page: {e}. Saving content for debug.') - # If page content is available, save it on error - with contextlib.suppress(Exception): - if page and not content: - content = await page.content() + if NO_WEBINAR_MARKER not in text: + # Extract name (topic) from first column + name_elem = row.locator('td').nth(0) + name = await name_elem.inner_text() + name = name.strip() + + # Extract join URL from fourth column + url_elem = row.locator('td').nth(3).locator('a[href*="/webinar/join/"]').first + url = await url_elem.get_attribute('href') + + if name and url: + current_webinars.append({'name': name, 'url': url, 'text': text.strip()}) + logger.info(f'Found webinar: {name} -> {url}') + else: + logger.info('No webinars found (expected message present)') + + except Exception as e: + logger.error(f'Error checking page: {e}. Saving content for debug.') + # If page content is available, save it on error + with contextlib.suppress(Exception): + if page and not content: + content = await page.content() + + _metric_check_fail(time.time() - _t0) + return None + finally: + with contextlib.suppress(Exception): + await asyncio.wait_for(page.close(), timeout=5) + with contextlib.suppress(Exception): + await asyncio.wait_for(context_browser.close(), timeout=5) - return None finally: - await page.close() - await context_browser.close() - - finally: - await browser.close() + with contextlib.suppress(Exception): + await asyncio.wait_for(browser.close(), timeout=5) + except TimeoutError: + logger.error('Webinar check timed out after 90s (playwright hang)') + _metric_check_fail(time.time() - _t0) + return None except Exception as e: logger.error(f'Playwright error: {e}') + _metric_check_fail(time.time() - _t0) return None # --- DEBUG LOGGING (Saving last response content) --- @@ -1637,6 +1767,7 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE): else: logger.info(f'Found {len(current_webinars)} webinar(s), but all are already known') + _metric_check_ok(time.time() - _t0) return len(current_webinars) @@ -1680,6 +1811,8 @@ def main(): job_queue = app.job_queue job_queue.run_repeating(check_webinars_job, interval=WEBINAR_CHECK_INTERVAL, first=10) + start_metrics_server() + logger.info('Bot started polling...') app.run_polling() diff --git a/gitea/k8s/gitea.yaml b/gitea/k8s/gitea.yaml index 6c75c6f..3ce7aa7 100644 --- a/gitea/k8s/gitea.yaml +++ b/gitea/k8s/gitea.yaml @@ -31,7 +31,7 @@ spec: spec: containers: - name: gitea - image: docker.gitea.com/gitea:1.27.3 + image: gitea/gitea:1.27.3 envFrom: - configMapRef: name: gitea-config diff --git a/postgres/.env.example b/postgres/.env.example index ae8600c..c99ea0b 100644 --- a/postgres/.env.example +++ b/postgres/.env.example @@ -3,3 +3,4 @@ AUTHENTIK_DB_PASSWORD= GITEA_DB_PASSWORD= NETRONOME_DB_PASSWORD= PENPOT_DB_PASSWORD= +STATUSPAGE_DB_PASSWORD= diff --git a/postgres/README.md b/postgres/README.md index 8e2fa7e..d839d48 100644 --- a/postgres/README.md +++ b/postgres/README.md @@ -1,23 +1,22 @@ # Shared PostgreSQL -This directory contains a PostgreSQL 15 deployment draft for Authentik, Gitea, -Netronome, and Penpot. It creates one database and one login role per service; -it does not migrate existing data or change application connection settings. +This directory contains the shared PostgreSQL 17 deployment for Authentik, +Gitea, Netronome, and Statuspage. It creates one database and one login role +per service. Per-service standalone databases were removed after the +migration (Sep 2026); Penpot stays on its own compose PostgreSQL (archived, +not part of the shared instance). ## Compatibility baseline -| Service | Current application | Current standalone PostgreSQL | Common PostgreSQL 15 | -| --------- | ------------------- | ----------------------------: | ------------------------------------------------------------------------------------ | -| Authentik | 2025.10.2 | 15 | Supported (Authentik requires 14+) | -| Gitea | 1.27.3 | 14 | Supported (Gitea requires 12+) | -| Netronome | 0.14.0 | 17 | Validate in staging; upstream's example uses 17 but no 17-only feature is documented | -| Penpot | 2.17.2 | 15 | Supported by the official deployment | +| Service | Current application | Shared PostgreSQL 17 | +| --------- | ------------------- | -------------------- | +| Authentik | 2025.10.x | Supported (Authentik requires 14+) | +| Gitea | 1.27.3 | Supported (Gitea requires 12+) | +| Netronome | 0.14.0 | Supported (upstream's example uses 17) | +| Statuspage| custom | Supported | -PostgreSQL 15 is the conservative common major. A major-version downgrade or -change must use a logical dump/restore; changing only the image tag while -keeping a data directory is not supported. Back up and migrate one application -at a time, starting with Netronome because its current standalone deployment -uses PostgreSQL 17. +A major-version change must use a logical dump/restore; changing only the +image tag while keeping a data directory is not supported. For Compose, copy `.env.example` to `.env`, set all passwords, and start it with `docker compose -f shared-compose.yaml up -d`. This file is intentionally not diff --git a/postgres/initdb/01-create-databases.sh b/postgres/initdb/01-create-databases.sh index ffdb01d..4da0148 100755 --- a/postgres/initdb/01-create-databases.sh +++ b/postgres/initdb/01-create-databases.sh @@ -5,12 +5,12 @@ set -euo pipefail : "${GITEA_DB_PASSWORD:?GITEA_DB_PASSWORD is required}" : "${NETRONOME_DB_PASSWORD:?NETRONOME_DB_PASSWORD is required}" : "${PENPOT_DB_PASSWORD:?PENPOT_DB_PASSWORD is required}" +: "${STATUSPAGE_DB_PASSWORD:?STATUSPAGE_DB_PASSWORD is required}" create_role_and_database() { local role="$1" local database="$2" local password="$3" - psql --username "$POSTGRES_USER" --dbname postgres \ -v role="$role" -v database="$database" -v password="$password" \ <<'SQL' @@ -25,3 +25,4 @@ create_role_and_database authentik authentik "$AUTHENTIK_DB_PASSWORD" create_role_and_database gitea gitea "$GITEA_DB_PASSWORD" create_role_and_database netronome netronome "$NETRONOME_DB_PASSWORD" create_role_and_database penpot penpot "$PENPOT_DB_PASSWORD" +create_role_and_database statuspage statuspage "$STATUSPAGE_DB_PASSWORD" diff --git a/postgres/k8s/postgres.yaml b/postgres/k8s/postgres.yaml index 42b691e..e452d52 100644 --- a/postgres/k8s/postgres.yaml +++ b/postgres/k8s/postgres.yaml @@ -60,11 +60,23 @@ spec: command: ["pg_isready", "-U", "postgres", "-d", "postgres"] initialDelaySeconds: 10 periodSeconds: 10 + startupProbe: + exec: + command: ["pg_isready", "-U", "postgres", "-d", "postgres"] + failureThreshold: 30 + periodSeconds: 10 livenessProbe: exec: command: ["pg_isready", "-U", "postgres", "-d", "postgres"] initialDelaySeconds: 30 periodSeconds: 20 + resources: + requests: + memory: "512Mi" + cpu: "500m" + limits: + memory: "2Gi" + cpu: "2000m" volumes: - name: postgres-data persistentVolumeClaim: diff --git a/postgres/shared-compose.yaml b/postgres/shared-compose.yaml index 872078d..0dac300 100644 --- a/postgres/shared-compose.yaml +++ b/postgres/shared-compose.yaml @@ -1,6 +1,6 @@ services: postgres: - image: postgres:15.19-alpine + image: postgres:17.11-alpine container_name: homelab-postgres restart: unless-stopped env_file: diff --git a/renovate.json b/renovate.json index e8e8fc0..b548672 100644 --- a/renovate.json +++ b/renovate.json @@ -1,14 +1,43 @@ { "$schema": "https://docs.renovatebot.com/renovate-schema.json", "extends": ["config:recommended"], - "enabledManagers": ["docker-compose", "kubernetes", "helm-values"], + "enabledManagers": ["dockerfile", "docker-compose", "kubernetes", "helm-values", "custom.regex"], "helm-values": { "managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"] }, "kubernetes": { "managerFilePatterns": ["/k8s/.+\\.ya?ml$/"] }, + "customManagers": [ + { + "customType": "regex", + "description": "singlesource: playwright npm version pinned in npx command (k8s + compose)", + "fileMatch": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"], + "matchStrings": ["playwright@(?\\d+\\.\\d+\\.\\d+)"], + "datasourceTemplate": "npm", + "depNameTemplate": "playwright" + }, + { + "customType": "regex", + "description": "singlesource: PLAYWRIGHT_VERSION file", + "fileMatch": ["^edu_master/PLAYWRIGHT_VERSION$"], + "matchStrings": ["^(?\\d+\\.\\d+\\.\\d+)$"], + "datasourceTemplate": "pypi", + "depNameTemplate": "playwright" + } + ], "packageRules": [ + { + "description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync", + "matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"], + "groupName": "playwright singlesource", + "groupSlug": "playwright" + }, + { + "description": "playwright must not automerge - version skew breaks WS handshake (checker.py:1523 vs playwright.yaml:20)", + "matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"], + "automerge": false + }, { "description": "Keep private homelab images unchanged", "matchDatasources": ["docker"],