From a5d384a4d8babdaccdb09f34f1661348f8794ee9 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Sun, 27 Sep 2026 10:50:07 +0200 Subject: [PATCH] feat(deploy): probe every active service after a deploy, rollouts included verify-k8s watches rollouts, which reports that pods converged. It cannot tell a converged pod from a serving one. A Service selector pointing at a port nothing listens on, a 500 from the app itself, a Traefik route that stopped matching, a pod that OOMKilled early enough to still count as Available for the duration of the check -- all of those are green at the rollout level and broken for whoever opens the URL. So ask what users ask. A smoke stage probes the public route of every active service and fails on a transport error, a 5xx, or a 000, which curl reports when it exits cleanly and nothing replied. Everything else passes, including 4xx: a 404 from a path the service does not serve and a 302 to a login both prove Traefik matched the host, the Service resolved to a pod and the pod answered, which is the whole claim being tested. An empty host list is an error, not a pass. Zero names means the extraction broke, and reporting a clean deploy off a broken grep is the failure mode this job exists to catch. It runs on always() and after verify-k8s rather than before it, because a rollback is when a route most needs re-checking. It only skips when verify-k8s did, which is when nothing was deployed at all. Two things worth writing down, because both were wrong on the first pass: Stripping comments before reading the routes is not optional. naio and xui are still in the tree commented out, and a plain grep picks both up and then reports two services as unreachable when nobody ever deployed them. The apex forust.xyz also needs a filter that admits it, so /\.forust\.xyz$/ quietly dropped the site root. And the 5xx test was written as ${code%%[0-9]*} != 5, which is empty for every three-digit code, so a 500 was reported as ok. A case glob on the leading digit is what actually works. 23 routes answer today, in 1.6s. The 5xx branch is the one part no live service here exercises, so it was checked by running the block over 200 through 599 and 000 rather than against a real response. --- .gitea/workflows/deploy-lib.sh | 97 ++++++++++++++++++++++++++++++++++ .gitea/workflows/deploy.yaml | 26 +++++++++ 2 files changed, 123 insertions(+) diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 39d5113..3299b44 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -778,6 +778,102 @@ verify_compose_stack() { return 0 } +# The public hostname of every active service, one per line. +# +# Comments are stripped first, and deliberately so: a route that someone +# disabled by commenting it out is not a service to probe, and naio and xui are +# both still in the tree that way. A `#` only starts a comment when it is at the +# start of a line or after whitespace, so `s/#.*//` alone would also cut a +# legitimate value in half. +# +# Only the public names. The *.internal names are the same Traefik and the same +# Services, reached by a different label, so probing both would double the run +# to learn the same thing. The public name is also the one a user types. +smoke_hosts() { + local m k + # The backticks below are literal. They are Traefik's Host() delimiter, and the + # single quotes are precisely what keeps the shell from reading them as a + # command substitution, so the warning is the opposite of a real problem. + # shellcheck disable=SC2016 + { + for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do + [ -f "$m" ] && cat "$m" + done + for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do + kubectl kustomize "$k" 2>/dev/null || true + done + } | sed -E 's/(^|[[:space:]])#.*$//' \ + | grep -oE 'Host\(`[^`]+`\)' \ + | sed -E 's/^Host\(`//; s/`\)$//' \ + | grep -E '(^|\.)forust\.xyz$' \ + | grep -v '\${' \ + | sort -u +} + +# stage_verify_k8s watches the rollout, which reports that the pods converged. +# It cannot tell a converged pod from a serving one: a route pointing at the +# wrong port, a Service selector that matches nothing the app listens on, a 500 +# from the app itself, an OOMKill loop that still counts as Available for long +# enough to pass. All of those are green at the rollout level. +# +# So ask the thing users ask. Any HTTP response proves Traefik matched the +# host, the Service resolved to a pod and the pod answered -- a 302 to a login +# or a 404 from a path the service does not serve still means the chain is +# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means +# the service is not serving, and only those fail the run. +stage_smoke() { + cd "$REPO" + select_manifests >/dev/null + local -a hosts=() + # Not named failed: an array of that name already exists in restart_stale_images + # above, and a scalar shadowing an array is a trap rather than a shadow. + local h code rc bad=0 + while IFS= read -r h; do + [ -n "$h" ] && hosts+=("$h") + done < <(smoke_hosts) + + if [ "${#hosts[@]}" -eq 0 ]; then + # Nothing to probe means the extraction broke, not that the cluster is empty. + echo "ERROR: no public hostnames found in active manifests, refusing to report success" + return 1 + fi + + log "Probing ${#hosts[@]} public route(s)" + for h in "${hosts[@]}"; do + code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$? + if [ "$rc" -ne 0 ]; then + echo " UNREACHABLE $h (curl exit $rc)" + bad=1 + continue + fi + # A glob, not a string compare. `case` on the leading digit is the only one + # of these that survives a three-digit code, and the obvious expansion to + # try first -- ${code%%[0-9]*} -- is empty for every input, so it silently + # reports a 500 as healthy. + case "$code" in + 5*) + echo " SERVER ERROR $h $code" + bad=1 + ;; + 000) + # curl exited 0 and still no status, so nothing on the far end replied. + # Not a pass, whatever the transport thought. + echo " NO RESPONSE $h" + bad=1 + ;; + *) + echo " ok $h $code" + ;; + esac + done + + if [ "$bad" -ne 0 ]; then + echo "ERROR: at least one active service is not serving over its public route" + return 1 + fi + echo "all ${#hosts[@]} route(s) answered" +} + stage_apply_compose() { cd "$REPO" select_manifests >/dev/null @@ -816,6 +912,7 @@ run_stage() { validate) stage_validate ;; apply-k8s) stage_apply_k8s ;; verify-k8s) stage_verify_k8s ;; + smoke) stage_smoke ;; apply-compose) stage_apply_compose ;; *) echo "ERROR: unknown stage: $1" diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 395ef52..2ce9f83 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -123,3 +123,29 @@ jobs: run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh verify-k8s + + # Asks the public route of every active service whether it is actually + # serving, which the rollout check above structurally cannot: a pod can + # converge and still be crash-looping, or be listening on a port no Service + # points at, or answer 500. + # + # `always()` for the same reason verify-k8s has it, and it runs after that job + # specifically because a rollback is when a route most needs re-checking. The + # needs is a barrier, not a filter: whether verify-k8s passed, failed or was + # cancelled, the probes are what say whether the cluster is serving, and + # suppressing them on a rollback would hide the one run where the answer + # matters most. + smoke: + needs: [verify-k8s] + if: always() && needs.verify-k8s.result != 'skipped' + runs-on: [self-hosted, linux, arch, homelab, prod] + timeout-minutes: 10 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + + - name: Probe the public route of every active service + shell: bash + run: | + set -euo pipefail + ./.gitea/workflows/ssh-run.sh smoke