diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 39d5113..3299b44 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -778,6 +778,102 @@ verify_compose_stack() { return 0 } +# The public hostname of every active service, one per line. +# +# Comments are stripped first, and deliberately so: a route that someone +# disabled by commenting it out is not a service to probe, and naio and xui are +# both still in the tree that way. A `#` only starts a comment when it is at the +# start of a line or after whitespace, so `s/#.*//` alone would also cut a +# legitimate value in half. +# +# Only the public names. The *.internal names are the same Traefik and the same +# Services, reached by a different label, so probing both would double the run +# to learn the same thing. The public name is also the one a user types. +smoke_hosts() { + local m k + # The backticks below are literal. They are Traefik's Host() delimiter, and the + # single quotes are precisely what keeps the shell from reading them as a + # command substitution, so the warning is the opposite of a real problem. + # shellcheck disable=SC2016 + { + for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do + [ -f "$m" ] && cat "$m" + done + for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do + kubectl kustomize "$k" 2>/dev/null || true + done + } | sed -E 's/(^|[[:space:]])#.*$//' \ + | grep -oE 'Host\(`[^`]+`\)' \ + | sed -E 's/^Host\(`//; s/`\)$//' \ + | grep -E '(^|\.)forust\.xyz$' \ + | grep -v '\${' \ + | sort -u +} + +# stage_verify_k8s watches the rollout, which reports that the pods converged. +# It cannot tell a converged pod from a serving one: a route pointing at the +# wrong port, a Service selector that matches nothing the app listens on, a 500 +# from the app itself, an OOMKill loop that still counts as Available for long +# enough to pass. All of those are green at the rollout level. +# +# So ask the thing users ask. Any HTTP response proves Traefik matched the +# host, the Service resolved to a pod and the pod answered -- a 302 to a login +# or a 404 from a path the service does not serve still means the chain is +# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means +# the service is not serving, and only those fail the run. +stage_smoke() { + cd "$REPO" + select_manifests >/dev/null + local -a hosts=() + # Not named failed: an array of that name already exists in restart_stale_images + # above, and a scalar shadowing an array is a trap rather than a shadow. + local h code rc bad=0 + while IFS= read -r h; do + [ -n "$h" ] && hosts+=("$h") + done < <(smoke_hosts) + + if [ "${#hosts[@]}" -eq 0 ]; then + # Nothing to probe means the extraction broke, not that the cluster is empty. + echo "ERROR: no public hostnames found in active manifests, refusing to report success" + return 1 + fi + + log "Probing ${#hosts[@]} public route(s)" + for h in "${hosts[@]}"; do + code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$? + if [ "$rc" -ne 0 ]; then + echo " UNREACHABLE $h (curl exit $rc)" + bad=1 + continue + fi + # A glob, not a string compare. `case` on the leading digit is the only one + # of these that survives a three-digit code, and the obvious expansion to + # try first -- ${code%%[0-9]*} -- is empty for every input, so it silently + # reports a 500 as healthy. + case "$code" in + 5*) + echo " SERVER ERROR $h $code" + bad=1 + ;; + 000) + # curl exited 0 and still no status, so nothing on the far end replied. + # Not a pass, whatever the transport thought. + echo " NO RESPONSE $h" + bad=1 + ;; + *) + echo " ok $h $code" + ;; + esac + done + + if [ "$bad" -ne 0 ]; then + echo "ERROR: at least one active service is not serving over its public route" + return 1 + fi + echo "all ${#hosts[@]} route(s) answered" +} + stage_apply_compose() { cd "$REPO" select_manifests >/dev/null @@ -816,6 +912,7 @@ run_stage() { validate) stage_validate ;; apply-k8s) stage_apply_k8s ;; verify-k8s) stage_verify_k8s ;; + smoke) stage_smoke ;; apply-compose) stage_apply_compose ;; *) echo "ERROR: unknown stage: $1" diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 395ef52..2ce9f83 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -123,3 +123,29 @@ jobs: run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh verify-k8s + + # Asks the public route of every active service whether it is actually + # serving, which the rollout check above structurally cannot: a pod can + # converge and still be crash-looping, or be listening on a port no Service + # points at, or answer 500. + # + # `always()` for the same reason verify-k8s has it, and it runs after that job + # specifically because a rollback is when a route most needs re-checking. The + # needs is a barrier, not a filter: whether verify-k8s passed, failed or was + # cancelled, the probes are what say whether the cluster is serving, and + # suppressing them on a rollback would hide the one run where the answer + # matters most. + smoke: + needs: [verify-k8s] + if: always() && needs.verify-k8s.result != 'skipped' + runs-on: [self-hosted, linux, arch, homelab, prod] + timeout-minutes: 10 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + + - name: Probe the public route of every active service + shell: bash + run: | + set -euo pipefail + ./.gitea/workflows/ssh-run.sh smoke