#!/usr/bin/env bash # Workstation deploy stages; invoked by the durable controller against pinned source. # REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF' # source "$REPO/.gitea/workflows/deploy-lib.sh" # run_stage "$STAGE" # EOF set -euo pipefail : "${REPO:?REPO must be set}" APPLY_PRUNE="${APPLY_PRUNE:-false}" CONFIG_REPO="${CONFIG_REPO:-$REPO}" # Exact SHA accepted by the CI gate for both manual and automatic deploys. DEPLOY_SHA="${DEPLOY_SHA:-}" # Handoff point between the apply stage (writes) and the verify stage (reads). # Under the deploy user's own XDG state directory rather than /var/backups: the # deploy is unprivileged, /var/backups does not exist on a minimal Arch host, and # creating it would need root — which is why the first real deploy died here with # "is not writable" before touching a single workload. $HOME comes from sshd. DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state}/homelab-deploy}" # Per-workload rollout budget and how many workloads to watch at once. The whole # apply job has its own timeout-minutes as a backstop. ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}" ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-4}" WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps" log() { echo "== $* ==" } # Store operation results without command output or local configuration values. record_apply() { [ -n "${RUN_DIR:-}" ] || return 0 jq -cn --arg action "$1" --arg target "$2" --arg result "$3" \ '{action: $action, target: $target, result: $result}' >>"$RUN_DIR/apply-events.jsonl" \ || echo 'WARNING: cannot record an apply result' >&2 return 0 } warn() { echo "WARNING: $*" >&2 } # Prune needs the complete desired set in one invocation. Per-file pruning # treats resources from the other files as absent and can delete them. check_prune_mode() { if [ "$APPLY_PRUNE" = "true" ]; then echo "ERROR: APPLY_PRUNE=true is unsupported by the per-file deploy loop." >&2 echo "Disable it; remove obsolete resources explicitly after review." >&2 return 1 fi } collect_k8s() { git -C "$REPO" ls-files -- "$1" \ | grep -E '\.ya?ml$' \ | grep -Ev '/overlays/' \ | grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \ | grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \ | sort } kustomize_overlay() { if [ -f "$1/overlays/prod/kustomization.yaml" ]; then echo "$1/overlays/prod" elif [ -f "$1/base/kustomization.yaml" ]; then echo "$1/base" elif [ -f "$1/kustomization.yaml" ]; then echo "$1" fi } selected_service() { local kind="$1" service="$2" section=selected [ -n "${DEPLOY_PLAN:-}" ] || return 0 if [ "${DEPLOY_SMOKE_ALL:-false}" = true ]; then section=active; fi jq -e --arg kind "$kind" --arg service "$service" --arg section "$section" \ '.[$section][$kind] | index($service) != null' "$DEPLOY_PLAN" >/dev/null } # Resolve .env and relative binds on the persistent workstation tree. Locked # JSON configs keep the same Compose project name and volume names. compose() { local cf="$1" locked project_dir project_dir="$CONFIG_REPO/$(basename "$(dirname "$cf")")" shift locked="${RUN_DIR:-/nonexistent}/compose/$(basename "$(dirname "$cf")").json" if [ -f "$locked" ]; then cf="$locked"; fi (cd "$CONFIG_REPO" && docker compose --project-directory "$project_dir" -f "$cf" "$@") } select_manifests() { K8S_MANIFESTS=() KUSTOMIZE_APPS=() COMPOSE_STACKS=() local kd_rel kd overlay cf_rel cf f while IFS= read -r kd_rel; do kd="$REPO/$kd_rel" selected_service k8s "${kd_rel%/k8s}" || continue if [ ! -f "$kd/active" ]; then echo "skip (no k8s/active): $kd_rel" continue fi overlay="$(kustomize_overlay "$kd" || true)" if [ -n "${overlay:-}" ]; then echo "kustomize app: ${overlay#"$REPO"/}" KUSTOMIZE_APPS+=("$overlay") else while IFS= read -r f; do [ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f") done < <(collect_k8s "$kd_rel" || true) fi done < <( git -C "$REPO" ls-files '*.yaml' '*.yml' \ | grep -E '(^|/)k8s/' \ | sed -E 's#((^|.*/)k8s)/.*#\1#' \ | sort -u ) while IFS= read -r cf_rel; do cf="$REPO/$cf_rel" selected_service compose "$(dirname "$cf_rel")" || continue if [ -f "$(dirname "$cf")/active" ]; then echo "compose: $cf_rel" COMPOSE_STACKS+=("$cf") else echo "skip (no root active): $cf_rel" fi done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort) } # --- post-apply verification and rollback ------------------------------------- # # A green `kubectl apply` says nothing about the cluster being healthy. These # helpers watch exactly the workloads whose spec changed during this apply, and # on failure roll them back to the revision that was running before, so a bad # push to main cannot leave a service crash-looping. # # The workstation controller runs apply and verification as separate durable # stages. Runner jobs only follow their logs. ExecStopPost recovers interrupted # runs using the per-run snapshot, even when the SSH connection has gone away. # # Creates this run's snapshot directory and publishes it as the handoff point for # the verify stage. Fails hard by design: a deploy that cannot record what it is # about to change must not start, because then nothing can be rolled back for it # automatically. Publishing happens before the first apply, so an apply killed # mid-flight still leaves a usable baseline behind. snapshot_dir() { local stamp dir stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}" dir="$DEPLOY_SNAPSHOT_DIR/$stamp" if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2 echo "The verify job needs it to learn which workloads this deploy touches." >&2 echo "Refusing to deploy without a way to roll back." >&2 return 1 fi if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then echo "ERROR: cannot create snapshot dir $dir" >&2 return 1 fi if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2 return 1 fi printf '%s\n' "$dir" } save_snapshot() { local dir="$1" releases revision status log "Saving pre-apply snapshot to $dir" workload_generations >"$dir/generations.before" || return 1 kubectl get "$WORKLOAD_KINDS" -A -o json >"$dir/workloads.json" || return 1 kubectl get controllerrevisions.apps -A -o json >"$dir/controller-revisions.json" || return 1 jq --slurpfile revisions "$dir/controller-revisions.json" ' [.items[] | . as $w | { kind: (.kind | ascii_downcase), namespace: .metadata.namespace, name: .metadata.name, uid: .metadata.uid, revision: (if .kind == "Deployment" then (.metadata.annotations["deployment.kubernetes.io/revision"] // "0" | tonumber) else ([$revisions[0].items[] | select(.metadata.namespace == $w.metadata.namespace) | select(any(.metadata.ownerReferences[]?; .uid == $w.metadata.uid)) | select($w.kind != "StatefulSet" or .metadata.name == $w.status.currentRevision) | .revision] | max // 0) end) }]' "$dir/workloads.json" >"$dir/revisions.json" || return 1 # Helm 4 lists every release status by default and removed the --all flag. releases="$(helm list -A -o json)" || return 1 for entry in "${HELM_RELEASES[@]}"; do IFS='|' read -r release _ namespace _ _ _ <<<"$entry" if ! jq -e --arg r "$release" --arg n "$namespace" \ 'any(.[]; .name == $r and .namespace == $n)' <<<"$releases" >/dev/null; then continue fi helm status "$release" -n "$namespace" -o json >"$dir/helm-$release.json" || return 1 status="$(jq -r '.info.status' "$dir/helm-$release.json")" if [ "$status" != deployed ]; then # Never capture a pending/failed revision as the recovery target. helm history "$release" -n "$namespace" -o json >"$dir/helm-$release.history.json" || return 1 revision="$(jq '[.[] | select(.status == "deployed" or .status == "superseded") | .revision] | max // 0' \ "$dir/helm-$release.history.json")" jq --argjson revision "$revision" '.version = $revision' "$dir/helm-$release.json" >"$dir/helm-$release.tmp" mv "$dir/helm-$release.tmp" "$dir/helm-$release.json" fi done # The verify stage compares this against the commit it is deploying, to refuse # rolling back against a baseline left by an earlier run. A snapshot we cannot # attribute to a commit is unusable for that, so fail before anything is applied. if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then echo "ERROR: cannot record the deploy commit in $dir/commit" >&2 return 1 fi } # Prints " " for every workload in the cluster. workload_generations() { kubectl get "$WORKLOAD_KINDS" -A \ -o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \ --no-headers 2>/dev/null \ | awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }' } # Prints " " for every workload that is new or whose generation # moved since the snapshot, i.e. the ones this apply actually touched. changed_workloads() { local before="$1" local ns name kind gen old current current="$(workload_generations)" || return 1 while read -r ns name kind gen; do [ -n "${gen:-}" ] || continue old="$(awk -v want_ns="$ns" -v want_name="$name" -v want_kind="$kind" \ '$1 == want_ns && $2 == want_name && $3 == want_kind { print $4; exit }' "$before" 2>/dev/null || true)" if [ -n "${RUN_DIR:-}" ] && ! grep -qxF "$kind $ns $name" "$RUN_DIR/workload-refs"; then continue fi if [ "$old" != "$gen" ]; then printf '%s %s %s\n' "$kind" "$ns" "$name" fi done <<<"$current" } # Resolve owned image references exclusively from the checked CI artifact. render_pinned() { python3 "$REPO/.gitea/workflows/release.py" render } # verify_workloads ... # Watches every workload in parallel and records the ones that never became # healthy. Returns non-zero if any of them failed. verify_workloads() { local failed_file="$1" shift [ "$#" -gt 0 ] || return 0 : >"$failed_file" local running=0 pid kind ns name local -a pids=() for entry in "$@"; do read -r kind ns name <<<"$entry" ( if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then echo " ok: ${kind}/${ns}/${name}" else echo " FAILED: ${kind}/${ns}/${name}" printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file" fi ) & pids+=($!) running=$((running + 1)) if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then wait -n 2>/dev/null || true running=$((running - 1)) fi done for pid in ${pids[@]+"${pids[@]}"}; do wait "$pid" || true done # Non-zero when the file holds at least one failure, i.e. a workload never # became healthy. `[ -s ]` alone is the opposite test and silently disabled # every rollback this stage is meant to perform. [ ! -s "$failed_file" ] } # rollback_workloads # Restores the previous revision of every failed workload and waits for it to # settle. Prints a report and returns non-zero if any workload is still unhealthy, # so the operator knows manual recovery is required. rollback_workloads() { local failed_file="$1" snapshot kind ns name index=0 running=0 pid revision uid local -a pids=() snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")" while read -r kind ns name; do [[ "$kind" =~ ^(deployment|statefulset|daemonset)$ ]] || continue index=$((index + 1)) ( if kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.annotations}' | grep -q 'meta.helm.sh/release-name'; then echo " skip (Helm recovery owns this workload): $kind/$ns/$name" exit 1 fi revision="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \ '.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .revision' "$snapshot/revisions.json")" uid="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \ '.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .uid' "$snapshot/revisions.json")" if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]] || [ "$uid" != "$(kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.uid}')" ]; then echo " no safe previous revision: $kind/$ns/$name (new or replaced workload)" exit 1 fi kubectl rollout undo "$kind/$name" -n "$ns" --to-revision="$revision" \ && kubectl rollout status "$kind/$name" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" ) >"$snapshot/rollback-$index.log" 2>&1 & pids+=($!) running=$((running + 1)) if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then wait -n 2>/dev/null || true running=$((running - 1)) fi done <"$failed_file" local recovered=0 unrecovered=0 i=0 for pid in "${pids[@]}"; do i=$((i + 1)) if wait "$pid"; then recovered=$((recovered + 1)); else unrecovered=$((unrecovered + 1)); fi cat "$snapshot/rollback-$i.log" done echo "ROLLED_BACK=$recovered" >>"$failed_file" echo "UNRECOVERED=$unrecovered" >>"$failed_file" [ "$unrecovered" -eq 0 ] } # Helm releases owned by this stage, one line each: # # release|chart|namespace|chart version|values file (rel. to $REPO)|active marker # # The chart version is the field Renovate keeps current. The helmv3 manager only # understands Chart.yaml and the helm-values manager only values files, so a pin # written straight into a `helm upgrade` command would never be updated: these # have to be declared as custom.regex managers in renovate/renovate.json. HELM_RELEASES=( "prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active" "victoria-operator|victoriametrics/victoria-metrics-operator|prometheus|0.68.1|prometheus-stack/k8s/victoria-operator-values.yaml|prometheus-stack/k8s/active" "loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active" "alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active" "reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active" ) # "name url" for the Helm repository hosting a chart, empty if unknown. helm_repo_for() { case "$1" in prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;; grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;; stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;; victoriametrics/*) echo "victoriametrics https://victoriametrics.github.io/helm-charts" ;; esac } # helm_release_status # Prints the release status in lowercase (deployed, failed, pending-rollback, # ...) or "not-found" when the release does not exist yet. helm_release_status() { local out if ! out="$(helm status "$1" -n "$2" 2>&1)"; then if [[ "$out" == *"release: not found"* ]]; then echo "not-found" return 0 fi printf 'ERROR: cannot read Helm status: %s\n' "$out" >&2 return 1 fi awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]' } # recover_pending_release # Rolls a release out of a pending-* state left by a failed upgrade with --rollback-on-failure # whose own rollback never completed. Without this every future upgrade errors # out until a human runs `helm rollback`. Passes through releases that are not # pending (deployed, failed, not-found). Returns non-zero when the release is # still not recoverable, so the pipeline fails loud instead of wedging. recover_pending_release() { local release="$1" namespace="$2" status revision snapshot status="$(helm_release_status "$release" "$namespace")" || return 1 case "$status" in pending-upgrade|pending-rollback|pending-install) log "Release $release is $status, rolling back to the last deployed revision" revision="" if [ -s "$DEPLOY_SNAPSHOT_DIR/current" ]; then snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")" if [ -s "$snapshot/helm-$release.json" ]; then revision="$(jq -r '.version' "$snapshot/helm-$release.json")" fi fi if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]]; then echo "ERROR: no captured Helm revision for $release; manual recovery required" return 1 fi record_apply helm-rollback "$namespace/$release" started if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then record_apply helm-rollback "$namespace/$release" failure echo "WARN: helm rollback of $release did not complete" return 1 fi status="$(helm_release_status "$release" "$namespace")" || return 1 if [ "$status" != "deployed" ]; then record_apply helm-rollback "$namespace/$release" failure echo "WARN: $release is $status after rollback" return 1 fi record_apply helm-rollback "$namespace/$release" success ;; esac return 0 } # wait_for_calm # The deploy itself is heavy enough to melt this single node (helm churn plus # apply churn drove load past 40, killed netbird/ssh, left helm pending-*). # Never pile a heavy step onto an already-hot node: wait up to 10 minutes for # the 1-minute load average to drop below the ceiling, then proceed anyway # with a warning so a permanently busy node cannot wedge the pipeline forever. wait_for_calm() { local load waited=0 while [ "$waited" -lt 600 ]; do load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)" if [ "$load" -lt 28 ]; then return 0 fi if [ "$((waited % 60))" -eq 0 ]; then log "$1: load $load, waiting for calm (<28)..." fi sleep 15 waited=$((waited + 15)) done echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway" } upgrade_helm_releases() { local entry release chart namespace version values marker repo for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do IFS='|' read -r release chart namespace version values marker <<<"$entry" if [ -n "${DEPLOY_PLAN:-}" ] && ! jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null; then echo "skip (unchanged Helm release): $release" continue fi if [ ! -f "$REPO/$values" ] && [ -f "$CONFIG_REPO/$values" ]; then values="$CONFIG_REPO/$values"; else values="$REPO/$values"; fi if [ ! -f "$REPO/$marker" ]; then echo "skip (no $marker): $release" continue fi if [ ! -f "$values" ]; then echo "ERROR: $values is gitignored but missing on the workstation, restore it first." return 1 fi repo="$(helm_repo_for "$chart")" if [ -z "$repo" ]; then echo "ERROR: no Helm repository configured for chart $chart" return 1 fi helm repo add "${repo%% *}" "${repo#* }" >/dev/null helm repo update "${repo%% *}" >/dev/null log "Upgrading $release ($chart $version)" wait_for_calm "helm $release" # A previous run with --rollback-on-failure whose own rollback never finished leaves the # release in pending-*, which blocks every future upgrade. Recover first # so one wedged revision cannot wedge the pipeline forever. if ! recover_pending_release "$release" "$namespace"; then echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand." return 1 fi # --rollback-on-failure (+ --wait) rolls the release back when the upgrade # times out or the workloads it touches never become ready, so a bad chart # bump is not left half applied. (--atomic was this combo; deprecated.) record_apply helm-upgrade "$namespace/$release" started if ! helm upgrade --install "$release" "$chart" \ --namespace "$namespace" \ --version "$version" \ --values "$values" \ --wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then record_apply helm-upgrade "$namespace/$release" failure echo "WARN: upgrade of $release failed, checking release state" # --rollback-on-failure already attempted its own rollback; finish the job when that # rollback never completed, otherwise the release stays pending-* and # blocks every future run. if ! recover_pending_release "$release" "$namespace"; then echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand." else echo "ERROR: upgrade of $release failed (release is back on its previous revision)." fi record_apply helm-recovery-state "$namespace/$release" "$(helm_release_status "$release" "$namespace" || echo unknown)" return 1 fi record_apply helm-upgrade "$namespace/$release" success done } stage_doctor() { local tool entry release chart namespace version values marker for tool in git docker kubectl helm jq curl timeout flock python3; do command -v "$tool" >/dev/null || { echo "Missing workstation tool: $tool"; return 1; } done docker compose version >/dev/null docker buildx version >/dev/null [ "$(kubectl config current-context)" = "${KUBE_CONTEXT:?configure KUBE_CONTEXT}" ] || { echo "Unexpected Kubernetes context"; return 1; } [ "$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" = "${EXPECTED_CLUSTER_UID:?configure EXPECTED_CLUSTER_UID}" ] || { echo "Unexpected Kubernetes cluster"; return 1; } kubectl get --raw=/readyz --request-timeout=10s >/dev/null [ "$(git -C "$REPO" rev-parse HEAD)" = "$DEPLOY_SHA" ] || return 1 select_manifests for entry in "${HELM_RELEASES[@]}"; do IFS='|' read -r release chart namespace version values marker <<<"$entry" [ -f "$REPO/$marker" ] || continue [ -f "$REPO/$values" ] || [ -f "$CONFIG_REPO/$values" ] || { echo "Missing values: $values"; return 1; } done jq '{sha, selected, helm, removed}' "$DEPLOY_PLAN" local cf for cf in "${COMPOSE_STACKS[@]}"; do compose "$cf" config --quiet while IFS= read -r network; do docker network inspect "$network" >/dev/null || return 1 done < <(compose "$cf" config --format json | jq -r '.networks // {} | to_entries[] | select(.value.external == true) | .value.name') python3 "$REPO/.gitea/workflows/compose-release.py" "$cf" done local image refs m k refs="$( for m in "${K8S_MANIFESTS[@]}"; do render_pinned <"$m" || return 1; done for k in "${KUSTOMIZE_APPS[@]}"; do kubectl kustomize "$k" | render_pinned || return 1; done )" || return 1 refs="$(grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+@sha256:[0-9a-f]{64}' <<<"$refs" | sort -u || true)" while IFS= read -r image; do [ -n "$image" ] || continue timeout 60s docker buildx imagetools inspect "$image" >/dev/null done <<<"$refs" } # Required pod Secrets, scoped to the resource namespace. TLS route Secrets are # created by cert-manager and are not prerequisites for applying a Certificate. check_referenced_secrets() { local m k objects refs extracted ns name local missing=() refs="" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do if skip_uninstalled_vmagent_crd "$m"; then continue fi objects="$(kubectl create --dry-run=client --validate=false -f "$m" -o json)" || return 1 extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1 refs+="$extracted"$'\n' done for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do objects="$(kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json)" || return 1 extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1 refs+="$extracted"$'\n' done while read -r ns name; do [ -n "${name:-}" ] || continue if kubectl get secret "$name" -n "$ns" -o name >/dev/null 2>&1; then echo " ok: $ns/$name" else echo " MISSING OR UNREADABLE: $ns/$name" missing+=("$ns/$name") fi done < <(printf '%s' "$refs" | sort -u) if [ "${#missing[@]}" -gt 0 ]; then echo "ERROR: required pod Secrets are missing or unreadable:" printf ' - %s\n' "${missing[@]}" echo "Create them in the listed namespaces from the service's secret example." return 1 fi } # The VMAgent CRD is installed by the VictoriaMetrics Operator Helm release in # stage_apply_k8s, after this preflight stage. Skip only its dry-run until then. skip_uninstalled_vmagent_crd() { local manifest="$1" if [[ "$manifest" == "$REPO/prometheus-stack/k8s/vmagent.yaml" ]] \ && ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then echo " skip: VMAgent CRD is installed by Helm during apply: ${manifest#"$REPO"/}" return 0 fi return 1 } stage_validate() { check_prune_mode || return 1 cd "$REPO" select_manifests local m k cf # The deploy host has the local .env and secret files. Resolve them here so # missing configuration fails before either apply job changes workloads. # CI keeps the structure-only check for inactive stacks. # shellcheck source=compose-lint.sh source "$REPO/.gitea/workflows/compose-lint.sh" log "Validate compose stacks" for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do echo " config: $cf" compose "$cf" config --quiet done log "Validate k8s manifests (kubectl dry-run=client)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do if skip_uninstalled_vmagent_crd "$m"; then continue fi kubectl apply --dry-run=client -f "$m" >/dev/null done for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do kubectl apply -k "$k" --dry-run=client >/dev/null done log "Validate k8s manifests (kubectl dry-run=server)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do if skip_uninstalled_vmagent_crd "$m"; then continue fi kubectl apply --dry-run=server -f "$m" >/dev/null done for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do kubectl apply -k "$k" --dry-run=server >/dev/null done log "Checking referenced Secrets exist" echo " (deploy never applies *secret*.yaml; create missing ones manually)" check_referenced_secrets } selected_workload_refs() { local m k for m in "${K8S_MANIFESTS[@]}"; do if skip_uninstalled_vmagent_crd "$m" >/dev/null; then continue; fi kubectl create --dry-run=client --validate=false -f "$m" -o json | jq -r ' (if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) | "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"' done for k in "${KUSTOMIZE_APPS[@]}"; do kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json | jq -r ' (if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) | "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"' done } stage_apply_k8s() { check_prune_mode || return 1 cd "$REPO" select_manifests >/dev/null local ns_files=() other_files=() m k for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do case "$m" in */namespace.yaml|*/namespace.yml) ns_files+=("$m") ;; *) other_files+=("$m") ;; esac done # Record what is about to change, and publish it for the verify job, before # the first apply. Both are fatal on failure: see snapshot_dir. selected_workload_refs >"$RUN_DIR/workload-refs" local snapshot snapshot="$(snapshot_dir)" || return 1 save_snapshot "$snapshot" || return 1 touch "$snapshot/ready" if [ "${#ns_files[@]}" -gt 0 ]; then log "Applying namespaces (${#ns_files[@]} files)" for m in "${ns_files[@]}"; do record_apply kubectl "${m#"$REPO"/}" started if ! kubectl apply -f "$m"; then record_apply kubectl "${m#"$REPO"/}" failure return 1 fi record_apply kubectl "${m#"$REPO"/}" success done fi if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then if [ ! -f "$CONFIG_REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first." exit 1 fi fi upgrade_helm_releases wait_for_calm "apply resources" if [ "${#other_files[@]}" -gt 0 ]; then log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do log "Applying ${m#"$REPO"/}" record_apply kubectl "${m#"$REPO"/}" started if ! render_pinned <"$m" | kubectl apply -f -; then record_apply kubectl "${m#"$REPO"/}" failure echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 exit 1 fi record_apply kubectl "${m#"$REPO"/}" success done fi for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)" record_apply kustomize "${k#"$REPO"/}" started if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then record_apply kustomize "${k#"$REPO"/}" failure echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2 exit 1 fi record_apply kustomize "${k#"$REPO"/}" success done # No verification here on purpose. This stage may be killed at any point by # timeout-minutes, by the runner cancelling the job, or by a dropped SSH # connection, and any code below that line would simply not run. stage_verify_k8s # picks the work up from the snapshot instead. log "Applied. Verification and rollback are the verify job's job, not this one's." } # Runs as its own workflow job, after apply-k8s (and apply-compose) are done — # including when they failed, timed out or were cancelled. Reads the baseline the # apply stage published and works out what it changed, watches those workloads, # and rolls back the ones that never became healthy. stage_verify_k8s() { local pointer="$DEPLOY_SNAPSHOT_DIR/current" local snapshot want have generations changed local -a touched=() if [ ! -s "$pointer" ]; then echo "ERROR: no snapshot pointer at $pointer." echo "The apply stage died before publishing any state, so there is no baseline to" echo "tell which workloads it touched. Nothing can be rolled back automatically —" echo "inspect the cluster by hand." return 1 fi snapshot="$(head -1 "$pointer")" if [ ! -d "$snapshot" ] || [ ! -f "$snapshot/ready" ]; then echo "ERROR: snapshot pointer refers to a missing directory: $snapshot" return 1 fi # Never trust the pointer blindly. If the apply stage was killed before it # published its own snapshot, `current` still points at the previous deploy's # baseline. Verifying against that would watch the wrong workloads and the # rollback would revert the wrong revisions, so refuse instead. want="${DEPLOY_SHA:-}" if [ -z "$want" ]; then want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)" fi have="$(cat "$snapshot/commit" 2>/dev/null || true)" if [ -z "$want" ] || [ "$have" != "$want" ]; then echo "ERROR: refusing to verify or roll back against a stale snapshot." echo " snapshot: $snapshot" echo " snapshot commit: ${have:-}" echo " deploy commit: ${want:-}" return 1 fi echo " snapshot: $snapshot (commit ${have:0:12})" local entry release chart namespace version values marker for entry in "${HELM_RELEASES[@]}"; do IFS='|' read -r release chart namespace version values marker <<<"$entry" jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null || continue recover_pending_release "$release" "$namespace" || return 1 done generations="$snapshot/generations.before" if [ ! -s "$generations" ]; then # Without a baseline we cannot tell which workloads the apply touched, so # fall back to watching everything rather than silently skipping the check. warn "no pre-apply baseline, verifying every workload in the cluster" : >"$generations" fi changed="$(changed_workloads "$generations")" || return 1 while read -r kind ns name; do [ -n "${kind:-}" ] && touched+=("$kind $ns $name") done <<<"$changed" log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)" if [ "${#touched[@]}" -eq 0 ]; then echo " nothing to verify" return 0 fi printf ' watching: %s\n' "${touched[@]/#/ }" local failed_file="$snapshot/failed-workloads" if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then echo echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:" grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /' echo log "Rolling back to the previous revision" if rollback_workloads "$failed_file"; then echo echo "Rolled back successfully. The cluster is back on the pre-deploy revision." echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and" echo "externally created resources from this commit are still in place. Review the failed" echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)." else echo echo "Rollback did NOT fully recover the cluster. Manual intervention required:" grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /' echo "Pre-apply snapshot: $snapshot" fi return 1 fi } # verify_compose_stack # `docker compose up -d` exits 0 as soon as containers are created, so a stack can # come back broken with a green pipeline. Require every long-running service to # actually be running. verify_compose_stack() { local cf="$1" local expected running missing=() expected="$(compose "$cf" config --format json | jq -r ' .services | to_entries[] | select(.value.restart != "no") | .key' | sort)" || return 1 running="$(compose "$cf" ps --status running --services | sort)" || return 1 [ -n "$expected" ] || return 0 while IFS= read -r svc; do [ -n "$svc" ] || continue # restart:"no" services are allowed to have exited. if ! printf '%s\n' "$running" | grep -qx "$svc"; then missing+=("$svc") fi done <<<"$expected" if [ "${#missing[@]}" -gt 0 ]; then echo " NOT RUNNING: ${missing[*]}" compose "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true return 1 fi echo " all ${#expected} service(s) running" return 0 } # The public hostname of every active service, one per line. # # Comments are stripped first, and deliberately so: a route that someone # disabled by commenting it out is not a service to probe, and naio and xui are # both still in the tree that way. A `#` only starts a comment when it is at the # start of a line or after whitespace, so `s/#.*//` alone would also cut a # legitimate value in half. # # Only the public names. The *.internal names are the same Traefik and the same # Services, reached by a different label, so probing both would double the run # to learn the same thing. The public name is also the one a user types. smoke_hosts() { local m k # The backticks below are literal. They are Traefik's Host() delimiter, and the # single quotes are precisely what keeps the shell from reading them as a # command substitution, so the warning is the opposite of a real problem. # shellcheck disable=SC2016 { for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do [ -f "$m" ] && cat "$m" done for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do kubectl kustomize "$k" 2>/dev/null || true done } | sed -E 's/(^|[[:space:]])#.*$//' \ | grep -oE 'Host\(`[^`]+`\)' \ | sed -E 's/^Host\(`//; s/`\)$//' \ | grep -E '(^|\.)forust\.xyz$' \ | grep -v '\${' \ | sort -u } # Traefik's own list of the routes it actually built. The Kubernetes CRs are the # wrong source for this: when a middleware fails to load, Traefik drops the # router that referenced it and leaves the CR behind looking perfectly healthy. # # api.insecure is already on for the internal `traefik` entrypoint, but the pod # IP is not routable from the node, so read it through kubectl exec rather than # standing up a port-forward. HTTP only: the TCP routers match on HostSNI(`*`) # and the UDP ones carry no rule at all, both selected by entrypoint and port, # so neither can answer whether a given host has a route. traefik_http_routes() { kubectl -n traefik exec deploy/traefik -- \ wget -qO- --timeout=10 http://127.0.0.1:8080/api/http/routers 2>/dev/null \ | jq -c '[.[] | {status, rule: (.rule // "")}]' } # The hosts Traefik currently routes to, one per line. Every backticked token of # an enabled rule counts, which is a superset of the hosts -- PathPrefix values # land here too, harmlessly -- but it keeps the host syntax in one place instead # of a matcher per host. The scan keeps the delimiters, so strip them: what # belongs in a comparison against a hostname is the bare name. traefik_routed_hosts() { jq -r '[.[] | select(.status == "enabled") | (.rule // "") | scan("`[^`]+`") | ltrimstr("`") | rtrimstr("`")] | unique | .[]' <<<"$1" } # stage_verify_k8s watches the rollout, which reports that the pods converged. # It cannot tell a converged pod from a serving one: a route pointing at the # wrong port, a Service selector that matches nothing the app listens on, a 500 # from the app itself, an OOMKill loop that still counts as Available for long # enough to pass. All of those are green at the rollout level. # # So ask the thing users ask. Any HTTP response proves Traefik matched the # host, the Service resolved to a pod and the pod answered -- a 302 to a login # or a 404 from a path the service does not serve still means the chain is # intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means # the service is not serving, and only those fail the run. # # Except that a 404 is not evidence on its own. A router Traefik refused to # build answers with the same 404 and nothing behind it, so a middleware that # fails to load takes down every route that referenced it while # this stage reports `ok` for all of them. No status code separates those two # cases, so ask Traefik which routes it built and fail on the difference. stage_smoke() { cd "$REPO" if [ -n "${DEPLOY_PLAN:-}" ] && jq -e '.full_smoke' "$DEPLOY_PLAN" >/dev/null; then DEPLOY_SMOKE_ALL=true fi select_manifests >/dev/null local -a hosts=() local h code rc bad=0 while IFS= read -r h; do [ -n "$h" ] && hosts+=("$h") done < <(smoke_hosts) if [ "${#hosts[@]}" -eq 0 ]; then # Nothing to probe means the extraction broke, not that the cluster is empty. echo "No public routes in the selected components" return 0 fi log "Probing ${#hosts[@]} public route(s)" for h in "${hosts[@]}"; do code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$? if [ "$rc" -ne 0 ]; then echo " UNREACHABLE $h (curl exit $rc)" bad=1 continue fi # A glob, not a string compare. `case` on the leading digit is the only one # of these that survives a three-digit code, and the obvious expansion to # try first -- ${code%%[0-9]*} -- is empty for every input, so it silently # reports a 500 as healthy. case "$code" in 5*) echo " SERVER ERROR $h $code" bad=1 ;; 000) # curl exited 0 and still no status, so nothing on the far end replied. # Not a pass, whatever the transport thought. echo " NO RESPONSE $h" bad=1 ;; *) echo " ok $h $code" ;; esac done # Second gate. The probe above only means something if a router matched the # host in the first place, so compare the hosts we expect against the hosts # Traefik reports and fail on the difference. local routes routed if ! routes="$(traefik_http_routes)"; then echo "ERROR: could not read Traefik's router list, refusing to report success" return 1 fi routed="$(traefik_routed_hosts "$routes")" local -a unrouted=() local tries=3 while :; do unrouted=() for h in "${hosts[@]}"; do grep -qxF "$h" <<<"$routed" || unrouted+=("$h") done if [ "${#unrouted[@]}" -eq 0 ]; then break fi # A router mid-rollout is legitimately absent for a moment. A middleware # that failed to load stays absent, so waiting cannot paper over it. if [ "$tries" -le 1 ]; then break fi tries=$((tries - 1)) warn "${#unrouted[@]} host(s) have no enabled route yet, re-checking in 10s" sleep 10 if ! routes="$(traefik_http_routes)"; then break fi routed="$(traefik_routed_hosts "$routes")" done if [ "${#unrouted[@]}" -ne 0 ]; then for h in "${unrouted[@]}"; do echo " NO ROUTE $h (Traefik has no enabled router for this host)" done bad=1 fi if [ "$bad" -ne 0 ]; then echo "ERROR: at least one active service is not serving over its public route" return 1 fi echo "all ${#hosts[@]} route(s) answered and have a router" } stage_apply_compose() { cd "$REPO" select_manifests >/dev/null local cf for cf in "${COMPOSE_STACKS[@]}"; do log "Applying Compose ${cf#"$REPO"/}" record_apply compose "${cf#"$REPO"/}" started if ! compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans; then record_apply compose "${cf#"$REPO"/}" failure return 1 fi record_apply compose "${cf#"$REPO"/}" success verify_compose_stack "$cf" done echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)" } run_stage() { case "${1:?stage required}" in doctor) stage_doctor ;; validate) stage_validate ;; apply-k8s) stage_apply_k8s ;; verify-k8s) stage_verify_k8s ;; smoke) stage_smoke ;; apply-compose) stage_apply_compose ;; *) echo "ERROR: unknown stage: $1" exit 1 ;; esac }