Files
homelab/.gitea/workflows/deploy-lib.sh
T
forust 0c76426c17
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (push) Skipped
ci / Shell (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Workflows (pull_request) Failing after 8s
ci / Formatting (pull_request) Canceled after 0s
ci / Python and tests (pull_request) Canceled after 0s
ci / YAML (pull_request) Canceled after 0s
ci / Dockerfiles (pull_request) Canceled after 0s
ci / Kubernetes (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
ci / Shell (pull_request) Canceled after 7s
Show failure details in CI and deploy summaries
2026-10-06 23:44:00 +02:00

1015 lines
42 KiB
Bash

#!/usr/bin/env bash
# Workstation deploy stages; invoked by the durable controller against pinned source.
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
# source "$REPO/.gitea/workflows/deploy-lib.sh"
# run_stage "$STAGE"
# EOF
set -euo pipefail
: "${REPO:?REPO must be set}"
APPLY_PRUNE="${APPLY_PRUNE:-false}"
CONFIG_REPO="${CONFIG_REPO:-$REPO}"
# Exact SHA accepted by the CI gate for both manual and automatic deploys.
DEPLOY_SHA="${DEPLOY_SHA:-}"
# Handoff point between the apply stage (writes) and the verify stage (reads).
# Under the deploy user's own XDG state directory rather than /var/backups: the
# deploy is unprivileged, /var/backups does not exist on a minimal Arch host, and
# creating it would need root — which is why the first real deploy died here with
# "is not writable" before touching a single workload. $HOME comes from sshd.
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state}/homelab-deploy}"
# Per-workload rollout budget and how many workloads to watch at once. The whole
# apply job has its own timeout-minutes as a backstop.
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-4}"
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
log() {
echo "== $* =="
}
# Store operation results without command output or local configuration values.
record_apply() {
[ -n "${RUN_DIR:-}" ] || return 0
jq -cn --arg action "$1" --arg target "$2" --arg result "$3" \
'{action: $action, target: $target, result: $result}' >>"$RUN_DIR/apply-events.jsonl" \
|| echo 'WARNING: cannot record an apply result' >&2
return 0
}
warn() {
echo "WARNING: $*" >&2
}
# Prune needs the complete desired set in one invocation. Per-file pruning
# treats resources from the other files as absent and can delete them.
check_prune_mode() {
if [ "$APPLY_PRUNE" = "true" ]; then
echo "ERROR: APPLY_PRUNE=true is unsupported by the per-file deploy loop." >&2
echo "Disable it; remove obsolete resources explicitly after review." >&2
return 1
fi
}
collect_k8s() {
git -C "$REPO" ls-files -- "$1" \
| grep -E '\.ya?ml$' \
| grep -Ev '/overlays/' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
| sort
}
kustomize_overlay() {
if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
echo "$1/overlays/prod"
elif [ -f "$1/base/kustomization.yaml" ]; then
echo "$1/base"
elif [ -f "$1/kustomization.yaml" ]; then
echo "$1"
fi
}
selected_service() {
local kind="$1" service="$2" section=selected
[ -n "${DEPLOY_PLAN:-}" ] || return 0
if [ "${DEPLOY_SMOKE_ALL:-false}" = true ]; then section=active; fi
jq -e --arg kind "$kind" --arg service "$service" --arg section "$section" \
'.[$section][$kind] | index($service) != null' "$DEPLOY_PLAN" >/dev/null
}
# Resolve .env and relative binds on the persistent workstation tree. Locked
# JSON configs keep the same Compose project name and volume names.
compose() {
local cf="$1" locked project_dir
project_dir="$CONFIG_REPO/$(basename "$(dirname "$cf")")"
shift
locked="${RUN_DIR:-/nonexistent}/compose/$(basename "$(dirname "$cf")").json"
if [ -f "$locked" ]; then cf="$locked"; fi
(cd "$CONFIG_REPO" && docker compose --project-directory "$project_dir" -f "$cf" "$@")
}
select_manifests() {
K8S_MANIFESTS=()
KUSTOMIZE_APPS=()
COMPOSE_STACKS=()
local kd_rel kd overlay cf_rel cf f
while IFS= read -r kd_rel; do
kd="$REPO/$kd_rel"
selected_service k8s "${kd_rel%/k8s}" || continue
if [ ! -f "$kd/active" ]; then
echo "skip (no k8s/active): $kd_rel"
continue
fi
overlay="$(kustomize_overlay "$kd" || true)"
if [ -n "${overlay:-}" ]; then
echo "kustomize app: ${overlay#"$REPO"/}"
KUSTOMIZE_APPS+=("$overlay")
else
while IFS= read -r f; do
[ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f")
done < <(collect_k8s "$kd_rel" || true)
fi
done < <(
git -C "$REPO" ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
while IFS= read -r cf_rel; do
cf="$REPO/$cf_rel"
selected_service compose "$(dirname "$cf_rel")" || continue
if [ -f "$(dirname "$cf")/active" ]; then
echo "compose: $cf_rel"
COMPOSE_STACKS+=("$cf")
else
echo "skip (no root active): $cf_rel"
fi
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
}
# --- post-apply verification and rollback -------------------------------------
#
# A green `kubectl apply` says nothing about the cluster being healthy. These
# helpers watch exactly the workloads whose spec changed during this apply, and
# on failure roll them back to the revision that was running before, so a bad
# push to main cannot leave a service crash-looping.
#
# The workstation controller runs apply and verification as separate durable
# stages. Runner jobs only follow their logs. ExecStopPost recovers interrupted
# runs using the per-run snapshot, even when the SSH connection has gone away.
#
# Creates this run's snapshot directory and publishes it as the handoff point for
# the verify stage. Fails hard by design: a deploy that cannot record what it is
# about to change must not start, because then nothing can be rolled back for it
# automatically. Publishing happens before the first apply, so an apply killed
# mid-flight still leaves a usable baseline behind.
snapshot_dir() {
local stamp dir
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
echo "The verify job needs it to learn which workloads this deploy touches." >&2
echo "Refusing to deploy without a way to roll back." >&2
return 1
fi
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
echo "ERROR: cannot create snapshot dir $dir" >&2
return 1
fi
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
return 1
fi
printf '%s\n' "$dir"
}
save_snapshot() {
local dir="$1" releases revision status
log "Saving pre-apply snapshot to $dir"
workload_generations >"$dir/generations.before" || return 1
kubectl get "$WORKLOAD_KINDS" -A -o json >"$dir/workloads.json" || return 1
kubectl get controllerrevisions.apps -A -o json >"$dir/controller-revisions.json" || return 1
jq --slurpfile revisions "$dir/controller-revisions.json" '
[.items[] | . as $w | {
kind: (.kind | ascii_downcase), namespace: .metadata.namespace, name: .metadata.name, uid: .metadata.uid,
revision: (if .kind == "Deployment" then (.metadata.annotations["deployment.kubernetes.io/revision"] // "0" | tonumber)
else ([$revisions[0].items[] | select(.metadata.namespace == $w.metadata.namespace)
| select(any(.metadata.ownerReferences[]?; .uid == $w.metadata.uid))
| select($w.kind != "StatefulSet" or .metadata.name == $w.status.currentRevision) | .revision] | max // 0) end)
}]' "$dir/workloads.json" >"$dir/revisions.json" || return 1
releases="$(helm list --all -A -o json)" || return 1
for entry in "${HELM_RELEASES[@]}"; do
IFS='|' read -r release _ namespace _ _ _ <<<"$entry"
if ! jq -e --arg r "$release" --arg n "$namespace" \
'any(.[]; .name == $r and .namespace == $n)' <<<"$releases" >/dev/null; then
continue
fi
helm status "$release" -n "$namespace" -o json >"$dir/helm-$release.json" || return 1
status="$(jq -r '.info.status' "$dir/helm-$release.json")"
if [ "$status" != deployed ]; then
# Never capture a pending/failed revision as the recovery target.
helm history "$release" -n "$namespace" -o json >"$dir/helm-$release.history.json" || return 1
revision="$(jq '[.[] | select(.status == "deployed" or .status == "superseded") | .revision] | max // 0' \
"$dir/helm-$release.history.json")"
jq --argjson revision "$revision" '.version = $revision' "$dir/helm-$release.json" >"$dir/helm-$release.tmp"
mv "$dir/helm-$release.tmp" "$dir/helm-$release.json"
fi
done
# The verify stage compares this against the commit it is deploying, to refuse
# rolling back against a baseline left by an earlier run. A snapshot we cannot
# attribute to a commit is unusable for that, so fail before anything is applied.
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
return 1
fi
}
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
workload_generations() {
kubectl get "$WORKLOAD_KINDS" -A \
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
--no-headers 2>/dev/null \
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
}
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
# moved since the snapshot, i.e. the ones this apply actually touched.
changed_workloads() {
local before="$1"
local ns name kind gen old current
current="$(workload_generations)" || return 1
while read -r ns name kind gen; do
[ -n "${gen:-}" ] || continue
old="$(awk -v want_ns="$ns" -v want_name="$name" -v want_kind="$kind" \
'$1 == want_ns && $2 == want_name && $3 == want_kind { print $4; exit }' "$before" 2>/dev/null || true)"
if [ -n "${RUN_DIR:-}" ] && ! grep -qxF "$kind $ns $name" "$RUN_DIR/workload-refs"; then
continue
fi
if [ "$old" != "$gen" ]; then
printf '%s %s %s\n' "$kind" "$ns" "$name"
fi
done <<<"$current"
}
# Resolve owned image references exclusively from the checked CI artifact.
render_pinned() {
python3 "$REPO/.gitea/workflows/release.py" render
}
# verify_workloads <failed-file> <kind> <ns> <name> ...
# Watches every workload in parallel and records the ones that never became
# healthy. Returns non-zero if any of them failed.
verify_workloads() {
local failed_file="$1"
shift
[ "$#" -gt 0 ] || return 0
: >"$failed_file"
local running=0 pid kind ns name
local -a pids=()
for entry in "$@"; do
read -r kind ns name <<<"$entry"
(
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " ok: ${kind}/${ns}/${name}"
else
echo " FAILED: ${kind}/${ns}/${name}"
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
fi
) &
pids+=($!)
running=$((running + 1))
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
wait -n 2>/dev/null || true
running=$((running - 1))
fi
done
for pid in ${pids[@]+"${pids[@]}"}; do
wait "$pid" || true
done
# Non-zero when the file holds at least one failure, i.e. a workload never
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
# every rollback this stage is meant to perform.
[ ! -s "$failed_file" ]
}
# rollback_workloads <failed-file>
# Restores the previous revision of every failed workload and waits for it to
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
# so the operator knows manual recovery is required.
rollback_workloads() {
local failed_file="$1" snapshot kind ns name index=0 running=0 pid revision uid
local -a pids=()
snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")"
while read -r kind ns name; do
[[ "$kind" =~ ^(deployment|statefulset|daemonset)$ ]] || continue
index=$((index + 1))
(
if kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.annotations}' | grep -q 'meta.helm.sh/release-name'; then
echo " skip (Helm recovery owns this workload): $kind/$ns/$name"
exit 1
fi
revision="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \
'.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .revision' "$snapshot/revisions.json")"
uid="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \
'.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .uid' "$snapshot/revisions.json")"
if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]] || [ "$uid" != "$(kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.uid}')" ]; then
echo " no safe previous revision: $kind/$ns/$name (new or replaced workload)"
exit 1
fi
kubectl rollout undo "$kind/$name" -n "$ns" --to-revision="$revision" \
&& kubectl rollout status "$kind/$name" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s"
) >"$snapshot/rollback-$index.log" 2>&1 &
pids+=($!)
running=$((running + 1))
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
wait -n 2>/dev/null || true
running=$((running - 1))
fi
done <"$failed_file"
local recovered=0 unrecovered=0 i=0
for pid in "${pids[@]}"; do
i=$((i + 1))
if wait "$pid"; then recovered=$((recovered + 1)); else unrecovered=$((unrecovered + 1)); fi
cat "$snapshot/rollback-$i.log"
done
echo "ROLLED_BACK=$recovered" >>"$failed_file"
echo "UNRECOVERED=$unrecovered" >>"$failed_file"
[ "$unrecovered" -eq 0 ]
}
# Helm releases owned by this stage, one line each:
#
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
#
# The chart version is the field Renovate keeps current. The helmv3 manager only
# understands Chart.yaml and the helm-values manager only values files, so a pin
# written straight into a `helm upgrade` command would never be updated: these
# have to be declared as custom.regex managers in renovate/renovate.json.
HELM_RELEASES=(
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
"victoria-operator|victoriametrics/victoria-metrics-operator|prometheus|0.68.1|prometheus-stack/k8s/victoria-operator-values.yaml|prometheus-stack/k8s/active"
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
"reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active"
)
# "name url" for the Helm repository hosting a chart, empty if unknown.
helm_repo_for() {
case "$1" in
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;;
victoriametrics/*) echo "victoriametrics https://victoriametrics.github.io/helm-charts" ;;
esac
}
# helm_release_status <release> <namespace>
# Prints the release status in lowercase (deployed, failed, pending-rollback,
# ...) or "not-found" when the release does not exist yet.
helm_release_status() {
local out
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
if [[ "$out" == *"release: not found"* ]]; then
echo "not-found"
return 0
fi
printf 'ERROR: cannot read Helm status: %s\n' "$out" >&2
return 1
fi
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
}
# recover_pending_release <release> <namespace>
# Rolls a release out of a pending-* state left by a failed upgrade with --rollback-on-failure
# whose own rollback never completed. Without this every future upgrade errors
# out until a human runs `helm rollback`. Passes through releases that are not
# pending (deployed, failed, not-found). Returns non-zero when the release is
# still not recoverable, so the pipeline fails loud instead of wedging.
recover_pending_release() {
local release="$1" namespace="$2" status revision snapshot
status="$(helm_release_status "$release" "$namespace")" || return 1
case "$status" in
pending-upgrade|pending-rollback|pending-install)
log "Release $release is $status, rolling back to the last deployed revision"
revision=""
if [ -s "$DEPLOY_SNAPSHOT_DIR/current" ]; then
snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")"
if [ -s "$snapshot/helm-$release.json" ]; then
revision="$(jq -r '.version' "$snapshot/helm-$release.json")"
fi
fi
if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]]; then
echo "ERROR: no captured Helm revision for $release; manual recovery required"
return 1
fi
record_apply helm-rollback "$namespace/$release" started
if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then
record_apply helm-rollback "$namespace/$release" failure
echo "WARN: helm rollback of $release did not complete"
return 1
fi
status="$(helm_release_status "$release" "$namespace")" || return 1
if [ "$status" != "deployed" ]; then
record_apply helm-rollback "$namespace/$release" failure
echo "WARN: $release is $status after rollback"
return 1
fi
record_apply helm-rollback "$namespace/$release" success
;;
esac
return 0
}
# wait_for_calm <stage>
# The deploy itself is heavy enough to melt this single node (helm churn plus
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
# the 1-minute load average to drop below the ceiling, then proceed anyway
# with a warning so a permanently busy node cannot wedge the pipeline forever.
wait_for_calm() {
local load waited=0
while [ "$waited" -lt 600 ]; do
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
if [ "$load" -lt 28 ]; then
return 0
fi
if [ "$((waited % 60))" -eq 0 ]; then
log "$1: load $load, waiting for calm (<28)..."
fi
sleep 15
waited=$((waited + 15))
done
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
}
upgrade_helm_releases() {
local entry release chart namespace version values marker repo
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
IFS='|' read -r release chart namespace version values marker <<<"$entry"
if [ -n "${DEPLOY_PLAN:-}" ] && ! jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null; then
echo "skip (unchanged Helm release): $release"
continue
fi
if [ ! -f "$REPO/$values" ] && [ -f "$CONFIG_REPO/$values" ]; then values="$CONFIG_REPO/$values"; else values="$REPO/$values"; fi
if [ ! -f "$REPO/$marker" ]; then
echo "skip (no $marker): $release"
continue
fi
if [ ! -f "$values" ]; then
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
return 1
fi
repo="$(helm_repo_for "$chart")"
if [ -z "$repo" ]; then
echo "ERROR: no Helm repository configured for chart $chart"
return 1
fi
helm repo add "${repo%% *}" "${repo#* }" >/dev/null
helm repo update "${repo%% *}" >/dev/null
log "Upgrading $release ($chart $version)"
wait_for_calm "helm $release"
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
# release in pending-*, which blocks every future upgrade. Recover first
# so one wedged revision cannot wedge the pipeline forever.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
return 1
fi
# --rollback-on-failure (+ --wait) rolls the release back when the upgrade
# times out or the workloads it touches never become ready, so a bad chart
# bump is not left half applied. (--atomic was this combo; deprecated.)
record_apply helm-upgrade "$namespace/$release" started
if ! helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \
--version "$version" \
--values "$values" \
--wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then
record_apply helm-upgrade "$namespace/$release" failure
echo "WARN: upgrade of $release failed, checking release state"
# --rollback-on-failure already attempted its own rollback; finish the job when that
# rollback never completed, otherwise the release stays pending-* and
# blocks every future run.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
else
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
fi
record_apply helm-recovery-state "$namespace/$release" "$(helm_release_status "$release" "$namespace" || echo unknown)"
return 1
fi
record_apply helm-upgrade "$namespace/$release" success
done
}
stage_doctor() {
local tool entry release chart namespace version values marker
for tool in git docker kubectl helm jq curl timeout flock python3; do
command -v "$tool" >/dev/null || { echo "Missing workstation tool: $tool"; return 1; }
done
docker compose version >/dev/null
docker buildx version >/dev/null
[ "$(kubectl config current-context)" = "${KUBE_CONTEXT:?configure KUBE_CONTEXT}" ] || { echo "Unexpected Kubernetes context"; return 1; }
[ "$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" = "${EXPECTED_CLUSTER_UID:?configure EXPECTED_CLUSTER_UID}" ] || { echo "Unexpected Kubernetes cluster"; return 1; }
kubectl get --raw=/readyz --request-timeout=10s >/dev/null
[ "$(git -C "$REPO" rev-parse HEAD)" = "$DEPLOY_SHA" ] || return 1
select_manifests
for entry in "${HELM_RELEASES[@]}"; do
IFS='|' read -r release chart namespace version values marker <<<"$entry"
[ -f "$REPO/$marker" ] || continue
[ -f "$REPO/$values" ] || [ -f "$CONFIG_REPO/$values" ] || { echo "Missing values: $values"; return 1; }
done
jq '{sha, selected, helm, removed}' "$DEPLOY_PLAN"
local cf
for cf in "${COMPOSE_STACKS[@]}"; do
compose "$cf" config --quiet
while IFS= read -r network; do
docker network inspect "$network" >/dev/null || return 1
done < <(compose "$cf" config --format json | jq -r '.networks // {} | to_entries[] | select(.value.external == true) | .value.name')
python3 "$REPO/.gitea/workflows/compose-release.py" "$cf"
done
local image refs m k
refs="$(
for m in "${K8S_MANIFESTS[@]}"; do render_pinned <"$m" || return 1; done
for k in "${KUSTOMIZE_APPS[@]}"; do kubectl kustomize "$k" | render_pinned || return 1; done
)" || return 1
refs="$(grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+@sha256:[0-9a-f]{64}' <<<"$refs" | sort -u || true)"
while IFS= read -r image; do
[ -n "$image" ] || continue
timeout 60s docker buildx imagetools inspect "$image" >/dev/null
done <<<"$refs"
}
# Required pod Secrets, scoped to the resource namespace. TLS route Secrets are
# created by cert-manager and are not prerequisites for applying a Certificate.
check_referenced_secrets() {
local m k objects refs extracted ns name
local missing=()
refs=""
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
if skip_uninstalled_vmagent_crd "$m"; then
continue
fi
objects="$(kubectl create --dry-run=client --validate=false -f "$m" -o json)" || return 1
extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1
refs+="$extracted"$'\n'
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
objects="$(kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json)" || return 1
extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1
refs+="$extracted"$'\n'
done
while read -r ns name; do
[ -n "${name:-}" ] || continue
if kubectl get secret "$name" -n "$ns" -o name >/dev/null 2>&1; then
echo " ok: $ns/$name"
else
echo " MISSING OR UNREADABLE: $ns/$name"
missing+=("$ns/$name")
fi
done < <(printf '%s' "$refs" | sort -u)
if [ "${#missing[@]}" -gt 0 ]; then
echo "ERROR: required pod Secrets are missing or unreadable:"
printf ' - %s\n' "${missing[@]}"
echo "Create them in the listed namespaces from the service's secret example."
return 1
fi
}
# The VMAgent CRD is installed by the VictoriaMetrics Operator Helm release in
# stage_apply_k8s, after this preflight stage. Skip only its dry-run until then.
skip_uninstalled_vmagent_crd() {
local manifest="$1"
if [[ "$manifest" == "$REPO/prometheus-stack/k8s/vmagent.yaml" ]] \
&& ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then
echo " skip: VMAgent CRD is installed by Helm during apply: ${manifest#"$REPO"/}"
return 0
fi
return 1
}
stage_validate() {
check_prune_mode || return 1
cd "$REPO"
select_manifests
local m k cf
# The deploy host has the local .env and secret files. Resolve them here so
# missing configuration fails before either apply job changes workloads.
# CI keeps the structure-only check for inactive stacks.
# shellcheck source=compose-lint.sh
source "$REPO/.gitea/workflows/compose-lint.sh"
log "Validate compose stacks"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " config: $cf"
compose "$cf" config --quiet
done
log "Validate k8s manifests (kubectl dry-run=client)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
if skip_uninstalled_vmagent_crd "$m"; then
continue
fi
kubectl apply --dry-run=client -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=client >/dev/null
done
log "Validate k8s manifests (kubectl dry-run=server)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
if skip_uninstalled_vmagent_crd "$m"; then
continue
fi
kubectl apply --dry-run=server -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=server >/dev/null
done
log "Checking referenced Secrets exist"
echo " (deploy never applies *secret*.yaml; create missing ones manually)"
check_referenced_secrets
}
selected_workload_refs() {
local m k
for m in "${K8S_MANIFESTS[@]}"; do
if skip_uninstalled_vmagent_crd "$m" >/dev/null; then continue; fi
kubectl create --dry-run=client --validate=false -f "$m" -o json | jq -r '
(if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
| "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"'
done
for k in "${KUSTOMIZE_APPS[@]}"; do
kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json | jq -r '
(if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
| "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"'
done
}
stage_apply_k8s() {
check_prune_mode || return 1
cd "$REPO"
select_manifests >/dev/null
local ns_files=() other_files=() m k
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
case "$m" in
*/namespace.yaml|*/namespace.yml) ns_files+=("$m") ;;
*) other_files+=("$m") ;;
esac
done
# Record what is about to change, and publish it for the verify job, before
# the first apply. Both are fatal on failure: see snapshot_dir.
selected_workload_refs >"$RUN_DIR/workload-refs"
local snapshot
snapshot="$(snapshot_dir)" || return 1
save_snapshot "$snapshot" || return 1
touch "$snapshot/ready"
if [ "${#ns_files[@]}" -gt 0 ]; then
log "Applying namespaces (${#ns_files[@]} files)"
for m in "${ns_files[@]}"; do
record_apply kubectl "${m#"$REPO"/}" started
if ! kubectl apply -f "$m"; then
record_apply kubectl "${m#"$REPO"/}" failure
return 1
fi
record_apply kubectl "${m#"$REPO"/}" success
done
fi
if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then
if [ ! -f "$CONFIG_REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
exit 1
fi
fi
upgrade_helm_releases
wait_for_calm "apply resources"
if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
for m in "${other_files[@]}"; do
log "Applying ${m#"$REPO"/}"
record_apply kubectl "${m#"$REPO"/}" started
if ! render_pinned <"$m" | kubectl apply -f -; then
record_apply kubectl "${m#"$REPO"/}" failure
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
exit 1
fi
record_apply kubectl "${m#"$REPO"/}" success
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
record_apply kustomize "${k#"$REPO"/}" started
if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then
record_apply kustomize "${k#"$REPO"/}" failure
echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2
exit 1
fi
record_apply kustomize "${k#"$REPO"/}" success
done
# No verification here on purpose. This stage may be killed at any point by
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
# connection, and any code below that line would simply not run. stage_verify_k8s
# picks the work up from the snapshot instead.
log "Applied. Verification and rollback are the verify job's job, not this one's."
}
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
# including when they failed, timed out or were cancelled. Reads the baseline the
# apply stage published and works out what it changed, watches those workloads,
# and rolls back the ones that never became healthy.
stage_verify_k8s() {
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
local snapshot want have generations changed
local -a touched=()
if [ ! -s "$pointer" ]; then
echo "ERROR: no snapshot pointer at $pointer."
echo "The apply stage died before publishing any state, so there is no baseline to"
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
echo "inspect the cluster by hand."
return 1
fi
snapshot="$(head -1 "$pointer")"
if [ ! -d "$snapshot" ] || [ ! -f "$snapshot/ready" ]; then
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
return 1
fi
# Never trust the pointer blindly. If the apply stage was killed before it
# published its own snapshot, `current` still points at the previous deploy's
# baseline. Verifying against that would watch the wrong workloads and the
# rollback would revert the wrong revisions, so refuse instead.
want="${DEPLOY_SHA:-}"
if [ -z "$want" ]; then
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
fi
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
if [ -z "$want" ] || [ "$have" != "$want" ]; then
echo "ERROR: refusing to verify or roll back against a stale snapshot."
echo " snapshot: $snapshot"
echo " snapshot commit: ${have:-<missing>}"
echo " deploy commit: ${want:-<unknown>}"
return 1
fi
echo " snapshot: $snapshot (commit ${have:0:12})"
local entry release chart namespace version values marker
for entry in "${HELM_RELEASES[@]}"; do
IFS='|' read -r release chart namespace version values marker <<<"$entry"
jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null || continue
recover_pending_release "$release" "$namespace" || return 1
done
generations="$snapshot/generations.before"
if [ ! -s "$generations" ]; then
# Without a baseline we cannot tell which workloads the apply touched, so
# fall back to watching everything rather than silently skipping the check.
warn "no pre-apply baseline, verifying every workload in the cluster"
: >"$generations"
fi
changed="$(changed_workloads "$generations")" || return 1
while read -r kind ns name; do
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
done <<<"$changed"
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
if [ "${#touched[@]}" -eq 0 ]; then
echo " nothing to verify"
return 0
fi
printf ' watching: %s\n' "${touched[@]/#/ }"
local failed_file="$snapshot/failed-workloads"
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
echo
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
echo
log "Rolling back to the previous revision"
if rollback_workloads "$failed_file"; then
echo
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
echo "externally created resources from this commit are still in place. Review the failed"
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
else
echo
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
echo "Pre-apply snapshot: $snapshot"
fi
return 1
fi
}
# verify_compose_stack <compose-file>
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
# come back broken with a green pipeline. Require every long-running service to
# actually be running.
verify_compose_stack() {
local cf="$1"
local expected running missing=()
expected="$(compose "$cf" config --format json | jq -r ' .services | to_entries[] | select(.value.restart != "no") | .key' | sort)" || return 1
running="$(compose "$cf" ps --status running --services | sort)" || return 1
[ -n "$expected" ] || return 0
while IFS= read -r svc; do
[ -n "$svc" ] || continue
# restart:"no" services are allowed to have exited.
if ! printf '%s\n' "$running" | grep -qx "$svc"; then
missing+=("$svc")
fi
done <<<"$expected"
if [ "${#missing[@]}" -gt 0 ]; then
echo " NOT RUNNING: ${missing[*]}"
compose "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
return 1
fi
echo " all ${#expected} service(s) running"
return 0
}
# The public hostname of every active service, one per line.
#
# Comments are stripped first, and deliberately so: a route that someone
# disabled by commenting it out is not a service to probe, and naio and xui are
# both still in the tree that way. A `#` only starts a comment when it is at the
# start of a line or after whitespace, so `s/#.*//` alone would also cut a
# legitimate value in half.
#
# Only the public names. The *.internal names are the same Traefik and the same
# Services, reached by a different label, so probing both would double the run
# to learn the same thing. The public name is also the one a user types.
smoke_hosts() {
local m k
# The backticks below are literal. They are Traefik's Host() delimiter, and the
# single quotes are precisely what keeps the shell from reading them as a
# command substitution, so the warning is the opposite of a real problem.
# shellcheck disable=SC2016
{
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
[ -f "$m" ] && cat "$m"
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl kustomize "$k" 2>/dev/null || true
done
} | sed -E 's/(^|[[:space:]])#.*$//' \
| grep -oE 'Host\(`[^`]+`\)' \
| sed -E 's/^Host\(`//; s/`\)$//' \
| grep -E '(^|\.)forust\.xyz$' \
| grep -v '\${' \
| sort -u
}
# Traefik's own list of the routes it actually built. The Kubernetes CRs are the
# wrong source for this: when a middleware fails to load, Traefik drops the
# router that referenced it and leaves the CR behind looking perfectly healthy.
#
# api.insecure is already on for the internal `traefik` entrypoint, but the pod
# IP is not routable from the node, so read it through kubectl exec rather than
# standing up a port-forward. HTTP only: the TCP routers match on HostSNI(`*`)
# and the UDP ones carry no rule at all, both selected by entrypoint and port,
# so neither can answer whether a given host has a route.
traefik_http_routes() {
kubectl -n traefik exec deploy/traefik -- \
wget -qO- --timeout=10 http://127.0.0.1:8080/api/http/routers 2>/dev/null \
| jq -c '[.[] | {status, rule: (.rule // "")}]'
}
# The hosts Traefik currently routes to, one per line. Every backticked token of
# an enabled rule counts, which is a superset of the hosts -- PathPrefix values
# land here too, harmlessly -- but it keeps the host syntax in one place instead
# of a matcher per host. The scan keeps the delimiters, so strip them: what
# belongs in a comparison against a hostname is the bare name.
traefik_routed_hosts() {
jq -r '[.[] | select(.status == "enabled") | (.rule // "")
| scan("`[^`]+`") | ltrimstr("`") | rtrimstr("`")]
| unique | .[]' <<<"$1"
}
# stage_verify_k8s watches the rollout, which reports that the pods converged.
# It cannot tell a converged pod from a serving one: a route pointing at the
# wrong port, a Service selector that matches nothing the app listens on, a 500
# from the app itself, an OOMKill loop that still counts as Available for long
# enough to pass. All of those are green at the rollout level.
#
# So ask the thing users ask. Any HTTP response proves Traefik matched the
# host, the Service resolved to a pod and the pod answered -- a 302 to a login
# or a 404 from a path the service does not serve still means the chain is
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
# the service is not serving, and only those fail the run.
#
# Except that a 404 is not evidence on its own. A router Traefik refused to
# build answers with the same 404 and nothing behind it, so a middleware that
# fails to load takes down every route that referenced it while
# this stage reports `ok` for all of them. No status code separates those two
# cases, so ask Traefik which routes it built and fail on the difference.
stage_smoke() {
cd "$REPO"
if [ -n "${DEPLOY_PLAN:-}" ] && jq -e '.full_smoke' "$DEPLOY_PLAN" >/dev/null; then
DEPLOY_SMOKE_ALL=true
fi
select_manifests >/dev/null
local -a hosts=()
local h code rc bad=0
while IFS= read -r h; do
[ -n "$h" ] && hosts+=("$h")
done < <(smoke_hosts)
if [ "${#hosts[@]}" -eq 0 ]; then
# Nothing to probe means the extraction broke, not that the cluster is empty.
echo "No public routes in the selected components"
return 0
fi
log "Probing ${#hosts[@]} public route(s)"
for h in "${hosts[@]}"; do
code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$?
if [ "$rc" -ne 0 ]; then
echo " UNREACHABLE $h (curl exit $rc)"
bad=1
continue
fi
# A glob, not a string compare. `case` on the leading digit is the only one
# of these that survives a three-digit code, and the obvious expansion to
# try first -- ${code%%[0-9]*} -- is empty for every input, so it silently
# reports a 500 as healthy.
case "$code" in
5*)
echo " SERVER ERROR $h $code"
bad=1
;;
000)
# curl exited 0 and still no status, so nothing on the far end replied.
# Not a pass, whatever the transport thought.
echo " NO RESPONSE $h"
bad=1
;;
*)
echo " ok $h $code"
;;
esac
done
# Second gate. The probe above only means something if a router matched the
# host in the first place, so compare the hosts we expect against the hosts
# Traefik reports and fail on the difference.
local routes routed
if ! routes="$(traefik_http_routes)"; then
echo "ERROR: could not read Traefik's router list, refusing to report success"
return 1
fi
routed="$(traefik_routed_hosts "$routes")"
local -a unrouted=()
local tries=3
while :; do
unrouted=()
for h in "${hosts[@]}"; do
grep -qxF "$h" <<<"$routed" || unrouted+=("$h")
done
if [ "${#unrouted[@]}" -eq 0 ]; then
break
fi
# A router mid-rollout is legitimately absent for a moment. A middleware
# that failed to load stays absent, so waiting cannot paper over it.
if [ "$tries" -le 1 ]; then
break
fi
tries=$((tries - 1))
warn "${#unrouted[@]} host(s) have no enabled route yet, re-checking in 10s"
sleep 10
if ! routes="$(traefik_http_routes)"; then
break
fi
routed="$(traefik_routed_hosts "$routes")"
done
if [ "${#unrouted[@]}" -ne 0 ]; then
for h in "${unrouted[@]}"; do
echo " NO ROUTE $h (Traefik has no enabled router for this host)"
done
bad=1
fi
if [ "$bad" -ne 0 ]; then
echo "ERROR: at least one active service is not serving over its public route"
return 1
fi
echo "all ${#hosts[@]} route(s) answered and have a router"
}
stage_apply_compose() {
cd "$REPO"
select_manifests >/dev/null
local cf
for cf in "${COMPOSE_STACKS[@]}"; do
log "Applying Compose ${cf#"$REPO"/}"
record_apply compose "${cf#"$REPO"/}" started
if ! compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans; then
record_apply compose "${cf#"$REPO"/}" failure
return 1
fi
record_apply compose "${cf#"$REPO"/}" success
verify_compose_stack "$cf"
done
echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)"
}
run_stage() {
case "${1:?stage required}" in
doctor) stage_doctor ;;
validate) stage_validate ;;
apply-k8s) stage_apply_k8s ;;
verify-k8s) stage_verify_k8s ;;
smoke) stage_smoke ;;
apply-compose) stage_apply_compose ;;
*)
echo "ERROR: unknown stage: $1"
exit 1
;;
esac
}