Files
homelab/.gitea/workflows/deploy-lib.sh
T
forust 0859479c0f
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 12s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Failing after 14m22s
feat(ingress): replace traefik crowdsec plugin with firewall bouncer
Move L3 enforcement to the host firewall-bouncer (systemd, nftables): drop the Traefik plugin, its secrets volume and the crowdsec Middleware, remove bouncer refs from all IngressRoutes. Disable the http-generic-bf scenario (403-burst bans hurt legit automation under L3 enforcement). Add a Gateway API PoC for homepages prod and CrowdSec PrometheusRule alerts.
2026-09-30 20:14:25 +02:00

1140 lines
45 KiB
Bash

#!/usr/bin/env bash
# Shared stages for the deploy workflow. Runs on the workstation, invoked as:
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
# source "$REPO/.gitea/workflows/deploy-lib.sh"
# run_stage "$STAGE"
# EOF
set -euo pipefail
: "${REPO:?REPO must be set}"
APPLY_PRUNE="${APPLY_PRUNE:-false}"
# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to
# the current origin/main.
DEPLOY_SHA="${DEPLOY_SHA:-}"
# Handoff point between the apply stage (writes) and the verify stage (reads).
# Under the deploy user's own XDG state directory rather than /var/backups: the
# deploy is unprivileged, /var/backups does not exist on a minimal Arch host, and
# creating it would need root — which is why the first real deploy died here with
# "is not writable" before touching a single workload. $HOME comes from sshd.
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state}/homelab-deploy}"
# Per-workload rollout budget and how many workloads to watch at once. The whole
# apply job has its own timeout-minutes as a backstop.
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}"
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
log() {
echo "== $* =="
}
warn() {
echo "WARNING: $*" >&2
}
collect_k8s() {
git -C "$REPO" ls-files -- "$1" \
| grep -E '\.ya?ml$' \
| grep -Ev '/overlays/' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
| sort
}
kustomize_overlay() {
if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
echo "$1/overlays/prod"
elif [ -f "$1/base/kustomization.yaml" ]; then
echo "$1/base"
elif [ -f "$1/kustomization.yaml" ]; then
echo "$1"
fi
}
select_manifests() {
K8S_MANIFESTS=()
KUSTOMIZE_APPS=()
COMPOSE_STACKS=()
local kd_rel kd overlay cf_rel cf f
while IFS= read -r kd_rel; do
kd="$REPO/$kd_rel"
if [ ! -f "$kd/active" ]; then
echo "skip (no k8s/active): $kd_rel"
continue
fi
overlay="$(kustomize_overlay "$kd" || true)"
if [ -n "${overlay:-}" ]; then
echo "kustomize app: ${overlay#"$REPO"/}"
KUSTOMIZE_APPS+=("$overlay")
else
while IFS= read -r f; do
[ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f")
done < <(collect_k8s "$kd_rel" || true)
fi
done < <(
git -C "$REPO" ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
while IFS= read -r cf_rel; do
cf="$REPO/$cf_rel"
if [ -f "$(dirname "$cf")/active" ]; then
echo "compose: $cf_rel"
COMPOSE_STACKS+=("$cf")
else
echo "skip (no root active): $cf_rel"
fi
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
}
# --- post-apply verification and rollback -------------------------------------
#
# A green `kubectl apply` says nothing about the cluster being healthy. These
# helpers watch exactly the workloads whose spec changed during this apply, and
# on failure roll them back to the revision that was running before, so a bad
# push to main cannot leave a service crash-looping.
#
# Verification lives in its own workflow job, not at the end of the apply stage.
# Inside a single process it is worthless exactly when it is needed most: a job
# killed by timeout-minutes or cancelled mid-apply never reaches the rollback
# code, and leaves a half-applied cluster behind. Split out, the apply job can
# die in any way and the verify job still runs.
#
# That split needs a handoff point on the workstation, because the two stages are
# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written
# before anything is applied, read by the verify stage afterwards.
# Creates this run's snapshot directory and publishes it as the handoff point for
# the verify stage. Fails hard by design: a deploy that cannot record what it is
# about to change must not start, because then nothing can be rolled back for it
# automatically. Publishing happens before the first apply, so an apply killed
# mid-flight still leaves a usable baseline behind.
snapshot_dir() {
local stamp dir
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
echo "The verify job needs it to learn which workloads this deploy touches." >&2
echo "Refusing to deploy without a way to roll back." >&2
return 1
fi
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
echo "ERROR: cannot create snapshot dir $dir" >&2
return 1
fi
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
return 1
fi
printf '%s\n' "$dir"
}
save_snapshot() {
local dir="$1"
log "Saving pre-apply snapshot to $dir"
workload_generations >"$dir/generations.before" 2>/dev/null \
|| warn "could not snapshot workload generations"
kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \
|| warn "could not snapshot workloads"
for release in prometheus-stack loki alloy; do
if helm status "$release" -n prometheus >/dev/null 2>&1; then
{
echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)"
helm get values "$release" -n prometheus --all 2>/dev/null
} >"$dir/helm-$release.txt"
fi
done
# The verify stage compares this against the commit it is deploying, to refuse
# rolling back against a baseline left by an earlier run. A snapshot we cannot
# attribute to a commit is unusable for that, so fail before anything is applied.
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
return 1
fi
}
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
workload_generations() {
kubectl get "$WORKLOAD_KINDS" -A \
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
--no-headers 2>/dev/null \
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
}
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
# moved since the snapshot, i.e. the ones this apply actually touched.
changed_workloads() {
local before="$1"
local ns name kind gen old
while read -r ns name kind gen; do
[ -n "${gen:-}" ] || continue
old="$(awk -v want_ns="$ns" -v want_name="$name" \
'$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)"
if [ "$old" != "$gen" ]; then
printf '%s %s %s\n' "$kind" "$ns" "$name"
fi
done < <(workload_generations)
}
# Prints "<ns> <kind>/<name> <image>" for every workload this repository owns that
# runs an image from our own registry.
#
# The repository is the scope, deliberately. The cluster also holds workloads on
# our registry that no manifest here declares (they are applied out of band), and
# those are somebody else's to deploy. Walking the manifests rather than the
# cluster means those can never be restarted by this pipeline, now or later.
owned_registry_workloads() {
local kd_rel f
while IFS= read -r kd_rel; do
[ -f "$REPO/$kd_rel/active" ] || continue
while IFS= read -r f; do
[ -n "$f" ] || continue
# A file that does not mention the registry cannot declare a workload on it,
# and parsing costs ~2.5s per file against a millisecond for the grep. The
# filter keeps this at a handful of parses instead of one per manifest.
grep -q 'gcr\.forust\.xyz/forust/' "$REPO/$f" 2>/dev/null || continue
# kubectl prints a bare object for a single-document file and a List for a
# multi-document one, so normalise both shapes before filtering.
kubectl apply --dry-run=client -f "$REPO/$f" -o json 2>/dev/null \
| jq -r '
(if .items then .items[] else . end)
| select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
| select(any((.spec.template.spec.containers // [])[]?;
(.image // "") | test("^gcr\\.forust\\.xyz/forust/")))
| (.metadata.namespace // "default") as $ns
| ([.spec.template.spec.containers[].image
| select(test("^gcr\\.forust\\.xyz/forust/"))][0]) as $img
| "\($ns) \(.kind | ascii_downcase)/\(.metadata.name) \($img)"
' 2>/dev/null || true
done < <(collect_k8s "$kd_rel" || true)
done < <(
git -C "$REPO" ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
}
# Prints the digest an image tag resolves to for this cluster's architecture, or
# nothing when it cannot be resolved.
#
# Only the manifest entry matching the node architecture counts. A multi-arch tag
# also carries `unknown/unknown` entries for the build attestation, and a pod's
# imageID is always the per-platform digest, so comparing the wrong entry would
# mark every workload stale forever and restart the whole cluster on every deploy.
registry_digest() {
local arch
arch="$(kubectl get nodes -o jsonpath='{.items[0].status.nodeInfo.architecture}' 2>/dev/null || true)"
[ -n "$arch" ] || arch=amd64
# The || true is load-bearing. Every caller runs under set -euo pipefail, and
# pipefail reports the rightmost non-zero stage, so a ref the registry does not
# have would abort the caller at the assignment instead of yielding an empty
# string. The callers check for empty themselves and report it by name.
#
# Retried with a hard timeout because the registry has a known hang mode (and
# a known blink mode: a single failed lookup aborts the whole apply file in
# render_pinned). A short sleep between attempts lets a restarting registry
# come back instead of failing the deploy on one bad second.
local attempt=0 digest=""
while [ "$attempt" -lt 3 ]; do
digest="$(timeout 25s docker manifest inspect "$1" 2>/dev/null \
| jq -r --arg arch "$arch" '
.manifests[]?
| select(.platform.os == "linux" and .platform.architecture == $arch)
| .digest
' 2>/dev/null \
| head -1 || true)"
[ -n "$digest" ] && break
attempt=$((attempt + 1))
if [ "$attempt" -lt 3 ]; then
echo "WARNING: registry lookup of $1 failed (attempt $attempt/3), retrying in 5s" >&2
sleep 5
fi
done
printf '%s' "$digest"
}
# The commit this deploy is for: what CI validated, or - on a manual dispatch,
# whatever stage_preflight just checked out.
deploy_commit() {
local c="${DEPLOY_SHA:-}"
[ -n "$c" ] || c="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
printf '%.12s' "${c:-}"
}
# Resolves one of our image refs to the digest THIS commit's build produced.
#
# A manifest naming `:prod` names a pointer, not a version, and the deploy
# resolves it when the apply runs - which is not when CI ran it. Deploy runs are
# queued rather than cancelled (see deploy.yaml), so two pushes in a row leave
# the first deploy resolving the second push's build: the right manifests with
# the wrong code, and nothing anywhere reports it. ci therefore publishes every
# image it ships under `sha-<commit12>`, a name that cannot move, and that is
# the name resolved here.
#
# The fallback to the plain tag is for an image this pipeline never built. It
# reports itself, because a fallback nobody sees is the failure this removes.
pinned_digest() {
local ref="$1" commit pinned
commit="$(deploy_commit)"
if [ -n "$commit" ]; then
pinned="$(registry_digest "${ref%:*}:sha-$commit")"
if [ -n "$pinned" ]; then
printf '%s' "$pinned"
return 0
fi
fi
pinned="$(registry_digest "$ref")"
if [ -n "$pinned" ]; then
echo "WARNING: ${ref} carries no sha-${commit:-<unknown>} tag; resolved the moving tag instead" >&2
fi
printf '%s' "$pinned"
}
# Rewrites our own images to immutable digests on the way into the cluster.
# Reads a manifest stream on stdin, writes the pinned stream to stdout.
#
# A digest is not knowable when a manifest is written, so it is never committed:
# git keeps a readable `:prod` tag and the exact bytes are chosen here, at apply
# time, from the tag ci published for the commit being deployed. That is what
# makes rollback mean something. `kubectl rollout undo` restores the previous
# ReplicaSet's pod template verbatim, and a template naming a digest restores the
# exact bytes that were serving before. A template naming a moving tag does not —
# the tag has already moved by the time the rollback runs, so the "rollback"
# re-pulls the very image that just failed and the cluster stays broken.
#
# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a
# reference that is not `:latest` defaults to IfNotPresent, which is what the
# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot
# change, so pulling again buys nothing.
#
# An image that cannot be resolved is fatal. Carrying on would quietly apply a
# mutable tag again, which is the exact failure this function exists to remove.
render_pinned() {
local src refs map ref digest missing=0
src="$(mktemp)"
refs="$(mktemp)"
map="$(mktemp)"
cat >"$src"
grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true
while read -r ref; do
[ -n "$ref" ] || continue
digest="$(pinned_digest "$ref")"
if [ -z "$digest" ]; then
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
echo " The build job has to push that tag before the deploy resolves it." >&2
missing=$((missing + 1))
continue
fi
printf '%s\t%s\n' "$ref" "$digest" >>"$map"
done <"$refs"
if [ "$missing" -gt 0 ]; then
rm -f "$src" "$refs" "$map"
return 1
fi
awk -v mapfile="$map" '
BEGIN {
while ((getline line < mapfile) > 0) {
i = index(line, "\t")
d[substr(line, 1, i - 1)] = substr(line, i + 1)
}
}
{
if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) {
name = $0
sub(/^[[:space:]]*image:[[:space:]]*/, "", name)
sub(/[[:space:]]*$/, "", name)
if (name in d) {
pad = $0
sub(/image:.*/, "", pad)
# Drop the tag: the canonical form used in the docs is repo@sha256:...,
# and leaving :prod next to the digest reads like it still matters.
repo = name
sub(/:[A-Za-z0-9._-]+$/, "", repo)
print pad "image: " repo "@" d[name]
next
}
}
print
}
' "$src"
rm -f "$src" "$refs" "$map"
}
# Restarts every owned workload whose running image is not the one its tag
# resolves to now.
#
# This used to be how a rebuild reached the cluster at all: the manifests pinned
# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply`
# decided there was nothing to do, and the cluster served the previous build
# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves
# the pod template and rolls out on its own.
#
# What is left is the drift check: a hand-run `kubectl set image`, or anything
# else that edits a live workload behind the deploy's back, is the only way to end
# up serving a digest the tag has moved past. It stays idempotent, so a redeploy
# that changed no image still does not bounce healthy services.
#
# The container is matched on its repository rather than on the exact reference:
# once render_pinned has run, a pod's status reports `repo@sha256:...` while this
# still reads the repository's `:prod` tag out of the manifest.
restart_stale_images() {
local ns target image want selector running entry one
local unchecked=0
local -A digests=()
local -a stale=()
while read -r ns target image; do
[ -n "${target:-}" ] || continue
if [ -z "${digests[$image]:-}" ]; then
digests[$image]="$(pinned_digest "$image")"
fi
want="${digests[$image]}"
if [ -z "$want" ]; then
warn "cannot resolve ${image##*/} in the registry, leaving $target alone"
unchecked=$((unchecked + 1))
continue
fi
selector="$(kubectl get "$target" -n "$ns" -o jsonpath='{.spec.selector.matchLabels}' 2>/dev/null \
| jq -r 'to_entries | map("\(.key)=\(.value)") | join(",")' 2>/dev/null)"
if [ -z "$selector" ]; then
warn "cannot read the pod selector of $target, skipping"
unchecked=$((unchecked + 1))
continue
fi
running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \
| jq -r --arg repo "${image%%:*}" '
.items[] | .status.containerStatuses[]?
| select(.image == $repo
or (.image | startswith($repo + ":"))
or (.image | startswith($repo + "@")))
| .imageID
' 2>/dev/null)"
if [ -z "$running" ]; then
# Scaled to zero. Nothing is serving stale code, and imagePullPolicy
# resolves the tag when it is scaled back up.
continue
fi
entry=""
while IFS= read -r one; do
[ -n "$one" ] || continue
entry="${one##*@}"
if [ "$entry" != "$want" ]; then
stale+=("$ns $target")
break
fi
done <<<"$running"
done < <(owned_registry_workloads)
if [ "${#stale[@]}" -eq 0 ]; then
if [ "$unchecked" -gt 0 ]; then
# Say so plainly. Reporting "everything is current" after checking nothing
# would tell the operator the deploy is fine when it may not be.
warn "No workload needed a restart, but $unchecked could not be checked"
else
log "All owned workloads already run the image their tag points at"
fi
return 0
fi
log "Restarting ${#stale[@]} workload(s) running an image their tag has moved past"
for ref in "${stale[@]}"; do
log " $ref"
done
local failed=()
for ref in "${stale[@]}"; do
ns="${ref%% *}"
target="${ref#* }"
if ! kubectl rollout restart "$target" -n "$ns" >/dev/null 2>&1; then
failed+=("$ref")
fi
done
if [ "${#failed[@]}" -gt 0 ]; then
warn "could not restart: ${failed[*]}"
return 1
fi
}
# verify_workloads <failed-file> <kind> <ns> <name> ...
# Watches every workload in parallel and records the ones that never became
# healthy. Returns non-zero if any of them failed.
verify_workloads() {
local failed_file="$1"
shift
[ "$#" -gt 0 ] || return 0
: >"$failed_file"
local running=0 pid kind ns name
local -a pids=()
for entry in "$@"; do
read -r kind ns name <<<"$entry"
(
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " ok: ${kind}/${ns}/${name}"
else
echo " FAILED: ${kind}/${ns}/${name}"
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
fi
) &
pids+=($!)
running=$((running + 1))
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
wait -n 2>/dev/null || true
running=$((running - 1))
fi
done
for pid in ${pids[@]+"${pids[@]}"}; do
wait "$pid" || true
done
# Non-zero when the file holds at least one failure, i.e. a workload never
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
# every rollback this stage is meant to perform.
[ ! -s "$failed_file" ]
}
# rollback_workloads <failed-file>
# Restores the previous revision of every failed workload and waits for it to
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
# so the operator knows manual recovery is required.
rollback_workloads() {
local failed_file="$1"
local kind ns name unrecovered=()
local -a recovered=()
while read -r kind ns name; do
[ -n "${kind:-}" ] || continue
# Helm-owned workloads are already rolled back by the release's --rollback-on-failure
# upgrade. `rollout undo` here would step back to the revision Helm just
# escaped (the failed one), so leave them for the operator instead.
if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then
echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}"
unrecovered+=("${kind}/${ns}/${name} (helm-managed)")
continue
fi
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " rolled back: ${kind}/${ns}/${name}"
recovered+=("${kind}/${ns}/${name}")
else
echo " NOT RECOVERED: ${kind}/${ns}/${name}"
unrecovered+=("${kind}/${ns}/${name}")
fi
done <"$failed_file"
echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file"
echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file"
[ "${#unrecovered[@]}" -eq 0 ]
}
# Helm releases owned by this stage, one line each:
#
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
#
# The chart version is the field Renovate keeps current. The helmv3 manager only
# understands Chart.yaml and the helm-values manager only values files, so a pin
# written straight into a `helm upgrade` command would never be updated: these
# have to be declared as custom.regex managers in renovate/renovate.json.
HELM_RELEASES=(
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
"reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active"
)
# "name url" for the Helm repository hosting a chart, empty if unknown.
helm_repo_for() {
case "$1" in
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;;
esac
}
# helm_release_status <release> <namespace>
# Prints the release status in lowercase (deployed, failed, pending-rollback,
# ...) or "not-found" when the release does not exist yet.
helm_release_status() {
local out
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
echo "not-found"
return 0
fi
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
}
# recover_pending_release <release> <namespace>
# Rolls a release out of a pending-* state left by a failed upgrade with --rollback-on-failure
# whose own rollback never completed. Without this every future upgrade errors
# out until a human runs `helm rollback`. Passes through releases that are not
# pending (deployed, failed, not-found). Returns non-zero when the release is
# still not recoverable, so the pipeline fails loud instead of wedging.
recover_pending_release() {
local release="$1" namespace="$2" status
status="$(helm_release_status "$release" "$namespace")"
case "$status" in
pending-upgrade|pending-rollback|pending-install)
log "Release $release is $status, rolling back to the last deployed revision"
if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then
echo "WARN: helm rollback of $release did not complete"
return 1
fi
status="$(helm_release_status "$release" "$namespace")"
if [ "$status" != "deployed" ]; then
echo "WARN: $release is $status after rollback"
return 1
fi
;;
esac
return 0
}
# wait_for_calm <stage>
# The deploy itself is heavy enough to melt this single node (helm churn plus
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
# the 1-minute load average to drop below the ceiling, then proceed anyway
# with a warning so a permanently busy node cannot wedge the pipeline forever.
wait_for_calm() {
local load waited=0
while [ "$waited" -lt 600 ]; do
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
if [ "$load" -lt 28 ]; then
return 0
fi
if [ "$((waited % 60))" -eq 0 ]; then
log "$1: load $load, waiting for calm (<28)..."
fi
sleep 15
waited=$((waited + 15))
done
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
}
upgrade_helm_releases() {
local entry release chart namespace version values marker repo
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
IFS='|' read -r release chart namespace version values marker <<<"$entry"
if [ ! -f "$REPO/$marker" ]; then
echo "skip (no $marker): $release"
continue
fi
if [ ! -f "$REPO/$values" ]; then
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
return 1
fi
repo="$(helm_repo_for "$chart")"
if [ -z "$repo" ]; then
echo "ERROR: no Helm repository configured for chart $chart"
return 1
fi
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
log "Upgrading $release ($chart $version)"
wait_for_calm "helm $release"
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
# release in pending-*, which blocks every future upgrade. Recover first
# so one wedged revision cannot wedge the pipeline forever.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
return 1
fi
# --rollback-on-failure (+ --wait) rolls the release back when the upgrade
# times out or the workloads it touches never become ready, so a bad chart
# bump is not left half applied. (--atomic was this combo; deprecated.)
if ! helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \
--version "$version" \
--values "$REPO/$values" \
--wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then
echo "WARN: upgrade of $release failed, checking release state"
# --rollback-on-failure already attempted its own rollback; finish the job when that
# rollback never completed, otherwise the release stays pending-* and
# blocks every future run.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
else
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
fi
return 1
fi
done
}
stage_preflight() {
if [ ! -d "$REPO/.git" ]; then
echo "Repository not found at $REPO"
exit 1
fi
if [ -n "$DEPLOY_SHA" ]; then
log "Checking out the commit CI validated ($DEPLOY_SHA)"
git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \
|| git -C "$REPO" fetch origin main
else
git -C "$REPO" fetch origin main
fi
target="${DEPLOY_SHA:-origin/main}"
log "Workstation state"
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
echo " target: $(git -C "$REPO" rev-parse --short "$target")"
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
echo "ERROR: workstation has local tracked modifications, refusing reset:"
git -C "$REPO" status --porcelain --untracked-files=no
git -C "$REPO" diff --stat
echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy."
exit 1
fi
git -C "$REPO" reset --hard "$target"
}
stage_validate() {
cd "$REPO"
select_manifests
local m k cf
# Compose .env files and secret files are gitignored by design, so the
# workstation never has real values for the inactive stacks. This stage only
# runs the full check on active stacks; the general structure check for every
# committed Compose file (active or not) lives in the ci workflow, which has no
# .env at all.
#
# Active stacks are still validated with interpolation and env-file resolution
# off, so required-variable guards (:?) and missing local files do not fail the
# deploy. Normalization and consistency checks stay enabled.
# shellcheck source=compose-lint.sh
source "$REPO/.gitea/workflows/compose-lint.sh"
local compose_validate_flags=()
mapfile -t compose_validate_flags < <(compose_safe_flags)
log "Validate compose stacks"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " config: $cf"
validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"}
done
log "Validate k8s manifests (kubectl dry-run=client)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
kubectl apply --dry-run=client -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=client >/dev/null
done
log "Validate k8s manifests (kubectl dry-run=server)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
kubectl apply --dry-run=server -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=server >/dev/null
done
log "Checking referenced Secrets exist"
echo " (deploy never applies *secret*.yaml; create missing ones manually)"
local ref_secrets=() missing_secrets=() all_secrets s
if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then
while IFS= read -r s; do
[ -n "$s" ] && ref_secrets+=("$s")
done < <(
{
grep -h -A1 -E 'secretRef:|secretKeyRef:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
grep -h -E 'secretName:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
} | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true
)
fi
all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)"
for s in ${ref_secrets[@]+"${ref_secrets[@]}"}; do
if printf '%s\n' "$all_secrets" | grep -qx "$s"; then
echo " ok: $s"
else
echo " MISSING: $s"
missing_secrets+=("$s")
fi
done
if [ "${#missing_secrets[@]}" -gt 0 ]; then
echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:"
printf ' - %s\n' "${missing_secrets[@]}"
echo "Create them manually from the laptop, e.g.:"
echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example"
exit 1
fi
}
stage_apply_k8s() {
cd "$REPO"
select_manifests >/dev/null
local ns_files=() other_files=() m k prune_opts=()
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
case "$m" in
*/namespace.y?ml) ns_files+=("$m") ;;
*) other_files+=("$m") ;;
esac
done
if [ "$APPLY_PRUNE" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
# Record what is about to change, and publish it for the verify job, before
# the first apply. Both are fatal on failure: see snapshot_dir.
local snapshot
snapshot="$(snapshot_dir)" || return 1
save_snapshot "$snapshot" || return 1
if [ "${#ns_files[@]}" -gt 0 ]; then
log "Applying namespaces (${#ns_files[@]} files)"
for m in "${ns_files[@]}"; do
kubectl apply -f "$m"
done
fi
if [ -f "$REPO/prometheus-stack/k8s/active" ]; then
if [ ! -f "$REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
exit 1
fi
fi
upgrade_helm_releases
wait_for_calm "apply resources"
if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
for m in "${other_files[@]}"; do
if ! render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -; then
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
exit 1
fi
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then
echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2
exit 1
fi
done
restart_stale_images
# No verification here on purpose. This stage may be killed at any point by
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
# connection, and any code below that line would simply not run. stage_verify_k8s
# picks the work up from the snapshot instead.
log "Applied. Verification and rollback are the verify job's job, not this one's."
}
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
# including when they failed, timed out or were cancelled. Reads the baseline the
# apply stage published and works out what it changed, watches those workloads,
# and rolls back the ones that never became healthy.
stage_verify_k8s() {
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
local snapshot want have generations
local -a touched=()
if [ ! -s "$pointer" ]; then
echo "ERROR: no snapshot pointer at $pointer."
echo "The apply stage died before publishing any state, so there is no baseline to"
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
echo "inspect the cluster by hand."
return 1
fi
snapshot="$(head -1 "$pointer")"
if [ ! -d "$snapshot" ]; then
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
return 1
fi
# Never trust the pointer blindly. If the apply stage was killed before it
# published its own snapshot, `current` still points at the previous deploy's
# baseline. Verifying against that would watch the wrong workloads and the
# rollback would revert the wrong revisions, so refuse instead.
want="${DEPLOY_SHA:-}"
if [ -z "$want" ]; then
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
fi
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
if [ -z "$want" ] || [ "$have" != "$want" ]; then
echo "ERROR: refusing to verify or roll back against a stale snapshot."
echo " snapshot: $snapshot"
echo " snapshot commit: ${have:-<missing>}"
echo " deploy commit: ${want:-<unknown>}"
return 1
fi
echo " snapshot: $snapshot (commit ${have:0:12})"
generations="$snapshot/generations.before"
if [ ! -s "$generations" ]; then
# Without a baseline we cannot tell which workloads the apply touched, so
# fall back to watching everything rather than silently skipping the check.
warn "no pre-apply baseline, verifying every workload in the cluster"
: >"$generations"
fi
while read -r kind ns name; do
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
done < <(changed_workloads "$generations")
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
if [ "${#touched[@]}" -eq 0 ]; then
echo " nothing to verify"
return 0
fi
printf ' watching: %s\n' "${touched[@]/#/ }"
local failed_file="$snapshot/failed-workloads"
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
echo
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
echo
log "Rolling back to the previous revision"
if rollback_workloads "$failed_file"; then
echo
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
echo "externally created resources from this commit are still in place. Review the failed"
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
else
echo
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
echo "Pre-apply snapshot: $snapshot"
fi
return 1
fi
}
# verify_compose_stack <compose-file>
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
# come back broken with a green pipeline. Require every long-running service to
# actually be running.
verify_compose_stack() {
local cf="$1"
local expected running missing=()
expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)"
running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)"
[ -n "$expected" ] || return 0
while IFS= read -r svc; do
[ -n "$svc" ] || continue
# restart:"no" services are allowed to have exited.
if ! printf '%s\n' "$running" | grep -qx "$svc" \
&& ! docker compose -f "$cf" config 2>/dev/null \
| grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then
missing+=("$svc")
fi
done <<<"$expected"
if [ "${#missing[@]}" -gt 0 ]; then
echo " NOT RUNNING: ${missing[*]}"
docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
return 1
fi
echo " all ${#expected} service(s) running"
return 0
}
# The public hostname of every active service, one per line.
#
# Comments are stripped first, and deliberately so: a route that someone
# disabled by commenting it out is not a service to probe, and naio and xui are
# both still in the tree that way. A `#` only starts a comment when it is at the
# start of a line or after whitespace, so `s/#.*//` alone would also cut a
# legitimate value in half.
#
# Only the public names. The *.internal names are the same Traefik and the same
# Services, reached by a different label, so probing both would double the run
# to learn the same thing. The public name is also the one a user types.
smoke_hosts() {
local m k
# The backticks below are literal. They are Traefik's Host() delimiter, and the
# single quotes are precisely what keeps the shell from reading them as a
# command substitution, so the warning is the opposite of a real problem.
# shellcheck disable=SC2016
{
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
[ -f "$m" ] && cat "$m"
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl kustomize "$k" 2>/dev/null || true
done
} | sed -E 's/(^|[[:space:]])#.*$//' \
| grep -oE 'Host\(`[^`]+`\)' \
| sed -E 's/^Host\(`//; s/`\)$//' \
| grep -E '(^|\.)forust\.xyz$' \
| grep -v '\${' \
| sort -u
}
# Traefik's own list of the routes it actually built. The Kubernetes CRs are the
# wrong source for this: when a middleware fails to load, Traefik drops the
# router that referenced it and leaves the CR behind looking perfectly healthy.
#
# api.insecure is already on for the internal `traefik` entrypoint, but the pod
# IP is not routable from the node, so read it through kubectl exec rather than
# standing up a port-forward. HTTP only: the TCP routers match on HostSNI(`*`)
# and the UDP ones carry no rule at all, both selected by entrypoint and port,
# so neither can answer whether a given host has a route.
traefik_http_routes() {
kubectl -n traefik exec deploy/traefik -- \
wget -qO- --timeout=10 http://127.0.0.1:8080/api/http/routers 2>/dev/null \
| jq -c '[.[] | {status, rule: (.rule // "")}]'
}
# The hosts Traefik currently routes to, one per line. Every backticked token of
# an enabled rule counts, which is a superset of the hosts -- PathPrefix values
# land here too, harmlessly -- but it keeps the host syntax in one place instead
# of a matcher per host. The scan keeps the delimiters, so strip them: what
# belongs in a comparison against a hostname is the bare name.
traefik_routed_hosts() {
jq -r '[.[] | select(.status == "enabled") | (.rule // "")
| scan("`[^`]+`") | ltrimstr("`") | rtrimstr("`")]
| unique | .[]' <<<"$1"
}
# stage_verify_k8s watches the rollout, which reports that the pods converged.
# It cannot tell a converged pod from a serving one: a route pointing at the
# wrong port, a Service selector that matches nothing the app listens on, a 500
# from the app itself, an OOMKill loop that still counts as Available for long
# enough to pass. All of those are green at the rollout level.
#
# So ask the thing users ask. Any HTTP response proves Traefik matched the
# host, the Service resolved to a pod and the pod answered -- a 302 to a login
# or a 404 from a path the service does not serve still means the chain is
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
# the service is not serving, and only those fail the run.
#
# Except that a 404 is not evidence on its own. A router Traefik refused to
# build answers with the same 404 and nothing behind it, so a middleware that
# fails to load takes down every route that referenced it while
# this stage reports `ok` for all of them. No status code separates those two
# cases, so ask Traefik which routes it built and fail on the difference.
stage_smoke() {
cd "$REPO"
select_manifests >/dev/null
local -a hosts=()
# Not named failed: an array of that name already exists in restart_stale_images
# above, and a scalar shadowing an array is a trap rather than a shadow.
local h code rc bad=0
while IFS= read -r h; do
[ -n "$h" ] && hosts+=("$h")
done < <(smoke_hosts)
if [ "${#hosts[@]}" -eq 0 ]; then
# Nothing to probe means the extraction broke, not that the cluster is empty.
echo "ERROR: no public hostnames found in active manifests, refusing to report success"
return 1
fi
log "Probing ${#hosts[@]} public route(s)"
for h in "${hosts[@]}"; do
code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$?
if [ "$rc" -ne 0 ]; then
echo " UNREACHABLE $h (curl exit $rc)"
bad=1
continue
fi
# A glob, not a string compare. `case` on the leading digit is the only one
# of these that survives a three-digit code, and the obvious expansion to
# try first -- ${code%%[0-9]*} -- is empty for every input, so it silently
# reports a 500 as healthy.
case "$code" in
5*)
echo " SERVER ERROR $h $code"
bad=1
;;
000)
# curl exited 0 and still no status, so nothing on the far end replied.
# Not a pass, whatever the transport thought.
echo " NO RESPONSE $h"
bad=1
;;
*)
echo " ok $h $code"
;;
esac
done
# Second gate. The probe above only means something if a router matched the
# host in the first place, so compare the hosts we expect against the hosts
# Traefik reports and fail on the difference.
local routes routed
if ! routes="$(traefik_http_routes)"; then
echo "ERROR: could not read Traefik's router list, refusing to report success"
return 1
fi
routed="$(traefik_routed_hosts "$routes")"
local -a unrouted=()
local tries=3
while :; do
unrouted=()
for h in "${hosts[@]}"; do
grep -qxF "$h" <<<"$routed" || unrouted+=("$h")
done
if [ "${#unrouted[@]}" -eq 0 ]; then
break
fi
# A router mid-rollout is legitimately absent for a moment. A middleware
# that failed to load stays absent, so waiting cannot paper over it.
if [ "$tries" -le 1 ]; then
break
fi
tries=$((tries - 1))
warn "${#unrouted[@]} host(s) have no enabled route yet, re-checking in 10s"
sleep 10
if ! routes="$(traefik_http_routes)"; then
break
fi
routed="$(traefik_routed_hosts "$routes")"
done
if [ "${#unrouted[@]}" -ne 0 ]; then
for h in "${unrouted[@]}"; do
echo " NO ROUTE $h (Traefik has no enabled router for this host)"
done
bad=1
fi
if [ "$bad" -ne 0 ]; then
echo "ERROR: at least one active service is not serving over its public route"
return 1
fi
echo "all ${#hosts[@]} route(s) answered and have a router"
}
stage_apply_compose() {
cd "$REPO"
select_manifests >/dev/null
local cf
log "Redeploying docker compose stacks (${#COMPOSE_STACKS[@]} stacks)"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " compose: $cf"
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
docker compose -f "$cf" build
docker compose -f "$cf" push
fi
docker compose -f "$cf" up -d --pull always --remove-orphans
done
local -a broken=()
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " verifying: $cf"
if ! verify_compose_stack "$cf"; then
broken+=("$cf")
fi
done
if [ "${#broken[@]}" -gt 0 ]; then
echo
echo "ERROR: ${#broken[@]} compose stack(s) did not come up:"
printf ' - %s\n' "${broken[@]}"
echo "Compose stacks are not rolled back automatically: their images use mutable"
echo "':latest' tags, so there is no previous version to return to. Check the logs"
echo "above, then re-run the deploy once the cause is fixed."
return 1
fi
}
run_stage() {
case "${1:?stage required}" in
preflight) stage_preflight ;;
validate) stage_validate ;;
apply-k8s) stage_apply_k8s ;;
verify-k8s) stage_verify_k8s ;;
smoke) stage_smoke ;;
apply-compose) stage_apply_compose ;;
*)
echo "ERROR: unknown stage: $1"
exit 1
;;
esac
}