fix(deploy): recover helm releases from pending-* and skip helm-owned rollbacks

An --atomic upgrade whose own rollback never finishes leaves the release in pending-*, blocking every future run until a human rolls back (loki rev 18/21). Recover automatically before and after each upgrade, and fail loud when recovery does not land on deployed. Also skip helm-managed workloads in rollback_workloads: rollout undo there would step back to the revision --atomic just escaped.
This commit is contained in:
forust committed 2026-09-28 12:03:26 +02:00
1 parent 7c4843c88c
commit dde6b1c743
1 file changed
+66 -2
+66 -2
View File
@@ -489,6 +489,14 @@ rollback_workloads() {
local -a recovered=() local -a recovered=()
while read -r kind ns name; do while read -r kind ns name; do
[ -n "${kind:-}" ] || continue [ -n "${kind:-}" ] || continue
# Helm-owned workloads are already rolled back by the release's --atomic
# upgrade. `rollout undo` here would step back to the revision Helm just
# escaped (the failed one), so leave them for the operator instead.
if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then
echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}"
unrecovered+=("${kind}/${ns}/${name} (helm-managed)")
continue
fi
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \ if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then && kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " rolled back: ${kind}/${ns}/${name}" echo " rolled back: ${kind}/${ns}/${name}"
@@ -527,6 +535,44 @@ helm_repo_for() {
esac esac
} }
# helm_release_status <release> <namespace>
# Prints the release status in lowercase (deployed, failed, pending-rollback,
# ...) or "not-found" when the release does not exist yet.
helm_release_status() {
local out
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
echo "not-found"
return 0
fi
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
}
# recover_pending_release <release> <namespace>
# Rolls a release out of a pending-* state left by a failed --atomic upgrade
# whose own rollback never completed. Without this every future upgrade errors
# out until a human runs `helm rollback`. Passes through releases that are not
# pending (deployed, failed, not-found). Returns non-zero when the release is
# still not recoverable, so the pipeline fails loud instead of wedging.
recover_pending_release() {
local release="$1" namespace="$2" status
status="$(helm_release_status "$release" "$namespace")"
case "$status" in
pending-upgrade|pending-rollback|pending-install)
log "Release $release is $status, rolling back to the last deployed revision"
if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then
echo "WARN: helm rollback of $release did not complete"
return 1
fi
status="$(helm_release_status "$release" "$namespace")"
if [ "$status" != "deployed" ]; then
echo "WARN: $release is $status after rollback"
return 1
fi
;;
esac
return 0
}
upgrade_helm_releases() { upgrade_helm_releases() {
local entry release chart namespace version values marker repo local entry release chart namespace version values marker repo
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
@@ -547,13 +593,31 @@ upgrade_helm_releases() {
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
helm repo update "${repo%% *}" >/dev/null 2>&1 || true helm repo update "${repo%% *}" >/dev/null 2>&1 || true
log "Upgrading $release ($chart $version)" log "Upgrading $release ($chart $version)"
# A previous --atomic run whose own rollback never finished leaves the
# release in pending-*, which blocks every future upgrade. Recover first
# so one wedged revision cannot wedge the pipeline forever.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
return 1
fi
# --atomic rolls the release back when the upgrade times out or the workloads # --atomic rolls the release back when the upgrade times out or the workloads
# it touches never become ready, so a bad chart bump is not left half applied. # it touches never become ready, so a bad chart bump is not left half applied.
helm upgrade --install "$release" "$chart" \ if ! helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \ --namespace "$namespace" \
--version "$version" \ --version "$version" \
--values "$REPO/$values" \ --values "$REPO/$values" \
--atomic --cleanup-on-fail --timeout 10m --atomic --cleanup-on-fail --timeout 10m; then
echo "WARN: upgrade of $release failed, checking release state"
# --atomic already attempted its own rollback; finish the job when that
# rollback never completed, otherwise the release stays pending-* and
# blocks every future run.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
else
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
fi
return 1
fi
done done
} }