feat(deploy): serialize apply stages and wait for calm node
apply-k8s and apply-compose share a workstation flock so host docker churn never overlaps cluster churn. Each helm upgrade and the apply loop wait up to 10m for load <28 first, so a deploy never piles onto an already-hot node (the load-40/netbird-death/pending-helm spiral).
This commit is contained in:
1 parent
e56662194b
commit
6f4cd03f4b
2 files changed
+36
-1
No files matched your search
@@ -588,6 +588,28 @@ recover_pending_release() {
|
|||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# wait_for_calm <stage>
|
||||||
|
# The deploy itself is heavy enough to melt this single node (helm churn plus
|
||||||
|
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
|
||||||
|
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
|
||||||
|
# the 1-minute load average to drop below the ceiling, then proceed anyway
|
||||||
|
# with a warning so a permanently busy node cannot wedge the pipeline forever.
|
||||||
|
wait_for_calm() {
|
||||||
|
local load waited=0
|
||||||
|
while [ "$waited" -lt 600 ]; do
|
||||||
|
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
|
||||||
|
if [ "$load" -lt 28 ]; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
if [ "$((waited % 60))" -eq 0 ]; then
|
||||||
|
log "$1: load $load, waiting for calm (<28)..."
|
||||||
|
fi
|
||||||
|
sleep 15
|
||||||
|
waited=$((waited + 15))
|
||||||
|
done
|
||||||
|
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
|
||||||
|
}
|
||||||
|
|
||||||
upgrade_helm_releases() {
|
upgrade_helm_releases() {
|
||||||
local entry release chart namespace version values marker repo
|
local entry release chart namespace version values marker repo
|
||||||
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||||
@@ -608,6 +630,7 @@ upgrade_helm_releases() {
|
|||||||
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||||
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||||
log "Upgrading $release ($chart $version)"
|
log "Upgrading $release ($chart $version)"
|
||||||
|
wait_for_calm "helm $release"
|
||||||
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
|
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
|
||||||
# release in pending-*, which blocks every future upgrade. Recover first
|
# release in pending-*, which blocks every future upgrade. Recover first
|
||||||
# so one wedged revision cannot wedge the pipeline forever.
|
# so one wedged revision cannot wedge the pipeline forever.
|
||||||
@@ -763,6 +786,7 @@ stage_apply_k8s() {
|
|||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
upgrade_helm_releases
|
upgrade_helm_releases
|
||||||
|
wait_for_calm "apply resources"
|
||||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||||
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
|
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
|
||||||
for m in "${other_files[@]}"; do
|
for m in "${other_files[@]}"; do
|
||||||
|
|||||||
@@ -36,6 +36,17 @@ ssh_opts=(
|
|||||||
)
|
)
|
||||||
|
|
||||||
rc=0
|
rc=0
|
||||||
|
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
|
||||||
|
# intact for the verify job, but on a single node they must not run at once:
|
||||||
|
# host docker churn on top of cluster churn is what melts the node (load 40+,
|
||||||
|
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
|
||||||
|
# with a shared lock; whoever arrives second waits.
|
||||||
|
remote_cmd=(bash -se)
|
||||||
|
case "$1" in
|
||||||
|
apply-k8s | apply-compose)
|
||||||
|
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
|
||||||
|
;;
|
||||||
|
esac
|
||||||
for attempt in 1 2 3; do
|
for attempt in 1 2 3; do
|
||||||
if [ "$attempt" -gt 1 ]; then
|
if [ "$attempt" -gt 1 ]; then
|
||||||
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
|
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
|
||||||
@@ -45,7 +56,7 @@ for attempt in 1 2 3; do
|
|||||||
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||||
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
|
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
|
||||||
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
|
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
|
||||||
"STAGE=$1" bash -se <<'EOF' || rc=$?
|
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
|
||||||
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||||
run_stage "$STAGE"
|
run_stage "$STAGE"
|
||||||
EOF
|
EOF
|
||||||
|
|||||||
Reference in new issue
Block a user