feat(deploy): serialize apply stages and wait for calm node
apply-k8s and apply-compose share a workstation flock so host docker churn never overlaps cluster churn. Each helm upgrade and the apply loop wait up to 10m for load <28 first, so a deploy never piles onto an already-hot node (the load-40/netbird-death/pending-helm spiral).
This commit is contained in:
1 parent
e56662194b
commit
6f4cd03f4b
2 files changed
+36
-1
No files matched your search
@@ -588,6 +588,28 @@ recover_pending_release() {
|
||||
return 0
|
||||
}
|
||||
|
||||
# wait_for_calm <stage>
|
||||
# The deploy itself is heavy enough to melt this single node (helm churn plus
|
||||
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
|
||||
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
|
||||
# the 1-minute load average to drop below the ceiling, then proceed anyway
|
||||
# with a warning so a permanently busy node cannot wedge the pipeline forever.
|
||||
wait_for_calm() {
|
||||
local load waited=0
|
||||
while [ "$waited" -lt 600 ]; do
|
||||
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
|
||||
if [ "$load" -lt 28 ]; then
|
||||
return 0
|
||||
fi
|
||||
if [ "$((waited % 60))" -eq 0 ]; then
|
||||
log "$1: load $load, waiting for calm (<28)..."
|
||||
fi
|
||||
sleep 15
|
||||
waited=$((waited + 15))
|
||||
done
|
||||
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
|
||||
}
|
||||
|
||||
upgrade_helm_releases() {
|
||||
local entry release chart namespace version values marker repo
|
||||
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||
@@ -608,6 +630,7 @@ upgrade_helm_releases() {
|
||||
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||
log "Upgrading $release ($chart $version)"
|
||||
wait_for_calm "helm $release"
|
||||
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
|
||||
# release in pending-*, which blocks every future upgrade. Recover first
|
||||
# so one wedged revision cannot wedge the pipeline forever.
|
||||
@@ -763,6 +786,7 @@ stage_apply_k8s() {
|
||||
fi
|
||||
fi
|
||||
upgrade_helm_releases
|
||||
wait_for_calm "apply resources"
|
||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
|
||||
for m in "${other_files[@]}"; do
|
||||
|
||||
Reference in new issue
Block a user