feat(deploy): serialize apply stages and wait for calm node

apply-k8s and apply-compose share a workstation flock so host docker churn never overlaps cluster churn. Each helm upgrade and the apply loop wait up to 10m for load <28 first, so a deploy never piles onto an already-hot node (the load-40/netbird-death/pending-helm spiral).
This commit is contained in:
forust committed 2026-09-29 14:42:56 +02:00
1 parent e56662194b
commit 6f4cd03f4b
2 files changed
+36 -1

No files matched your search

+12 -1
View File
@@ -36,6 +36,17 @@ ssh_opts=(
)
rc=0
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
# intact for the verify job, but on a single node they must not run at once:
# host docker churn on top of cluster churn is what melts the node (load 40+,
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
# with a shared lock; whoever arrives second waits.
remote_cmd=(bash -se)
case "$1" in
apply-k8s | apply-compose)
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
;;
esac
for attempt in 1 2 3; do
if [ "$attempt" -gt 1 ]; then
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
@@ -45,7 +56,7 @@ for attempt in 1 2 3; do
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
"STAGE=$1" bash -se <<'EOF' || rc=$?
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
source "$REPO/.gitea/workflows/deploy-lib.sh"
run_stage "$STAGE"
EOF