Files
homelab/.gitea/workflows/ssh-run.sh
T
forust 872f64b887
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 9s
ci / lint-prettier (push) Successful in 16s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 8s
renovate-ci / validate-renovate (push) Successful in 22s
ci / build (push) Successful in 38s
fix(ci): silence intentional SC2029 in ssh-run.sh
2026-09-29 14:54:17 +02:00

72 lines
2.8 KiB
Bash
Executable File

#!/usr/bin/env bash
# usage: ssh-run.sh <stage>
# Runs one deploy-lib.sh stage on the workstation over SSH.
set -euo pipefail
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
: "${DEPLOY_USER:?missing DEPLOY_USER}"
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
deploy_port="${DEPLOY_PORT:-22}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
# The private key is written to a per-run directory that is removed on exit, so a
# failed or cancelled job cannot leave deploy credentials in the runner's temp
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
trap 'rm -rf "$key_dir"' EXIT INT TERM
ssh_key="$key_dir/deploy_key"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key"
# A connection that died silently used to hang until the job timeout, and the
# stage was never re-run: one flaky TCP session cost a whole 45-minute apply.
# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds
# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that
# fails on its own merits exits with the remote's status, so a real failure
# still surfaces its own log instead of burning three attempts. The stages are
# declarative applies, so re-running one that had already committed is harmless.
ssh_opts=(
-i "$ssh_key" -p "$deploy_port"
-o BatchMode=yes -o StrictHostKeyChecking=accept-new
-o ConnectTimeout=15
-o ServerAliveInterval=15 -o ServerAliveCountMax=4
)
rc=0
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
# intact for the verify job, but on a single node they must not run at once:
# host docker churn on top of cluster churn is what melts the node (load 40+,
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
# with a shared lock; whoever arrives second waits.
remote_cmd=(bash -se)
case "$1" in
apply-k8s | apply-compose)
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
;;
esac
for attempt in 1 2 3; do
if [ "$attempt" -gt 1 ]; then
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
sleep $((attempt * 5))
fi
rc=0
# shellcheck disable=SC2029 # remote_cmd/ssh_opts expand on the client on purpose: they select the local ssh invocation, only the heredoc runs remotely.
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
source "$REPO/.gitea/workflows/deploy-lib.sh"
run_stage "$STAGE"
EOF
[ "$rc" -eq 0 ] && break
[ "$rc" -ne 255 ] && break
done
if [ "$rc" -ne 0 ]; then
echo ":: error::stage $1 failed over ssh (exit $rc)"
fi
exit "$rc"