ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 9s
ci / lint-prettier (push) Successful in 16s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 8s
renovate-ci / validate-renovate (push) Successful in 22s
ci / build (push) Successful in 38s
72 lines
2.8 KiB
Bash
Executable File
72 lines
2.8 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# usage: ssh-run.sh <stage>
|
|
# Runs one deploy-lib.sh stage on the workstation over SSH.
|
|
set -euo pipefail
|
|
|
|
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
|
|
: "${DEPLOY_USER:?missing DEPLOY_USER}"
|
|
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
|
|
|
|
deploy_port="${DEPLOY_PORT:-22}"
|
|
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
|
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
|
|
|
|
# The private key is written to a per-run directory that is removed on exit, so a
|
|
# failed or cancelled job cannot leave deploy credentials in the runner's temp
|
|
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
|
|
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
|
|
trap 'rm -rf "$key_dir"' EXIT INT TERM
|
|
|
|
ssh_key="$key_dir/deploy_key"
|
|
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
|
chmod 600 "$ssh_key"
|
|
|
|
# A connection that died silently used to hang until the job timeout, and the
|
|
# stage was never re-run: one flaky TCP session cost a whole 45-minute apply.
|
|
# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds
|
|
# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that
|
|
# fails on its own merits exits with the remote's status, so a real failure
|
|
# still surfaces its own log instead of burning three attempts. The stages are
|
|
# declarative applies, so re-running one that had already committed is harmless.
|
|
ssh_opts=(
|
|
-i "$ssh_key" -p "$deploy_port"
|
|
-o BatchMode=yes -o StrictHostKeyChecking=accept-new
|
|
-o ConnectTimeout=15
|
|
-o ServerAliveInterval=15 -o ServerAliveCountMax=4
|
|
)
|
|
|
|
rc=0
|
|
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
|
|
# intact for the verify job, but on a single node they must not run at once:
|
|
# host docker churn on top of cluster churn is what melts the node (load 40+,
|
|
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
|
|
# with a shared lock; whoever arrives second waits.
|
|
remote_cmd=(bash -se)
|
|
case "$1" in
|
|
apply-k8s | apply-compose)
|
|
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
|
|
;;
|
|
esac
|
|
for attempt in 1 2 3; do
|
|
if [ "$attempt" -gt 1 ]; then
|
|
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
|
|
sleep $((attempt * 5))
|
|
fi
|
|
rc=0
|
|
# shellcheck disable=SC2029 # remote_cmd/ssh_opts expand on the client on purpose: they select the local ssh invocation, only the heredoc runs remotely.
|
|
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
|
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
|
|
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
|
|
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
|
|
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
|
run_stage "$STAGE"
|
|
EOF
|
|
[ "$rc" -eq 0 ] && break
|
|
[ "$rc" -ne 255 ] && break
|
|
done
|
|
|
|
if [ "$rc" -ne 0 ]; then
|
|
echo ":: error::stage $1 failed over ssh (exit $rc)"
|
|
fi
|
|
exit "$rc"
|