#!/usr/bin/env bash # usage: ssh-run.sh # Runs one deploy-lib.sh stage on the workstation over SSH. set -euo pipefail : "${DEPLOY_HOST:?missing DEPLOY_HOST}" : "${DEPLOY_USER:?missing DEPLOY_USER}" : "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}" deploy_port="${DEPLOY_PORT:-22}" deploy_path="${DEPLOY_PATH:-/srv/homelab}" deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)" # The private key is written to a per-run directory that is removed on exit, so a # failed or cancelled job cannot leave deploy credentials in the runner's temp # directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel. key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")" trap 'rm -rf "$key_dir"' EXIT INT TERM ssh_key="$key_dir/deploy_key" printf '%s\n' "$DEPLOY_KEY" > "$ssh_key" chmod 600 "$ssh_key" # A connection that died silently used to hang until the job timeout, and the # stage was never re-run: one flaky TCP session cost a whole 45-minute apply. # ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds # setup. Only exit 255 - ssh's own transport failures - is retried. A stage that # fails on its own merits exits with the remote's status, so a real failure # still surfaces its own log instead of burning three attempts. The stages are # declarative applies, so re-running one that had already committed is harmless. ssh_opts=( -i "$ssh_key" -p "$deploy_port" -o BatchMode=yes -o StrictHostKeyChecking=accept-new -o ConnectTimeout=15 -o ServerAliveInterval=15 -o ServerAliveCountMax=4 ) rc=0 for attempt in 1 2 3; do if [ "$attempt" -gt 1 ]; then echo ":: warning::ssh transport failed, retrying (${attempt}/3)" sleep $((attempt * 5)) fi rc=0 ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \ env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \ "DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \ "STAGE=$1" bash -se <<'EOF' || rc=$? source "$REPO/.gitea/workflows/deploy-lib.sh" run_stage "$STAGE" EOF [ "$rc" -eq 0 ] && break [ "$rc" -ne 255 ] && break done if [ "$rc" -ne 0 ]; then echo ":: error::stage $1 failed over ssh (exit $rc)" fi exit "$rc"