name: deploy on: # Deploy only what CI already validated. workflow_run is used instead of # workflow_dispatch so a red lint/validate run can never reach the cluster. workflow_run: workflows: [ci] types: [completed] workflow_dispatch: # The deploy jobs read the tree, then reach the cluster over SSH with the # deploy key. The Actions token itself is not part of that path, so it gets # read-only contents and no more. permissions: contents: read concurrency: group: deploy-main # Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and # takes the verify job down with it, so a superseded deploy would leave the # cluster half-applied and unchecked — the exact failure the verify job exists # to catch. kubectl apply and docker compose up are both idempotent, so letting # the older run finish and then deploying the newer commit costs little. cancel-in-progress: false env: DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }} DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} # workflow_run's own GITHUB_SHA points at the branch head, not at the commit the # finished ci run checked. Pin the exact validated commit instead, so a push # landing mid-deploy cannot make the workstation deploy something else. Also # what the verify job checks the snapshot against. Empty for workflow_dispatch, # which falls back to the current origin/main. DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }} jobs: preflight: if: >- github.event_name != 'workflow_run' || (github.event.workflow_run.conclusion == 'success' && github.event.workflow_run.head_branch == 'main') runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 10 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Fetch and reset workstation shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh preflight validate: needs: [preflight] runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 20 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Dry-run manifests and check Secrets shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh validate apply-k8s: needs: [validate] runs-on: [self-hosted, linux, arch, homelab, prod] # Apply only, no verification, so this is just the work itself: snapshot, # then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop. # Verification has its own job and its own budget. # # 45 is roughly four times the measured cost of the stage, which is # deliberately not raised on a theory: # # helm, healthy 3 no-op upgrades ~3-5 min # helm, one release bad rollback-on-failure spends its 10m, ~10-15 min # then rolls that one back # apply loop ~40 manifests, 4 of which ~1 min # resolve an image digest # restart_stale_images 7.6s to find 8 workloads, ~0.5 min # 9.8s to resolve their digests # # The helm figure is one release, not three: `set -e` aborts # upgrade_helm_releases on the first failure, so a broken release costs # 10m and the other two are never attempted. Multiplying 10m by three # overstates the worst case by 20 minutes. # # The 45 minutes this was last raised to 45 were still not enough, and the # job logs for those runs no longer exist, so what actually consumed the # budget is not known - the two measurable candidates above account for # ~15 of it. The unbounded `docker manifest inspect` against the registry's # known hang mode is now bounded inside registry_digest (25s timeout, 3 # attempts): a dead registry fails each owned image after ~85s instead of # hanging the stage, and a blinking one is retried instead of failing the # whole apply file. Still open: make the stage announce which manifest it # is working on, so a killed run leaves a diagnosable last line. timeout-minutes: 45 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Apply Kubernetes manifests shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh apply-k8s apply-compose: needs: [validate] runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 30 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Redeploy docker compose stacks shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh apply-compose # Watches the workloads this deploy changed and rolls back the ones that never # became healthy. Runs even when the apply jobs failed, timed out or were # cancelled — that is the whole point of splitting it out. `always()` is what # lets it start after a failed dependency; the needs on apply-compose are a # barrier, so verification begins only once both applies are done. verify-k8s: needs: [apply-k8s, apply-compose] if: >- always() && needs.apply-k8s.result != 'skipped' && needs.apply-compose.result != 'skipped' runs-on: [self-hosted, linux, arch, homelab, prod] # Not raised, because the arithmetic does not close. # # 32 workloads are under management and the wave width is 8, so the verify # itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when # every rollout times out rather than converging. That is already 20 of 30. # # The other 10 would have to absorb rollback, and rollback_workloads is a # serial `while read` loop at 300s per failed workload. 10 minutes buys two. # Any larger number is buying a bigger multiple of an unbounded term rather # than covering a known cost: 60 minutes buys eight, and 60 minutes is # therefore not a bound, it is a guess with two digits. # # The number becomes derivable the moment rollback uses the same wave width # as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160, # and 45 covers verify plus rollback at full width. That change is to the # recovery path and is not folded into a timeout edit. timeout-minutes: 30 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Verify workloads and roll back on failure shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh verify-k8s # Asks the public route of every active service whether it is actually # serving, which the rollout check above structurally cannot: a pod can # converge and still be crash-looping, or be listening on a port no Service # points at, or answer 500. # # `always()` for the same reason verify-k8s has it, and it runs after that job # specifically because a rollback is when a route most needs re-checking. The # needs is a barrier, not a filter: whether verify-k8s passed, failed or was # cancelled, the probes are what say whether the cluster is serving, and # suppressing them on a rollback would hide the one run where the answer # matters most. smoke: needs: [verify-k8s] if: always() && needs.verify-k8s.result != 'skipped' runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 10 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Probe the public route of every active service shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh smoke