name: deploy on: # Deploy only what CI already validated. workflow_run is used instead of # workflow_dispatch so a red lint/validate run can never reach the cluster. workflow_run: workflows: [ci] types: [completed] workflow_dispatch: # The deploy jobs read the tree, then reach the cluster over SSH with the # deploy key. The Actions token itself is not part of that path, so it gets # read-only contents and no more. permissions: contents: read concurrency: group: deploy-main # Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and # takes the verify job down with it, so a superseded deploy would leave the # cluster half-applied and unchecked — the exact failure the verify job exists # to catch. kubectl apply and docker compose up are both idempotent, so letting # the older run finish and then deploying the newer commit costs little. cancel-in-progress: false env: DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }} DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} # workflow_run's own GITHUB_SHA points at the branch head, not at the commit the # finished ci run checked. Pin the exact validated commit instead, so a push # landing mid-deploy cannot make the workstation deploy something else. Also # what the verify job checks the snapshot against. Empty for workflow_dispatch, # which falls back to the current origin/main. DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }} jobs: preflight: if: >- github.event_name != 'workflow_run' || (github.event.workflow_run.conclusion == 'success' && github.event.workflow_run.head_branch == 'main') runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 10 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Fetch and reset workstation shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh preflight validate: needs: [preflight] runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 20 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Dry-run manifests and check Secrets shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh validate apply-k8s: needs: [validate] runs-on: [self-hosted, linux, arch, homelab, prod] # Apply only, no verification, so this is just the work itself: snapshot, # then up to three sequential `helm upgrade --atomic --timeout 10m`, then the # apply loop. Verification has its own job and its own budget. timeout-minutes: 45 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Apply Kubernetes manifests shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh apply-k8s apply-compose: needs: [validate] runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 30 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Redeploy docker compose stacks shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh apply-compose # Watches the workloads this deploy changed and rolls back the ones that never # became healthy. Runs even when the apply jobs failed, timed out or were # cancelled — that is the whole point of splitting it out. `always()` is what # lets it start after a failed dependency; the needs on apply-compose are a # barrier, so verification begins only once both applies are done. verify-k8s: needs: [apply-k8s, apply-compose] if: >- always() && needs.apply-k8s.result != 'skipped' && needs.apply-compose.result != 'skipped' runs-on: [self-hosted, linux, arch, homelab, prod] # ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback. timeout-minutes: 30 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Verify workloads and roll back on failure shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh verify-k8s # Asks the public route of every active service whether it is actually # serving, which the rollout check above structurally cannot: a pod can # converge and still be crash-looping, or be listening on a port no Service # points at, or answer 500. # # `always()` for the same reason verify-k8s has it, and it runs after that job # specifically because a rollback is when a route most needs re-checking. The # needs is a barrier, not a filter: whether verify-k8s passed, failed or was # cancelled, the probes are what say whether the cluster is serving, and # suppressing them on a rollback would hide the one run where the answer # matters most. smoke: needs: [verify-k8s] if: always() && needs.verify-k8s.result != 'skipped' runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 10 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - name: Probe the public route of every active service shell: bash run: | set -euo pipefail ./.gitea/workflows/ssh-run.sh smoke