verify_workloads ended on `[ -s "$failed_file" ]`, which is the opposite of what its own contract says. A non-empty file means something failed, so the function returned success exactly when a workload never came up, and failure when everything was fine. Every rollback was therefore skipped, and every deploy that changed anything ended red with an empty failure list and a bogus "Rolled back successfully". Worse, the check only ever ran at the end of stage_apply_k8s, inside the same process as the apply. A job killed by timeout-minutes, cancelled by a new push, or cut off by a dropped SSH connection never reached it, which is precisely when a rollback matters. The three helm upgrades alone can consume the whole 30-minute job budget, so that path was reachable. Verification now lives in its own job, gated on always(), so it runs whatever happened to the apply. The apply stage publishes its pre-apply snapshot through DEPLOY_SNAPSHOT_DIR/current before touching anything, and the verify stage picks it up from there. A snapshot whose recorded commit does not match the deploy is refused rather than trusted, so a stale pointer from an earlier run cannot make the rollback revert the wrong workloads. An unwritable snapshot directory now fails the deploy up front instead of silently continuing without a way back. cancel-in-progress becomes false for the same reason: cancelling a run kills the apply job and takes the verify job with it, which is the failure this change exists to prevent. Both applies are idempotent, so queueing costs little. The SSH key moves to a per-run directory removed on exit, and the deploy is pinned to the exact commit CI validated.
120 lines
4.3 KiB
YAML
120 lines
4.3 KiB
YAML
name: deploy
|
|
|
|
on:
|
|
# Deploy only what CI already validated. workflow_run is used instead of
|
|
# workflow_dispatch so a red lint/validate run can never reach the cluster.
|
|
workflow_run:
|
|
workflows: [ci]
|
|
types: [completed]
|
|
workflow_dispatch:
|
|
|
|
concurrency:
|
|
group: deploy-main
|
|
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
|
|
# takes the verify job down with it, so a superseded deploy would leave the
|
|
# cluster half-applied and unchecked — the exact failure the verify job exists
|
|
# to catch. kubectl apply and docker compose up are both idempotent, so letting
|
|
# the older run finish and then deploying the newer commit costs little.
|
|
cancel-in-progress: false
|
|
|
|
env:
|
|
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
|
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
|
|
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
|
|
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
|
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
|
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
|
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
|
|
# finished ci run checked. Pin the exact validated commit instead, so a push
|
|
# landing mid-deploy cannot make the workstation deploy something else. Also
|
|
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
|
|
# which falls back to the current origin/main.
|
|
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
|
|
|
|
jobs:
|
|
preflight:
|
|
if: >-
|
|
github.event_name != 'workflow_run' ||
|
|
(github.event.workflow_run.conclusion == 'success' &&
|
|
github.event.workflow_run.head_branch == 'main')
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
timeout-minutes: 10
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Fetch and reset workstation
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh preflight
|
|
|
|
validate:
|
|
needs: [preflight]
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
timeout-minutes: 20
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Dry-run manifests and check Secrets
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh validate
|
|
|
|
apply-k8s:
|
|
needs: [validate]
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
# Apply only, no verification, so this is just the work itself: snapshot,
|
|
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
|
|
# apply loop. Verification has its own job and its own budget.
|
|
timeout-minutes: 45
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Apply Kubernetes manifests
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh apply-k8s
|
|
|
|
apply-compose:
|
|
needs: [validate]
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
timeout-minutes: 30
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Redeploy docker compose stacks
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh apply-compose
|
|
|
|
# Watches the workloads this deploy changed and rolls back the ones that never
|
|
# became healthy. Runs even when the apply jobs failed, timed out or were
|
|
# cancelled — that is the whole point of splitting it out. `always()` is what
|
|
# lets it start after a failed dependency; the needs on apply-compose are a
|
|
# barrier, so verification begins only once both applies are done.
|
|
verify-k8s:
|
|
needs: [apply-k8s, apply-compose]
|
|
if: >-
|
|
always() &&
|
|
needs.apply-k8s.result != 'skipped' &&
|
|
needs.apply-compose.result != 'skipped'
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
|
|
timeout-minutes: 30
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Verify workloads and roll back on failure
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh verify-k8s
|