diff --git a/.gitea/deploy-dependencies.json b/.gitea/deploy-dependencies.json new file mode 100644 index 0000000..7c5f3f8 --- /dev/null +++ b/.gitea/deploy-dependencies.json @@ -0,0 +1,3 @@ +{ + "postgres": ["authentik", "gitea", "immich", "n8n", "netbox", "netronome"] +} diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md new file mode 100644 index 0000000..803d303 --- /dev/null +++ b/.gitea/runner/README.md @@ -0,0 +1,122 @@ +# Homelab CI/CD + +The native Gitea runner runs on **vps**; production runs on **workstation**. +Jobs run on `homelab:host`, one at a time. No job images or Kubernetes credentials +are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows. + +## Runner installation + +Install Docker Engine with Compose and Buildx, Git, Python 3.11+, Bash, curl, +GNU tar/xz, flock and systemd using the host's package manager. Keep the existing +Gitea runner 3.0.2 binary at `/usr/local/bin/gitea-runner`. + +From this checkout on the VPS: + +```sh +sudo bash .gitea/runner/setup-runner.sh +``` + +The installer reuses `/var/lib/gitea-runner/.runner` and the existing service. +For a new host, install the same runner binary and register as `gitea-runner` +using the registration token interactively, label `homelab:host`, and working +directory `/var/lib/gitea-runner`; then rerun the installer. Tokens never belong +in this repository or command-line examples. + +Pinned tools live in the runner user's `~/.cache/homelab-ci`; CI repairs version +drift there. Installations are locked. Buildx uses only the `homelab-ci` builder, +pushes directly to the registry, and caps retained local cache at 1 GiB with a +2 GiB free-space target. This is not a hard limit on peak build disk usage. +Nothing runs `docker system prune`, removes unrelated images, or deletes volumes. + +## Workstation setup + +As the existing SSH deploy user on workstation: + +```sh +sudo loginctl enable-linger forust +bash .gitea/runner/setup-workstation.sh +``` + +The controller uses `/srv/homelab` as the persistent configuration tree and makes +a detached source worktree for each SHA. It never resets `/srv/homelab`, moves +local configuration, renames Compose projects, or changes volume names. +The installer records the current Kubernetes context and cluster UID in +`~/.config/homelab-deploy/environment`. Check these before installing. + +Configure Gitea Actions Variables: + +- `DEPLOY_HOST`, `DEPLOY_USER`, `DEPLOY_PORT`: the existing VPS-to-workstation SSH endpoint. +- `DEPLOY_KNOWN_HOSTS`: workstation's verified SSH host key entry for that endpoint. +- `AUTODEPLOY`: `false` initially; `true` enables deployment after successful main CI. + +Keep `DEPLOY_SSH_KEY`, `REGISTRY_USERNAME` and `REGISTRY_PASSWORD` in Actions +Secrets. Legacy endpoint secrets remain accepted during migration. The Actions +token must have repository read and Actions read access for release downloads. +The deploy user's existing Docker registry authentication remains necessary. + +## Releases and deployment + +CI publishes `release-` as a Gitea artifact with all three owned image +digests and build input fingerprints. Unchanged images are reused only from a +successful main CI artifact, never from `:prod`. EDU images remain pinned to the +digests released by their application repository. Expired artifacts cause CI to +rebuild images; they block deployment until CI is rerun. + +Run deploy from main with `deploy_ref=main` or a checked SHA: + +- `full`: required for the first baseline; reconcile all active components. +- `changed`: compare with the last fully successful production deploy. +- `plan`: validate configuration and show selection without changing production resources. +- `refresh_images=true`: explicitly refresh mutable third-party Compose tags. + +The manual and automatic paths both require successful CI, a successful build +job and the exact SHA's release artifact. PRs cannot publish images or deploy. +Removed resources are reported and require explicit removal; no automatic prune. +Service dependencies are listed in `.gitea/deploy-dependencies.json`. + +A workstation user systemd service holds the deploy lock across validation, +sequential apply, verification and smoke checks. SSH clients only submit/follow: +disconnecting or cancelling the Actions client does not kill production apply. +Retrying the same run ID does not start another apply. `ExecStopPost` recovers +interrupted runs before the unit finishes. Kubernetes rolls back to captured +revisions; configuration and persistent data are not reverted. + +## Status and recovery + +`--retry` repeats failed verification and smoke checks, never apply. Recovery +keeps a failed deploy out of the successful baseline, even after rollback. + +On workstation (replace the numeric ID with Actions run ID and attempt): + +```sh +python3 ~/.local/lib/homelab-deploy/controller.py status 123-1 +python3 ~/.local/lib/homelab-deploy/controller.py recover 123-1 --retry +journalctl --user -u homelab-deploy@123-1 +``` + +Runs live in `~/.local/state/homelab-deploy/runs`. Compose stores resolved configs +with restricted permissions; these may contain credentials and must never be +uploaded as CI artifacts. Stage logs print the exact manual recovery command +using `compose-before/.json`, the original project directory and project +name. Compose does not automatically roll back, and Nextcloud AIO's child +containers remain managed by AIO. Preserve its own backups for data recovery. + +The controller retains twenty successful/planned runs and preserves failures. +Update the workstation dispatcher only when no deploy is running. + +## Validation and migration rollback + +```sh +python3 -m unittest discover -s tests -v +bash .gitea/tests/deploy-validation.sh +``` + +Test on a separate namespace before the initial production `full` run. Check a +failed rollout, interrupted SSH and repeated run ID, and verify that an isolated +service change does not upgrade unrelated Helm releases or Compose stacks. + +To roll back the migration, disable autodeploy and finish or recover the remote +run first. Restore the runner config/unit from `.before-` backups, +reload systemd and restart the runner. Restore the prior workflows from Git. +Production data and persistent volumes stay where they were. Do not remove run +state or Compose recovery files until recovery is confirmed. diff --git a/.gitea/runner/buildkitd.toml b/.gitea/runner/buildkitd.toml new file mode 100644 index 0000000..9b23273 --- /dev/null +++ b/.gitea/runner/buildkitd.toml @@ -0,0 +1,11 @@ +[worker.oci] + gc = true + reservedSpace = "256MB" + maxUsedSpace = "1GB" + minFreeSpace = "2GB" + +[[worker.oci.gcpolicy]] + reservedSpace = "256MB" + maxUsedSpace = "1GB" + minFreeSpace = "2GB" + all = true diff --git a/.gitea/runner/config.yaml b/.gitea/runner/config.yaml new file mode 100644 index 0000000..77028e5 --- /dev/null +++ b/.gitea/runner/config.yaml @@ -0,0 +1,10 @@ +runner: + file: /var/lib/gitea-runner/.runner + capacity: 1 + timeout: 5h + labels: + - homelab:host +cache: + enabled: false +container: + docker_host: unix:///var/run/docker.sock diff --git a/.gitea/runner/gitea-runner.service b/.gitea/runner/gitea-runner.service new file mode 100644 index 0000000..8665bfc --- /dev/null +++ b/.gitea/runner/gitea-runner.service @@ -0,0 +1,18 @@ +[Unit] +Description=Gitea Actions runner +After=network-online.target docker.service +Wants=network-online.target + +[Service] +User=gitea-runner +Group=gitea-runner +SupplementaryGroups=docker +WorkingDirectory=/var/lib/gitea-runner +Environment=PATH=/var/lib/gitea-runner/.cache/homelab-ci/bin:/usr/local/bin:/usr/bin:/bin +ExecStart=/usr/local/bin/gitea-runner daemon --config /etc/gitea-runner/config.yaml +Restart=on-failure +RestartSec=5 +UMask=0077 + +[Install] +WantedBy=multi-user.target diff --git a/.gitea/runner/homelab-deploy@.service b/.gitea/runner/homelab-deploy@.service new file mode 100644 index 0000000..1e9ceab --- /dev/null +++ b/.gitea/runner/homelab-deploy@.service @@ -0,0 +1,12 @@ +[Unit] +Description=Homelab deploy %i + +[Service] +Type=exec +EnvironmentFile=%h/.config/homelab-deploy/environment +ExecStart=/usr/bin/python3 %h/.local/lib/homelab-deploy/controller.py execute %i +ExecStopPost=/usr/bin/python3 %h/.local/lib/homelab-deploy/controller.py recover %i +RuntimeMaxSec=5h +TimeoutStopSec=135min +KillMode=control-group +UMask=0077 diff --git a/.gitea/runner/setup-runner.sh b/.gitea/runner/setup-runner.sh new file mode 100755 index 0000000..bacd7e1 --- /dev/null +++ b/.gitea/runner/setup-runner.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +# Native host runner, with pinned user-space tools and no extra CI images. +set -euo pipefail +here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +[ "$(id -u)" -eq 0 ] || { echo 'Run with sudo on the runner host' >&2; exit 1; } +for tool in docker curl python3 git tar xz flock runuser systemctl; do + command -v "$tool" >/dev/null || { echo "Install missing prerequisite: $tool" >&2; exit 1; } +done +docker info >/dev/null +docker compose version >/dev/null +docker buildx version >/dev/null +id gitea-runner >/dev/null 2>&1 || useradd --system --create-home --home-dir /var/lib/gitea-runner --shell /usr/bin/bash gitea-runner +# Reuse the established service account and runner registration. +runner_home="$(getent passwd gitea-runner | cut -d: -f6)" +[ "$runner_home" = /var/lib/gitea-runner ] || { echo 'Unexpected runner home; inspect the existing service first' >&2; exit 1; } +runuser -u gitea-runner -- docker info >/dev/null || { echo "The runner user needs access to Docker before setup" >&2; exit 1; } +command -v gitea-runner >/dev/null || { echo 'Install gitea-runner 3.0.2 at /usr/local/bin/gitea-runner first' >&2; exit 1; } +mkdir -p /etc/gitea-runner +stamp="$(date -u +%Y%m%dT%H%M%SZ)" +for existing in /etc/gitea-runner/config.yaml /etc/systemd/system/gitea-runner.service; do + [ ! -f "$existing" ] || cp -p "$existing" "$existing.before-$stamp" +done +scratch="$(mktemp -d)" +trap 'rm -rf "$scratch"' EXIT +chmod 755 "$scratch" +install -m 0644 "$here/../workflows/install-ci-tools.sh" "$here/../workflows/tool-versions.env" "$scratch/" +runuser -u gitea-runner -- bash "$scratch/install-ci-tools.sh" +install -m 0644 "$here/config.yaml" /etc/gitea-runner/config.yaml +python3 - <<'PYLABELS' +import json +from pathlib import Path +registration = Path('/var/lib/gitea-runner/.runner') +if registration.exists(): + labels = json.loads(registration.read_text()).get('labels', []) + labels = [label for label in labels if isinstance(label, str) and label.split(':')[0] != 'homelab'] + labels.append('homelab:host') + config = Path('/etc/gitea-runner/config.yaml') + config.write_text(config.read_text().replace(' - homelab:host', '\n'.join(' - ' + json.dumps(label) for label in labels))) +PYLABELS +install -m 0644 "$here/gitea-runner.service" /etc/systemd/system/gitea-runner.service +if [ ! -f /var/lib/gitea-runner/.runner ]; then + echo 'Register once as gitea-runner with homelab:host before starting the service.' + exit 0 +fi +systemctl daemon-reload +systemctl enable --now gitea-runner.service +systemctl restart gitea-runner.service +echo "Runner ready. Configuration backups: *.before-$stamp" diff --git a/.gitea/runner/setup-workstation.sh b/.gitea/runner/setup-workstation.sh new file mode 100755 index 0000000..a6d5f9b --- /dev/null +++ b/.gitea/runner/setup-workstation.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# Run as the existing deploy user on workstation. Never resets the working tree. +set -euo pipefail +here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +repo="${HOMELAB_REPO:-/srv/homelab}" +for tool in python3 git kubectl helm docker flock timeout; do + command -v "$tool" >/dev/null || { echo "Install missing dependency: $tool" >&2; exit 1; } +done +[ -d "$repo/.git" ] || { echo "Missing deploy checkout: $repo" >&2; exit 1; } +[[ "$repo" =~ ^/[A-Za-z0-9_./-]+$ ]] || { echo 'Deploy path must be absolute and contain no whitespace' >&2; exit 1; } +if [ "$(loginctl show-user "$USER" -p Linger --value)" != yes ]; then + echo "Run once: sudo loginctl enable-linger $USER" >&2 + exit 1 +fi +config="${XDG_CONFIG_HOME:-$HOME/.config}/homelab-deploy" +mkdir -p "$config" "$HOME/.local/lib/homelab-deploy" "$HOME/.config/systemd/user" +chmod 700 "$config" +if [ ! -f "$config/environment" ]; then + context="$(kubectl config current-context)" + cluster_uid="$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" + printf 'HOMELAB_REPO=%s\nKUBE_CONTEXT=%s\nEXPECTED_CLUSTER_UID=%s\n' "$repo" "$context" "$cluster_uid" >"$config/environment" + chmod 600 "$config/environment" +fi +# Do not replace a dispatcher while an existing deploy uses it. +if systemctl --user list-units 'homelab-deploy@*' --state=running --no-legend | grep -q .; then + echo 'An existing deploy is running; wait before updating the controller' >&2 + exit 1 +fi +install -m 0755 "$here/../workflows/deploy-controller.py" "$HOME/.local/lib/homelab-deploy/controller.py" +install -m 0644 "$here/homelab-deploy@.service" "$HOME/.config/systemd/user/homelab-deploy@.service" +systemctl --user daemon-reload +echo 'Controller ready. Run a checked main SHA in full mode for the initial baseline.' diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 70bde19..6bb9910 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -1,38 +1,29 @@ name: ci - -on: +"on": push: branches: - main - pull_request: - workflow_dispatch: - -# Every job here is checkout plus local tools. The token needs to read the tree -# and nothing else, and saying so keeps a future step that reaches for the API -# from quietly holding a token that can write to the repository. + pull_request: null + workflow_dispatch: null permissions: contents: read - + actions: read concurrency: group: ci-${{ github.ref }} cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} - -env: - REGISTRY: gcr.forust.xyz - jobs: - lint-compose: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 + checks: + runs-on: homelab + timeout-minutes: 30 steps: - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - # Structure check for every committed Compose file, active or not. - # Interpolation, env-file and bind-mount resolution are all switched off, - # because inactive stacks have no .env here and would only fail on their - # ${VAR:?} guards. Active stacks get the full check with interpolation in - # the deploy workflow, where the real .env files live. + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Validate Compose files shell: bash run: | @@ -61,35 +52,15 @@ jobs: exit 1 fi echo "checked ${#files[@]} Compose file(s)" - - lint-actionlint: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint Gitea Actions workflows with actionlint shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint)" - export PATH="$tools_dir:$PATH" actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml - - lint-shellcheck: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint shell scripts with ShellCheck shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" - export PATH="$tools_dir:$PATH" mapfile -t scripts < <( git ls-files '*.sh' ':(glob)**/*.bash' ) @@ -99,20 +70,10 @@ jobs: fi shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" bash .gitea/tests/deploy-validation.sh - - lint-prettier: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Check formatting with Prettier shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)" - export PATH="$tools_dir:$PATH" mapfile -t prettier_files < <( git ls-files \ @@ -126,37 +87,17 @@ jobs: fi prettier --check --ignore-unknown "${prettier_files[@]}" - - lint-ruff: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint and format-check Python with Ruff shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff)" - export PATH="$tools_dir:$PATH" - ruff check . - ruff format --check . + ruff check . .gitea/workflows + ruff format --check . .gitea/workflows python3 -m unittest discover -s tests -v - - lint-yaml: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint YAML syntax shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)" - export PATH="$tools_dir:$PATH" mapfile -t yaml_files < <( git ls-files '*.yaml' '*.yml' \ @@ -170,20 +111,10 @@ jobs: fi yamllint -c .yamllint "${yaml_files[@]}" - - lint-dockerfiles: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint Dockerfiles shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)" - export PATH="$tools_dir:$PATH" mapfile -t dockerfiles < <( git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*' @@ -195,20 +126,10 @@ jobs: fi hadolint -c .hadolint.yaml "${dockerfiles[@]}" - - validate: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 20 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Validate Kubernetes manifests against JSON schemas shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)" - export PATH="$tools_dir:$PATH" mapfile -t manifests < <( git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \ @@ -225,317 +146,27 @@ jobs: -ignore-missing-schemas \ -summary \ "${manifests[@]}" - - # kubeconform has no schemas for CRDs, so every IngressRoute, Certificate, - # PrometheusRule, Middleware, ServersTransport and ServiceMonitor is silently - # skipped above. The live API server knows the real CRD schemas (and runs the - # cert-manager / Traefik admission webhooks), so validate there too. - # - # Only services marked with a k8s/active marker are checked: server-side - # dry-run needs the target namespace to exist, and inactive services are not - # deployed. Services being enabled for the first time are still covered by - # the JSON-schema pass above. - # - # Main pushes only. `--dry-run=server` persists nothing, but it does execute - # the admission webhooks of the production API server, so anyone able to open - # a pull request would be able to run arbitrary manifest content through - # cert-manager and Traefik. A pull request has nothing to gain from it either: - # only main is ever deployed, and this job runs to completion before the - # deploy workflow is allowed to start, so a bad CRD is still caught before - # anything reaches the cluster -- just on the push rather than on the PR. - - name: Note the server-side check is not running here - if: github.event_name == 'pull_request' || github.ref != 'refs/heads/main' - shell: bash - run: | - echo "::notice::Skipping the server-side dry-run. It executes the cert-manager and" \ - "Traefik admission webhooks against the production API server, so it is limited" \ - "to pushes to main. CRDs are still schema-checked by kubeconform above, and the" \ - "server-side pass still runs on main before the deploy." - - - name: Validate active manifests against the live API server - if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main' - shell: bash - run: | - set -euo pipefail - - if ! kubectl get --raw='/readyz' --request-timeout=10s >/dev/null 2>&1; then - echo "::warning::Cluster unreachable — skipped server-side validation of CRDs (IngressRoute, Certificate, PrometheusRule). Review manifest changes manually." - exit 0 - fi - - mapfile -t k8s_dirs < <( - git ls-files '*.yaml' '*.yml' \ - | grep -E '(^|/)k8s/' \ - | sed -E 's#((^|.*/)k8s)/.*#\1#' \ - | sort -u - ) - - manifests=() - kustomize_apps=() - for dir in "${k8s_dirs[@]}"; do - if [ ! -f "${dir}/active" ]; then - echo "skip (no k8s/active): ${dir}" - continue - fi - if [ -f "${dir}/overlays/prod/kustomization.yaml" ]; then - kustomize_apps+=("${dir}/overlays/prod") - elif [ -f "${dir}/base/kustomization.yaml" ]; then - kustomize_apps+=("${dir}/base") - else - while IFS= read -r f; do - [ -n "$f" ] && manifests+=("$f") - done < <( - git ls-files "${dir}/*.yaml" "${dir}/*.yml" \ - | grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' - ) - fi - done - - echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps" - failed=0 - for m in ${manifests[@]+"${manifests[@]}"}; do - if [[ "$m" == "prometheus-stack/k8s/vmagent.yaml" ]] \ - && ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then - echo "skip server-side dry-run until the VictoriaMetrics Operator CRD is installed: $m" - continue - fi - if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then - failed=1 - echo "::error file=${m}::$(printf '%s' "$out" | head -1)" - fi - done - for k in ${kustomize_apps[@]+"${kustomize_apps[@]}"}; do - if ! out="$(kubectl apply -k "$k" --dry-run=server 2>&1)"; then - failed=1 - echo "::error file=${k}::$(printf '%s' "$out" | head -1)" - fi - done - - if [ "$failed" -ne 0 ]; then - echo "Server-side validation failed. The API server (or an admission webhook) rejected these manifests." - exit 1 - fi - echo "server-side dry-run: all active manifests accepted by the API server" - build: needs: - # The panel's scan-deps/test-backend/test-frontend jobs gated here until - # userbot moved to its own repo; upstream's code is upstream's gate now. - # The rule is unchanged: publishing and passing the checks are the same - # gate, so a commit that fails any of these still cannot move :prod. - [lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate] - if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev') && !startsWith(github.ref_name, 'renovate/') - runs-on: [self-hosted, linux, arch, homelab] + - checks + if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main' + runs-on: homelab timeout-minutes: 60 - outputs: - services: ${{ steps.services.outputs.services }} steps: - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 with: fetch-depth: 0 - - - name: Detect changed docker-built services - id: services - shell: bash - env: - PUSH_BEFORE: ${{ github.event.before }} - run: | - set -euo pipefail - base="${PUSH_BEFORE:-}" - empty_tree="$(git hash-object -t tree /dev/null)" - if [[ "$base" =~ ^0{40}$ ]]; then - base="$empty_tree" - elif [[ ! "$base" =~ ^[0-9a-fA-F]{40}$ ]] || ! git cat-file -e "${base}^{commit}" 2>/dev/null; then - # Some Gitea push payloads expose `before` as multiple root commits - # joined by newlines. It is not a usable diff base; use this push's - # first parent so image changes in the current commit are still built. - base="$(git rev-parse "${GITHUB_SHA}^" 2>/dev/null || printf '%s' "$empty_tree")" - echo "::warning::invalid push-before value; comparing against ${base}" - fi - - # A failed diff used to leave changed_files empty, which reads exactly - # like "nothing to build": the job went green having built nothing and - # the tag never moved. The status is checked, not assumed. - if ! changed="$(git diff --name-only "$base" "${GITHUB_SHA}")"; then - echo "::error::cannot diff ${base}..${GITHUB_SHA}" - exit 1 - fi - mapfile -t changed_files <<<"$changed" - - services=() - - add_service() { - local name="$1" - local seen=0 - for existing in "${services[@]}"; do - if [ "$existing" = "$name" ]; then - seen=1 - break - fi - done - if [ "$seen" -eq 0 ]; then - services+=("$name") - fi - } - - for file in "${changed_files[@]}"; do - case "$file" in - errorpages/*) - add_service errorpages - ;; - homepages/*) - add_service homepages - ;; - esac - done - - if [ "${#services[@]}" -eq 0 ]; then - echo "No docker-built services changed." - echo "services=" >> "$GITHUB_OUTPUT" - exit 0 - fi - - printf '%s\n' "${services[@]}" | tee /tmp/services.txt - echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT" - - - name: Log in to registry - # The pin step below also writes (manifest PUTs), and it runs on every - # main push — including manifest-only ones where services is empty. A - # stale persistent login on the old runner used to mask this; a clean - # runner pushes anonymously and gets 401. - if: steps.services.outputs.services != '' || github.ref_name == 'main' - shell: bash - # Through env, not by substitution into the script. A secret written - # into a run: block is pasted into the shell source before bash parses - # it, so a password containing a quote, a backtick or $(...) becomes - # code that runs. Masking the value in the log does not prevent that. + - name: Build changed images and write release env: + GITEA_TOKEN: ${{ github.token }} REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }} REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }} - run: | - set -euo pipefail - printf '%s' "$REGISTRY_PASSWORD" | docker login "${REGISTRY}" \ - -u "$REGISTRY_USERNAME" \ - --password-stdin - - - name: Build and push changed images - if: steps.services.outputs.services != '' - shell: bash - run: | - # This step was the one run: block in the workflow without it, and it - # is the one that cannot afford it: a docker push that failed partway - # through the loop used to be followed by more pushes, the loop's exit - # status came from the last one, and the job went green with half the - # images missing from the registry. - set -euo pipefail - IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}" - - # Tags for this push. The commit-pinned name is the point of this - # step: the deploy resolves it in preference to :prod, so a deploy - # that sat in the queue behind a later push still gets the build of - # the commit CI validated, instead of whatever :prod points at by the - # time it runs. See render_pinned in deploy-lib.sh. - commit_tag="" - if [ "${GITHUB_REF_NAME}" = "main" ]; then - commit_tag="sha-${GITHUB_SHA:0:12}" - fi - - set_tags() { - tags=() - case "${GITHUB_REF_NAME}" in - main) tags+=("main" "prod") ;; - dev) tags+=("dev") ;; - esac - if [ -n "$commit_tag" ]; then - tags+=("$commit_tag") - fi - } - - for service in "${services[@]}"; do - case "$service" in - errorpages) - image="${REGISTRY}/forust/error-pages" - set_tags - build_args=() - for tag in "${tags[@]}"; do - build_args+=(-t "${image}:${tag}") - done - docker build \ - --cache-from "type=registry,ref=${image}:buildcache" \ - --cache-to "type=registry,ref=${image}:buildcache,mode=max" \ - "${build_args[@]}" errorpages - for tag in "${tags[@]}"; do - docker push "${image}:${tag}" - done - ;; - homepages) - for variant in forust xdfnx; do - case "$variant" in - forust) - image="${REGISTRY}/forust/forust-homepage" - ;; - xdfnx) - image="${REGISTRY}/forust/xdfnx-homepage" - ;; - esac - set_tags - build_args=() - for tag in "${tags[@]}"; do - build_args+=(-t "${image}:${tag}") - done - docker build \ - --cache-from "type=registry,ref=${image}:buildcache" \ - --cache-to "type=registry,ref=${image}:buildcache,mode=max" \ - "${build_args[@]}" -f "homepages/Dockerfile.${variant}" homepages - for tag in "${tags[@]}"; do - docker push "${image}:${tag}" - done - done - ;; - esac - done - - # Every image the tree names has to carry the commit-pinned name, not only - # the ones this push rebuilt. A push that touches nothing but manifests - # builds nothing, and its deploy would then find no commit-pinned tag to - # resolve and quietly fall back to the moving :prod - which is the whole - # failure the commit-pinned name exists to remove. - # - # Re-tagging copies the manifest list and transfers no layers, so pinning - # six images that already exist costs six registry writes. - # - # The list is derived from the tree rather than written out here, so an - # image added to a manifest is covered without a second place to update. - - name: Pin the commit name on the images this push did not rebuild - if: github.ref_name == 'main' - shell: bash - run: | - set -euo pipefail - commit_tag="sha-${GITHUB_SHA:0:12}" - mapfile -t repos < <( - git grep -hoE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+' -- '*.yaml' '*.yml' \ - | sort -u - ) - if [ "${#repos[@]}" -eq 0 ]; then - echo "No own images referenced by the tree." - exit 0 - fi - echo "pinning ${#repos[@]} image(s) to $commit_tag" - for repo in "${repos[@]}"; do - # EDU images are released by the application repository and pinned - # directly by digest in edu_master manifests. Never retag them here. - case "$repo" in - */session-keeper|*/webinar-checker) continue ;; - esac - if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then - echo " already built by this push: ${repo##*/}" - continue - fi - if ! docker buildx imagetools inspect "$repo:prod" >/dev/null 2>&1; then - echo " WARNING: ${repo##*/} has no :prod to pin and no build produced it" - continue - fi - docker buildx imagetools create --tag "$repo:$commit_tag" "$repo:prod" - echo " pinned ${repo##*/}" - done + run: python3 .gitea/workflows/release.py build + - name: Store commit release + uses: actions/upload-artifact@c6a366c94c3e0affe28c06c8df20a878f24da3cf + with: + name: release-${{ github.sha }} + path: release.json + if-no-files-found: error + retention-days: 30 diff --git a/.gitea/workflows/compose-release.py b/.gitea/workflows/compose-release.py new file mode 100644 index 0000000..bad8d99 --- /dev/null +++ b/.gitea/workflows/compose-release.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Resolve Compose images without changing project names or local bind paths.""" + +import json +import os +import re +import subprocess +import sys +from pathlib import Path + + +def output(*args, **kwargs): + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603 + + +def resolve(reference): + if '@sha256:' in reference: + return reference + descriptor = json.loads( + output('docker', 'buildx', 'imagetools', 'inspect', reference, '--format', '{{json .Manifest}}') + ) + digest = descriptor['digest'] + if not re.fullmatch(r'sha256:[0-9a-f]{64}', digest): + raise ValueError(f'Invalid registry digest for {reference}') + # Strip tag only from the final path segment (registry ports are preserved). + repository = reference.rsplit('/', 1) + repository[-1] = repository[-1].split(':')[0] + return '/'.join(repository) + '@' + digest + + +def prepare(source_file): + config_repo = Path(os.environ['CONFIG_REPO']) + source_repo = Path(os.environ['REPO']) + directory = Path(os.environ['RUN_DIR']) + relative = source_file.relative_to(source_repo) + project_dir = config_repo / relative.parent + base = ['docker', 'compose', '--project-directory', str(project_dir), '-f', str(source_file)] + config = json.loads(output(*base, 'config', '--format', 'json', cwd=config_repo)) + project = config['name'] + previous_file = directory / 'previous.json' + previous = json.loads(previous_file.read_text()) if previous_file.exists() else {} + images_file = directory / 'compose-images.json' + locks = json.loads(images_file.read_text()) if images_file.exists() else previous.get('compose-images', {}) + release = json.loads((directory / 'release.json').read_text()) + before = json.loads(json.dumps(config)) + for service, settings in config['services'].items(): + reference = settings.get('image') + if not reference or settings.get('build'): + raise ValueError(f'{project}/{service}: Compose deploy requires a published image') + image_repo = reference.split('@')[0].rsplit('/', 1) + image_repo[-1] = image_repo[-1].split(':')[0] + image_repo = '/'.join(image_repo) + if image_repo in release['images']: + pinned = image_repo + '@' + release['images'][image_repo] + elif os.environ.get('REFRESH_IMAGES') != 'true' and reference in locks: + pinned = locks[reference] + else: + pinned = resolve(reference) + settings['image'] = pinned + locks[reference] = pinned + # Capture what is running, not the current value of its mutable tag. + ids = output( + 'docker', + 'ps', + '-aq', + '--filter', + f'label=com.docker.compose.project={project}', + '--filter', + f'label=com.docker.compose.service={service}', + ).splitlines() + actual = set() + for container in ids: + image_id = output('docker', 'inspect', container, '--format', '{{.Image}}') + digests = json.loads(output('docker', 'image', 'inspect', image_id, '--format', '{{json .RepoDigests}}')) + actual.add(next((d for d in digests or [] if d.split('@')[0] == image_repo), image_id)) + if len(actual) > 1: + raise ValueError(f'{project}/{service}: mixed running images, cannot capture one recovery config') + before['services'][service]['image'] = next(iter(actual)) if actual else reference + for name, data in (('compose', config), ('compose-before', before)): + folder = directory / name + folder.mkdir(mode=0o700, exist_ok=True) + destination = folder / f'{relative.parent.name}.json' + destination.write_text(json.dumps(data, indent=2) + '\n') + destination.chmod(0o600) + images_file.write_text(json.dumps(locks, indent=2) + '\n') + print(f'Compose {project}: images pinned; local paths preserved') + print( + f'Recovery: docker compose --project-directory {project_dir} -p {project} -f {directory}/compose-before/{relative.parent.name}.json up -d --pull never' + ) + + +if __name__ == '__main__': + prepare(Path(sys.argv[1])) diff --git a/.gitea/workflows/deploy-controller.py b/.gitea/workflows/deploy-controller.py new file mode 100644 index 0000000..432f672 --- /dev/null +++ b/.gitea/workflows/deploy-controller.py @@ -0,0 +1,323 @@ +#!/usr/bin/env python3 +"""Durable workstation deployment controller. Install with setup-workstation.sh.""" + +import argparse +import contextlib +import fcntl +import importlib.util +import json +import math +import os +import re +import shutil +import subprocess +import sys +import time +from pathlib import Path + +STATE = Path(os.environ.get('HOMELAB_STATE', Path.home() / '.local/state/homelab-deploy')) +CONFIG_REPO = Path(os.environ.get('HOMELAB_REPO', '/srv/homelab')) +RUN_ID = re.compile(r'[0-9]+-[0-9]+') + + +def command(*args, **kwargs): + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603, S607 + + +def atomic_json(path, data): + temporary = path.with_suffix('.tmp') + temporary.write_text(json.dumps(data, indent=2) + '\n') + temporary.chmod(0o600) + temporary.replace(path) + + +@contextlib.contextmanager +def lock(name): + STATE.mkdir(mode=0o700, parents=True, exist_ok=True) + with (STATE / name).open('a') as stream: + fcntl.flock(stream, fcntl.LOCK_EX) + yield + + +def load_module(name, path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def run_directory(run_id): + if not RUN_ID.fullmatch(run_id): + raise ValueError('Run ID must be numeric workflow-id and attempt') + return STATE / 'runs' / run_id + + +def start(run_id): + payload = sys.stdin.buffer.read(256 * 1024 + 1) + if len(payload) > 256 * 1024: + raise ValueError('Deploy request exceeds 256 KiB') + request = json.loads(payload) + sha = request['release']['sha'] + if not re.fullmatch(r'[0-9a-f]{40}', sha) or request['mode'] not in ('changed', 'full', 'plan'): + raise ValueError('Invalid deploy SHA or mode') + if not isinstance(request['refresh_images'], bool): + raise ValueError('refresh_images must be boolean') + directory = run_directory(run_id) + with lock('prepare.lock'): + if (directory / 'request.json').exists(): + if json.loads((directory / 'request.json').read_text()) != request: + raise ValueError('Run ID already belongs to a different request') + else: + directory.mkdir(mode=0o700, parents=True, exist_ok=True) + command('git', '-C', str(CONFIG_REPO), 'fetch', '--quiet', 'origin', 'main') + command('git', '-C', str(CONFIG_REPO), 'merge-base', '--is-ancestor', sha, 'origin/main') + if not (directory / 'source').exists(): + command('git', '-C', str(CONFIG_REPO), 'worktree', 'add', '--detach', str(directory / 'source'), sha) + if command('git', '-C', str(directory / 'source'), 'rev-parse', 'HEAD') != sha: + raise ValueError('Prepared source does not match deploy SHA') + release_module = load_module('release', directory / 'source/.gitea/workflows/release.py') + release_module.validate_release(request['release'], sha) + atomic_json(directory / 'release.json', request['release']) + atomic_json(directory / 'request.json', request) + if not (directory / 'status.json').exists(): + atomic_json(directory / 'status.json', {'state': 'queued', 'stages': {}}) + # Starting an existing active or finished ID is idempotent; never re-apply it. + if json.loads((directory / 'status.json').read_text())['state'] == 'queued': + command('systemctl', '--user', 'start', '--no-block', f'homelab-deploy@{run_id}.service') + print(f'Accepted deploy {run_id} ({sha})') + + +def environment(directory): + request = json.loads((directory / 'request.json').read_text()) + return { + **os.environ, + 'REPO': str(directory / 'source'), + 'CONFIG_REPO': str(CONFIG_REPO), + 'RUN_DIR': str(directory), + 'DEPLOY_SHA': request['release']['sha'], + 'RELEASE_FILE': str(directory / 'release.json'), + 'DEPLOY_PLAN': str(directory / 'plan.json'), + 'DEPLOY_SNAPSHOT_DIR': str(directory / 'snapshot'), + 'REFRESH_IMAGES': str(request['refresh_images']).lower(), + 'ROLLOUT_PARALLELISM': '4', + } + + +def stage(directory, name, budget): + status = json.loads((directory / 'status.json').read_text()) + if name in status['stages'] and status['stages'][name].get('result') in ('success', 'failure'): + return status['stages'][name]['result'] == 'success' + started = time.time() + status['stages'][name] = {'result': 'running', 'started': started} + atomic_json(directory / 'status.json', status) + script = directory / 'source/.gitea/workflows/deploy-stage.sh' + with (directory / f'{name}.log').open('a') as log: + # timeout kills the whole stage process group, including children, before recovery. + result = subprocess.run( # noqa: S603, S607 + [ + shutil.which('timeout') or '/usr/bin/timeout', + '--signal=TERM', + '--kill-after=30s', + str(budget), + 'bash', + str(script), + name, + ], + env=environment(directory), + stdout=log, + stderr=subprocess.STDOUT, + check=False, + ).returncode + status = json.loads((directory / 'status.json').read_text()) + status['stages'][name].update( + result='success' if result == 0 else 'failure', exit_code=result, seconds=round(time.time() - started) + ) + atomic_json(directory / 'status.json', status) + return result == 0 + + +def make_plan(directory): + source = directory / 'source' + planner = load_module('deploy_plan', source / '.gitea/workflows/deploy-plan.py') + request = json.loads((directory / 'request.json').read_text()) + previous = json.loads((STATE / 'last-success.json').read_text()) if (STATE / 'last-success.json').exists() else None + helm = json.loads(command('helm', 'list', '--all', '-A', '-o', 'json')) + plan = planner.make_plan(source, CONFIG_REPO, request['release'], previous, request['mode'], helm) + if request['refresh_images']: + plan['selected']['compose'] = plan['active']['compose'] + atomic_json(directory / 'plan.json', plan) + if previous: + atomic_json(directory / 'previous.json', previous) + # Local config is deliberately separate from the immutable Git source. + return plan + + +def finish_success(directory, plan): + # Repeating finalization after a crash is safe while holding deploy.lock. + plan['run_id'] = directory.name + path = directory / 'compose-images.json' + previous = directory / 'previous.json' + plan['compose-images'] = ( + json.loads(path.read_text()) + if path.exists() + else json.loads(previous.read_text()).get('compose-images', {}) + if previous.exists() + else {} + ) + atomic_json(STATE / 'last-success.json', plan) + status = json.loads((directory / 'status.json').read_text()) + status['state'] = 'success' + atomic_json(directory / 'status.json', status) + try: + retain_completed(directory) + except (OSError, subprocess.CalledProcessError) as error: + print(f'Retention deferred: {error}', flush=True) + + +def recover(directory, retry=False): + status = json.loads((directory / 'status.json').read_text()) + if status['state'] in ('success', 'planned'): + return + completed = ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke') + if all(status['stages'].get(name, {}).get('result') == 'success' for name in completed): + finish_success(directory, json.loads((directory / 'plan.json').read_text())) + return + if retry: + for name in ('verify-k8s', 'smoke'): + if status['stages'].get(name, {}).get('result') == 'failure': + del status['stages'][name] + atomic_json(directory / 'status.json', status) + snapshot = directory / 'snapshot/current' + if snapshot.exists(): + stage(directory, 'verify-k8s', 7200) + stage(directory, 'smoke', 600) + status = json.loads((directory / 'status.json').read_text()) + status['state'] = 'failure' + atomic_json(directory / 'status.json', status) + + +def execute(run_id): + directory = run_directory(run_id) + with lock('deploy.lock'): + status = json.loads((directory / 'status.json').read_text()) + if status['state'] != 'queued': + return + # A crashed predecessor must be recovered before another apply begins. + for other in (STATE / 'runs').iterdir(): + if ( + other != directory + and (other / 'status.json').exists() + and json.loads((other / 'status.json').read_text())['state'] == 'running' + ): + raise ValueError(f'Interrupted deploy {other.name}; run recover first') + status['state'] = 'running' + atomic_json(directory / 'status.json', status) + try: + plan = make_plan(directory) + print( + json.dumps({'selected': plan['selected'], 'helm': plan['helm'], 'manual_removals': plan['removed']}), + flush=True, + ) + if not stage(directory, 'doctor', 600): + raise RuntimeError('Preflight failed') + if not stage(directory, 'validate', 1200): + raise RuntimeError('Validation failed') + if json.loads((directory / 'request.json').read_text())['mode'] == 'plan': + status = json.loads((directory / 'status.json').read_text()) + status['state'] = 'planned' + atomic_json(directory / 'status.json', status) + return + # Budget includes both rollout checks and rollback waves, plus API overhead. + count = int( + command( + 'bash', + str(directory / 'source/.gitea/workflows/deploy-stage.sh'), + 'workload-count', + env=environment(directory), + ) + ) + verify_budget = max(600, 2 * math.ceil(count / 4) * 300 + 120) + if verify_budget > 7200: + raise ValueError('More than two hours of recovery required; split this deploy') + k8s_ok = stage(directory, 'apply-k8s', 2700) + compose_ok = stage(directory, 'apply-compose', 1800) if k8s_ok else False + verify_ok = stage(directory, 'verify-k8s', verify_budget) + smoke_ok = stage(directory, 'smoke', 600) + if not all((k8s_ok, compose_ok, verify_ok, smoke_ok)): + raise RuntimeError('Deploy failed; inspect stage logs and recovery report') + finish_success(directory, plan) + except Exception as error: + with (directory / 'controller.log').open('a') as stream: + stream.write(f'{error}\n') + recover(directory) + raise + + +def retain_completed(current): + finished = [] + for directory in (STATE / 'runs').iterdir(): + status_file = directory / 'status.json' + if status_file.exists() and json.loads(status_file.read_text())['state'] in ('success', 'planned'): + finished.append(directory) + for directory in sorted(finished, key=lambda p: p.stat().st_mtime, reverse=True)[20:]: + if directory == current: + continue + command('git', '-C', str(CONFIG_REPO), 'worktree', 'remove', '--force', str(directory / 'source')) + shutil.rmtree(directory) + + +def follow(run_id, phase): + directory = run_directory(run_id) + groups = { + 'apply': ('doctor', 'validate', 'apply-k8s', 'apply-compose'), + 'verify': ('verify-k8s',), + 'smoke': ('smoke',), + } + names = groups[phase] + offsets = {} + while True: + status = json.loads((directory / 'status.json').read_text()) + for name in (*names, 'controller'): + path = directory / f'{name}.log' + if path.exists(): + with path.open() as stream: + stream.seek(offsets.get(name, 0)) + content = stream.read() + if content: + print(content, end='', flush=True) + offsets[name] = stream.tell() + stages = status['stages'] + if all(stages.get(name, {}).get('result') in ('success', 'failure') for name in names): + return all(stages[name]['result'] == 'success' for name in names) + if status['state'] in ('success', 'failure', 'planned'): + return status['state'] in ('success', 'planned') + time.sleep(3) + + +def main(): + os.umask(0o077) + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow')) + parser.add_argument('run_id') + parser.add_argument('phase', nargs='?', choices=('apply', 'verify', 'smoke')) + parser.add_argument('--retry', action='store_true', help='Retry failed recovery checks; never repeat apply') + args = parser.parse_args() + directory = run_directory(args.run_id) + if args.action == 'start': + start(args.run_id) + elif args.action == 'execute': + execute(args.run_id) + elif args.action == 'recover': + with lock('deploy.lock'): + recover(directory, retry=args.retry) + elif args.action == 'status': + print((directory / 'status.json').read_text()) + if (directory / 'plan.json').exists(): + plan = json.loads((directory / 'plan.json').read_text()) + print(json.dumps({k: plan[k] for k in ('sha', 'selected', 'helm', 'removed')}, indent=2)) + elif not follow(args.run_id, args.phase): + sys.exit(1) + + +if __name__ == '__main__': + main() diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index b25c4cd..8731b90 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Shared stages for the deploy workflow. Runs on the workstation, invoked as: +# Workstation deploy stages; invoked by the durable controller against pinned source. # REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF' # source "$REPO/.gitea/workflows/deploy-lib.sh" # run_stage "$STAGE" @@ -8,8 +8,8 @@ set -euo pipefail : "${REPO:?REPO must be set}" APPLY_PRUNE="${APPLY_PRUNE:-false}" -# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to -# the current origin/main. +CONFIG_REPO="${CONFIG_REPO:-$REPO}" +# Exact SHA accepted by the CI gate for both manual and automatic deploys. DEPLOY_SHA="${DEPLOY_SHA:-}" # Handoff point between the apply stage (writes) and the verify stage (reads). # Under the deploy user's own XDG state directory rather than /var/backups: the @@ -20,7 +20,7 @@ DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state # Per-workload rollout budget and how many workloads to watch at once. The whole # apply job has its own timeout-minutes as a backstop. ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}" -ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}" +ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-4}" WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps" log() { @@ -60,6 +60,25 @@ kustomize_overlay() { fi } +selected_service() { + local kind="$1" service="$2" section=selected + [ -n "${DEPLOY_PLAN:-}" ] || return 0 + if [ "${DEPLOY_SMOKE_ALL:-false}" = true ]; then section=active; fi + jq -e --arg kind "$kind" --arg service "$service" --arg section "$section" \ + '.[$section][$kind] | index($service) != null' "$DEPLOY_PLAN" >/dev/null +} + +# Resolve .env and relative binds on the persistent workstation tree. Locked +# JSON configs keep the same Compose project name and volume names. +compose() { + local cf="$1" locked project_dir + project_dir="$CONFIG_REPO/$(basename "$(dirname "$cf")")" + shift + locked="${RUN_DIR:-/nonexistent}/compose/$(basename "$(dirname "$cf")").json" + if [ -f "$locked" ]; then cf="$locked"; fi + (cd "$CONFIG_REPO" && docker compose --project-directory "$project_dir" -f "$cf" "$@") +} + select_manifests() { K8S_MANIFESTS=() KUSTOMIZE_APPS=() @@ -67,6 +86,7 @@ select_manifests() { local kd_rel kd overlay cf_rel cf f while IFS= read -r kd_rel; do kd="$REPO/$kd_rel" + selected_service k8s "${kd_rel%/k8s}" || continue if [ ! -f "$kd/active" ]; then echo "skip (no k8s/active): $kd_rel" continue @@ -88,6 +108,7 @@ select_manifests() { ) while IFS= read -r cf_rel; do cf="$REPO/$cf_rel" + selected_service compose "$(dirname "$cf_rel")" || continue if [ -f "$(dirname "$cf")/active" ]; then echo "compose: $cf_rel" COMPOSE_STACKS+=("$cf") @@ -104,16 +125,10 @@ select_manifests() { # on failure roll them back to the revision that was running before, so a bad # push to main cannot leave a service crash-looping. # -# Verification lives in its own workflow job, not at the end of the apply stage. -# Inside a single process it is worthless exactly when it is needed most: a job -# killed by timeout-minutes or cancelled mid-apply never reaches the rollback -# code, and leaves a half-applied cluster behind. Split out, the apply job can -# die in any way and the verify job still runs. +# The workstation controller runs apply and verification as separate durable +# stages. Runner jobs only follow their logs. ExecStopPost recovers interrupted +# runs using the per-run snapshot, even when the SSH connection has gone away. # -# That split needs a handoff point on the workstation, because the two stages are -# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written -# before anything is applied, read by the verify stage afterwards. - # Creates this run's snapshot directory and publishes it as the handoff point for # the verify stage. Fails hard by design: a deploy that cannot record what it is # about to change must not start, because then nothing can be rolled back for it @@ -143,18 +158,35 @@ snapshot_dir() { } save_snapshot() { - local dir="$1" + local dir="$1" releases revision status log "Saving pre-apply snapshot to $dir" - workload_generations >"$dir/generations.before" 2>/dev/null \ - || warn "could not snapshot workload generations" - kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \ - || warn "could not snapshot workloads" - for release in prometheus-stack loki alloy; do - if helm status "$release" -n prometheus >/dev/null 2>&1; then - { - echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)" - helm get values "$release" -n prometheus --all 2>/dev/null - } >"$dir/helm-$release.txt" + workload_generations >"$dir/generations.before" || return 1 + kubectl get "$WORKLOAD_KINDS" -A -o json >"$dir/workloads.json" || return 1 + kubectl get controllerrevisions.apps -A -o json >"$dir/controller-revisions.json" || return 1 + jq --slurpfile revisions "$dir/controller-revisions.json" ' + [.items[] | . as $w | { + kind: (.kind | ascii_downcase), namespace: .metadata.namespace, name: .metadata.name, uid: .metadata.uid, + revision: (if .kind == "Deployment" then (.metadata.annotations["deployment.kubernetes.io/revision"] // "0" | tonumber) + else ([$revisions[0].items[] | select(.metadata.namespace == $w.metadata.namespace) + | select(any(.metadata.ownerReferences[]?; .uid == $w.metadata.uid)) + | select($w.kind != "StatefulSet" or .metadata.name == $w.status.currentRevision) | .revision] | max // 0) end) + }]' "$dir/workloads.json" >"$dir/revisions.json" || return 1 + releases="$(helm list --all -A -o json)" || return 1 + for entry in "${HELM_RELEASES[@]}"; do + IFS='|' read -r release _ namespace _ _ _ <<<"$entry" + if ! jq -e --arg r "$release" --arg n "$namespace" \ + 'any(.[]; .name == $r and .namespace == $n)' <<<"$releases" >/dev/null; then + continue + fi + helm status "$release" -n "$namespace" -o json >"$dir/helm-$release.json" || return 1 + status="$(jq -r '.info.status' "$dir/helm-$release.json")" + if [ "$status" != deployed ]; then + # Never capture a pending/failed revision as the recovery target. + helm history "$release" -n "$namespace" -o json >"$dir/helm-$release.history.json" || return 1 + revision="$(jq '[.[] | select(.status == "deployed" or .status == "superseded") | .revision] | max // 0' \ + "$dir/helm-$release.history.json")" + jq --argjson revision "$revision" '.version = $revision' "$dir/helm-$release.json" >"$dir/helm-$release.tmp" + mv "$dir/helm-$release.tmp" "$dir/helm-$release.json" fi done # The verify stage compares this against the commit it is deploying, to refuse @@ -178,294 +210,24 @@ workload_generations() { # moved since the snapshot, i.e. the ones this apply actually touched. changed_workloads() { local before="$1" - local ns name kind gen old + local ns name kind gen old current + current="$(workload_generations)" || return 1 while read -r ns name kind gen; do [ -n "${gen:-}" ] || continue - old="$(awk -v want_ns="$ns" -v want_name="$name" \ - '$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)" + old="$(awk -v want_ns="$ns" -v want_name="$name" -v want_kind="$kind" \ + '$1 == want_ns && $2 == want_name && $3 == want_kind { print $4; exit }' "$before" 2>/dev/null || true)" + if [ -n "${RUN_DIR:-}" ] && ! grep -qxF "$kind $ns $name" "$RUN_DIR/workload-refs"; then + continue + fi if [ "$old" != "$gen" ]; then printf '%s %s %s\n' "$kind" "$ns" "$name" fi - done < <(workload_generations) + done <<<"$current" } -# Prints " / " for every workload this repository owns that -# runs an image from our own registry. -# -# The repository is the scope, deliberately. The cluster also holds workloads on -# our registry that no manifest here declares (they are applied out of band), and -# those are somebody else's to deploy. Walking the manifests rather than the -# cluster means those can never be restarted by this pipeline, now or later. -owned_registry_workloads() { - local kd_rel f - while IFS= read -r kd_rel; do - [ -f "$REPO/$kd_rel/active" ] || continue - while IFS= read -r f; do - [ -n "$f" ] || continue - # A file that does not mention the registry cannot declare a workload on it, - # and parsing costs ~2.5s per file against a millisecond for the grep. The - # filter keeps this at a handful of parses instead of one per manifest. - grep -q 'gcr\.forust\.xyz/forust/' "$REPO/$f" 2>/dev/null || continue - # kubectl prints a bare object for a single-document file and a List for a - # multi-document one, so normalise both shapes before filtering. - kubectl apply --dry-run=client -f "$REPO/$f" -o json 2>/dev/null \ - | jq -r ' - (if .items then .items[] else . end) - | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) - | select(any((.spec.template.spec.containers // [])[]?; - (.image // "") | test("^gcr\\.forust\\.xyz/forust/"))) - | (.metadata.namespace // "default") as $ns - | ([.spec.template.spec.containers[].image - | select(test("^gcr\\.forust\\.xyz/forust/"))][0]) as $img - | "\($ns) \(.kind | ascii_downcase)/\(.metadata.name) \($img)" - ' 2>/dev/null || true - done < <(collect_k8s "$kd_rel" || true) - done < <( - git -C "$REPO" ls-files '*.yaml' '*.yml' \ - | grep -E '(^|/)k8s/' \ - | sed -E 's#((^|.*/)k8s)/.*#\1#' \ - | sort -u - ) -} - -# Prints the digest an image tag resolves to for this cluster's architecture, or -# nothing when it cannot be resolved. -# -# Only the manifest entry matching the node architecture counts. A multi-arch tag -# also carries `unknown/unknown` entries for the build attestation, and a pod's -# imageID is always the per-platform digest, so comparing the wrong entry would -# mark every workload stale forever and restart the whole cluster on every deploy. -registry_digest() { - local arch - arch="$(kubectl get nodes -o jsonpath='{.items[0].status.nodeInfo.architecture}' 2>/dev/null || true)" - [ -n "$arch" ] || arch=amd64 - # The || true is load-bearing. Every caller runs under set -euo pipefail, and - # pipefail reports the rightmost non-zero stage, so a ref the registry does not - # have would abort the caller at the assignment instead of yielding an empty - # string. The callers check for empty themselves and report it by name. - # - # Retried with a hard timeout because the registry has a known hang mode (and - # a known blink mode: a single failed lookup aborts the whole apply file in - # render_pinned). A short sleep between attempts lets a restarting registry - # come back instead of failing the deploy on one bad second. - local attempt=0 digest="" - while [ "$attempt" -lt 3 ]; do - digest="$(timeout 25s docker manifest inspect "$1" 2>/dev/null \ - | jq -r --arg arch "$arch" ' - .manifests[]? - | select(.platform.os == "linux" and .platform.architecture == $arch) - | .digest - ' 2>/dev/null \ - | head -1 || true)" - [ -n "$digest" ] && break - attempt=$((attempt + 1)) - if [ "$attempt" -lt 3 ]; then - echo "WARNING: registry lookup of $1 failed (attempt $attempt/3), retrying in 5s" >&2 - sleep 5 - fi - done - printf '%s' "$digest" -} - -# The commit this deploy is for: what CI validated, or - on a manual dispatch, -# whatever stage_preflight just checked out. -deploy_commit() { - local c="${DEPLOY_SHA:-}" - [ -n "$c" ] || c="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)" - printf '%.12s' "${c:-}" -} - -# Resolves one of our image refs to the digest THIS commit's build produced. -# -# A manifest naming `:prod` names a pointer, not a version, and the deploy -# resolves it when the apply runs - which is not when CI ran it. Deploy runs are -# queued rather than cancelled (see deploy.yaml), so two pushes in a row leave -# the first deploy resolving the second push's build: the right manifests with -# the wrong code, and nothing anywhere reports it. ci therefore publishes every -# image it ships under `sha-`, a name that cannot move, and that is -# the name resolved here. -# -# The fallback to the plain tag is for an image this pipeline never built. It -# reports itself, because a fallback nobody sees is the failure this removes. -pinned_digest() { - local ref="$1" commit pinned - commit="$(deploy_commit)" - if [ -n "$commit" ]; then - pinned="$(registry_digest "${ref%:*}:sha-$commit")" - if [ -n "$pinned" ]; then - printf '%s' "$pinned" - return 0 - fi - fi - pinned="$(registry_digest "$ref")" - if [ -n "$pinned" ]; then - echo "WARNING: ${ref} carries no sha-${commit:-} tag; resolved the moving tag instead" >&2 - fi - printf '%s' "$pinned" -} - -# Rewrites our own images to immutable digests on the way into the cluster. -# Reads a manifest stream on stdin, writes the pinned stream to stdout. -# -# A digest is not knowable when a manifest is written, so it is never committed: -# git keeps a readable `:prod` tag and the exact bytes are chosen here, at apply -# time, from the tag ci published for the commit being deployed. That is what -# makes rollback mean something. `kubectl rollout undo` restores the previous -# ReplicaSet's pod template verbatim, and a template naming a digest restores the -# exact bytes that were serving before. A template naming a moving tag does not — -# the tag has already moved by the time the rollback runs, so the "rollback" -# re-pulls the very image that just failed and the cluster stays broken. -# -# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a -# reference that is not `:latest` defaults to IfNotPresent, which is what the -# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot -# change, so pulling again buys nothing. -# -# An image that cannot be resolved is fatal. Carrying on would quietly apply a -# mutable tag again, which is the exact failure this function exists to remove. +# Resolve owned image references exclusively from the checked CI artifact. render_pinned() { - local src refs map ref digest missing=0 - src="$(mktemp)" - refs="$(mktemp)" - map="$(mktemp)" - - cat >"$src" - grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true - - while read -r ref; do - [ -n "$ref" ] || continue - digest="$(pinned_digest "$ref")" - if [ -z "$digest" ]; then - echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2 - echo " The build job has to push that tag before the deploy resolves it." >&2 - missing=$((missing + 1)) - continue - fi - printf '%s\t%s\n' "$ref" "$digest" >>"$map" - done <"$refs" - if [ "$missing" -gt 0 ]; then - rm -f "$src" "$refs" "$map" - return 1 - fi - - awk -v mapfile="$map" ' - BEGIN { - while ((getline line < mapfile) > 0) { - i = index(line, "\t") - d[substr(line, 1, i - 1)] = substr(line, i + 1) - } - } - { - if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) { - name = $0 - sub(/^[[:space:]]*image:[[:space:]]*/, "", name) - sub(/[[:space:]]*$/, "", name) - if (name in d) { - pad = $0 - sub(/image:.*/, "", pad) - # Drop the tag: the canonical form used in the docs is repo@sha256:..., - # and leaving :prod next to the digest reads like it still matters. - repo = name - sub(/:[A-Za-z0-9._-]+$/, "", repo) - print pad "image: " repo "@" d[name] - next - } - } - print - } - ' "$src" - rm -f "$src" "$refs" "$map" -} - -# Restarts every owned workload whose running image is not the one its tag -# resolves to now. -# -# This used to be how a rebuild reached the cluster at all: the manifests pinned -# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply` -# decided there was nothing to do, and the cluster served the previous build -# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves -# the pod template and rolls out on its own. -# -# What is left is the drift check: a hand-run `kubectl set image`, or anything -# else that edits a live workload behind the deploy's back, is the only way to end -# up serving a digest the tag has moved past. It stays idempotent, so a redeploy -# that changed no image still does not bounce healthy services. -# -# The container is matched on its repository rather than on the exact reference: -# once render_pinned has run, a pod's status reports `repo@sha256:...` while this -# still reads the repository's `:prod` tag out of the manifest. -restart_stale_images() { - local ns target image want selector running entry one - local unchecked=0 - local -A digests=() - local -a stale=() - while read -r ns target image; do - [ -n "${target:-}" ] || continue - if [ -z "${digests[$image]:-}" ]; then - digests[$image]="$(pinned_digest "$image")" - fi - want="${digests[$image]}" - if [ -z "$want" ]; then - warn "cannot resolve ${image##*/} in the registry, leaving $target alone" - unchecked=$((unchecked + 1)) - continue - fi - selector="$(kubectl get "$target" -n "$ns" -o jsonpath='{.spec.selector.matchLabels}' 2>/dev/null \ - | jq -r 'to_entries | map("\(.key)=\(.value)") | join(",")' 2>/dev/null)" - if [ -z "$selector" ]; then - warn "cannot read the pod selector of $target, skipping" - unchecked=$((unchecked + 1)) - continue - fi - running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \ - | jq -r --arg repo "${image%%:*}" ' - .items[] | .status.containerStatuses[]? - | select(.image == $repo - or (.image | startswith($repo + ":")) - or (.image | startswith($repo + "@"))) - | .imageID - ' 2>/dev/null)" - if [ -z "$running" ]; then - # Scaled to zero. Nothing is serving stale code, and imagePullPolicy - # resolves the tag when it is scaled back up. - continue - fi - entry="" - while IFS= read -r one; do - [ -n "$one" ] || continue - entry="${one##*@}" - if [ "$entry" != "$want" ]; then - stale+=("$ns $target") - break - fi - done <<<"$running" - done < <(owned_registry_workloads) - if [ "${#stale[@]}" -eq 0 ]; then - if [ "$unchecked" -gt 0 ]; then - # Say so plainly. Reporting "everything is current" after checking nothing - # would tell the operator the deploy is fine when it may not be. - warn "No workload needed a restart, but $unchecked could not be checked" - else - log "All owned workloads already run the image their tag points at" - fi - return 0 - fi - log "Restarting ${#stale[@]} workload(s) running an image their tag has moved past" - for ref in "${stale[@]}"; do - log " $ref" - done - local failed=() - for ref in "${stale[@]}"; do - ns="${ref%% *}" - target="${ref#* }" - if ! kubectl rollout restart "$target" -n "$ns" >/dev/null 2>&1; then - failed+=("$ref") - fi - done - if [ "${#failed[@]}" -gt 0 ]; then - warn "could not restart: ${failed[*]}" - return 1 - fi + python3 "$REPO/.gitea/workflows/release.py" render } # verify_workloads ... @@ -509,31 +271,44 @@ verify_workloads() { # settle. Prints a report and returns non-zero if any workload is still unhealthy, # so the operator knows manual recovery is required. rollback_workloads() { - local failed_file="$1" - local kind ns name unrecovered=() - local -a recovered=() + local failed_file="$1" snapshot kind ns name index=0 running=0 pid revision uid + local -a pids=() + snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")" while read -r kind ns name; do - [ -n "${kind:-}" ] || continue - # Helm-owned workloads are already rolled back by the release's --rollback-on-failure - # upgrade. `rollout undo` here would step back to the revision Helm just - # escaped (the failed one), so leave them for the operator instead. - if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then - echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}" - unrecovered+=("${kind}/${ns}/${name} (helm-managed)") - continue - fi - if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \ - && kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then - echo " rolled back: ${kind}/${ns}/${name}" - recovered+=("${kind}/${ns}/${name}") - else - echo " NOT RECOVERED: ${kind}/${ns}/${name}" - unrecovered+=("${kind}/${ns}/${name}") + [[ "$kind" =~ ^(deployment|statefulset|daemonset)$ ]] || continue + index=$((index + 1)) + ( + if kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.annotations}' | grep -q 'meta.helm.sh/release-name'; then + echo " skip (Helm recovery owns this workload): $kind/$ns/$name" + exit 1 + fi + revision="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \ + '.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .revision' "$snapshot/revisions.json")" + uid="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \ + '.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .uid' "$snapshot/revisions.json")" + if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]] || [ "$uid" != "$(kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.uid}')" ]; then + echo " no safe previous revision: $kind/$ns/$name (new or replaced workload)" + exit 1 + fi + kubectl rollout undo "$kind/$name" -n "$ns" --to-revision="$revision" \ + && kubectl rollout status "$kind/$name" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" + ) >"$snapshot/rollback-$index.log" 2>&1 & + pids+=($!) + running=$((running + 1)) + if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then + wait -n 2>/dev/null || true + running=$((running - 1)) fi done <"$failed_file" - echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file" - echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file" - [ "${#unrecovered[@]}" -eq 0 ] + local recovered=0 unrecovered=0 i=0 + for pid in "${pids[@]}"; do + i=$((i + 1)) + if wait "$pid"; then recovered=$((recovered + 1)); else unrecovered=$((unrecovered + 1)); fi + cat "$snapshot/rollback-$i.log" + done + echo "ROLLED_BACK=$recovered" >>"$failed_file" + echo "UNRECOVERED=$unrecovered" >>"$failed_file" + [ "$unrecovered" -eq 0 ] } # Helm releases owned by this stage, one line each: @@ -568,8 +343,12 @@ helm_repo_for() { helm_release_status() { local out if ! out="$(helm status "$1" -n "$2" 2>&1)"; then - echo "not-found" - return 0 + if [[ "$out" == *"release: not found"* ]]; then + echo "not-found" + return 0 + fi + printf 'ERROR: cannot read Helm status: %s\n' "$out" >&2 + return 1 fi awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]' } @@ -581,16 +360,27 @@ helm_release_status() { # pending (deployed, failed, not-found). Returns non-zero when the release is # still not recoverable, so the pipeline fails loud instead of wedging. recover_pending_release() { - local release="$1" namespace="$2" status - status="$(helm_release_status "$release" "$namespace")" + local release="$1" namespace="$2" status revision snapshot + status="$(helm_release_status "$release" "$namespace")" || return 1 case "$status" in pending-upgrade|pending-rollback|pending-install) log "Release $release is $status, rolling back to the last deployed revision" - if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then + revision="" + if [ -s "$DEPLOY_SNAPSHOT_DIR/current" ]; then + snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")" + if [ -s "$snapshot/helm-$release.json" ]; then + revision="$(jq -r '.version' "$snapshot/helm-$release.json")" + fi + fi + if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]]; then + echo "ERROR: no captured Helm revision for $release; manual recovery required" + return 1 + fi + if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then echo "WARN: helm rollback of $release did not complete" return 1 fi - status="$(helm_release_status "$release" "$namespace")" + status="$(helm_release_status "$release" "$namespace")" || return 1 if [ "$status" != "deployed" ]; then echo "WARN: $release is $status after rollback" return 1 @@ -626,11 +416,16 @@ upgrade_helm_releases() { local entry release chart namespace version values marker repo for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do IFS='|' read -r release chart namespace version values marker <<<"$entry" + if [ -n "${DEPLOY_PLAN:-}" ] && ! jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null; then + echo "skip (unchanged Helm release): $release" + continue + fi + if [ ! -f "$REPO/$values" ] && [ -f "$CONFIG_REPO/$values" ]; then values="$CONFIG_REPO/$values"; else values="$REPO/$values"; fi if [ ! -f "$REPO/$marker" ]; then echo "skip (no $marker): $release" continue fi - if [ ! -f "$REPO/$values" ]; then + if [ ! -f "$values" ]; then echo "ERROR: $values is gitignored but missing on the workstation, restore it first." return 1 fi @@ -639,8 +434,8 @@ upgrade_helm_releases() { echo "ERROR: no Helm repository configured for chart $chart" return 1 fi - helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true - helm repo update "${repo%% *}" >/dev/null 2>&1 || true + helm repo add "${repo%% *}" "${repo#* }" >/dev/null + helm repo update "${repo%% *}" >/dev/null log "Upgrading $release ($chart $version)" wait_for_calm "helm $release" # A previous run with --rollback-on-failure whose own rollback never finished leaves the @@ -656,7 +451,7 @@ upgrade_helm_releases() { if ! helm upgrade --install "$release" "$chart" \ --namespace "$namespace" \ --version "$version" \ - --values "$REPO/$values" \ + --values "$values" \ --wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then echo "WARN: upgrade of $release failed, checking release state" # --rollback-on-failure already attempted its own rollback; finish the job when that @@ -672,30 +467,42 @@ upgrade_helm_releases() { done } -stage_preflight() { - if [ ! -d "$REPO/.git" ]; then - echo "Repository not found at $REPO" - exit 1 - fi - if [ -n "$DEPLOY_SHA" ]; then - log "Checking out the commit CI validated ($DEPLOY_SHA)" - git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \ - || git -C "$REPO" fetch origin main - else - git -C "$REPO" fetch origin main - fi - target="${DEPLOY_SHA:-origin/main}" - log "Workstation state" - echo " local: $(git -C "$REPO" rev-parse --short HEAD)" - echo " target: $(git -C "$REPO" rev-parse --short "$target")" - if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then - echo "ERROR: workstation has local tracked modifications, refusing reset:" - git -C "$REPO" status --porcelain --untracked-files=no - git -C "$REPO" diff --stat - echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy." - exit 1 - fi - git -C "$REPO" reset --hard "$target" +stage_doctor() { + local tool entry release chart namespace version values marker + for tool in git docker kubectl helm jq curl timeout flock python3; do + command -v "$tool" >/dev/null || { echo "Missing workstation tool: $tool"; return 1; } + done + docker compose version >/dev/null + docker buildx version >/dev/null + [ "$(kubectl config current-context)" = "${KUBE_CONTEXT:?configure KUBE_CONTEXT}" ] || { echo "Unexpected Kubernetes context"; return 1; } + [ "$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" = "${EXPECTED_CLUSTER_UID:?configure EXPECTED_CLUSTER_UID}" ] || { echo "Unexpected Kubernetes cluster"; return 1; } + kubectl get --raw=/readyz --request-timeout=10s >/dev/null + [ "$(git -C "$REPO" rev-parse HEAD)" = "$DEPLOY_SHA" ] || return 1 + select_manifests + for entry in "${HELM_RELEASES[@]}"; do + IFS='|' read -r release chart namespace version values marker <<<"$entry" + [ -f "$REPO/$marker" ] || continue + [ -f "$REPO/$values" ] || [ -f "$CONFIG_REPO/$values" ] || { echo "Missing values: $values"; return 1; } + done + jq '{sha, selected, helm, removed}' "$DEPLOY_PLAN" + local cf + for cf in "${COMPOSE_STACKS[@]}"; do + compose "$cf" config --quiet + while IFS= read -r network; do + docker network inspect "$network" >/dev/null || return 1 + done < <(compose "$cf" config --format json | jq -r '.networks // {} | to_entries[] | select(.value.external == true) | .value.name') + python3 "$REPO/.gitea/workflows/compose-release.py" "$cf" + done + local image refs m k + refs="$( + for m in "${K8S_MANIFESTS[@]}"; do render_pinned <"$m" || return 1; done + for k in "${KUSTOMIZE_APPS[@]}"; do kubectl kustomize "$k" | render_pinned || return 1; done + )" || return 1 + refs="$(grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+@sha256:[0-9a-f]{64}' <<<"$refs" | sort -u || true)" + while IFS= read -r image; do + [ -n "$image" ] || continue + timeout 60s docker buildx imagetools inspect "$image" >/dev/null + done <<<"$refs" } # Required pod Secrets, scoped to the resource namespace. TLS route Secrets are @@ -759,7 +566,7 @@ stage_validate() { log "Validate compose stacks" for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do echo " config: $cf" - validate_compose_file "$cf" + compose "$cf" config --quiet done log "Validate k8s manifests (kubectl dry-run=client)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do @@ -786,6 +593,21 @@ stage_validate() { check_referenced_secrets } +selected_workload_refs() { + local m k + for m in "${K8S_MANIFESTS[@]}"; do + if skip_uninstalled_vmagent_crd "$m" >/dev/null; then continue; fi + kubectl create --dry-run=client --validate=false -f "$m" -o json | jq -r ' + (if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) + | "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"' + done + for k in "${KUSTOMIZE_APPS[@]}"; do + kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json | jq -r ' + (if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) + | "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"' + done +} + stage_apply_k8s() { check_prune_mode || return 1 cd "$REPO" @@ -793,16 +615,18 @@ stage_apply_k8s() { local ns_files=() other_files=() m k for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do case "$m" in - */namespace.y?ml) ns_files+=("$m") ;; + */namespace.yaml|*/namespace.yml) ns_files+=("$m") ;; *) other_files+=("$m") ;; esac done # Record what is about to change, and publish it for the verify job, before # the first apply. Both are fatal on failure: see snapshot_dir. + selected_workload_refs >"$RUN_DIR/workload-refs" local snapshot snapshot="$(snapshot_dir)" || return 1 save_snapshot "$snapshot" || return 1 + touch "$snapshot/ready" if [ "${#ns_files[@]}" -gt 0 ]; then log "Applying namespaces (${#ns_files[@]} files)" @@ -810,8 +634,8 @@ stage_apply_k8s() { kubectl apply -f "$m" done fi - if [ -f "$REPO/prometheus-stack/k8s/active" ]; then - if [ ! -f "$REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then + if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then + if [ ! -f "$CONFIG_REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first." exit 1 fi @@ -821,6 +645,7 @@ stage_apply_k8s() { if [ "${#other_files[@]}" -gt 0 ]; then log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do + log "Applying ${m#"$REPO"/}" if ! render_pinned <"$m" | kubectl apply -f -; then echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 exit 1 @@ -834,7 +659,6 @@ stage_apply_k8s() { exit 1 fi done - restart_stale_images # No verification here on purpose. This stage may be killed at any point by # timeout-minutes, by the runner cancelling the job, or by a dropped SSH @@ -849,7 +673,7 @@ stage_apply_k8s() { # and rolls back the ones that never became healthy. stage_verify_k8s() { local pointer="$DEPLOY_SNAPSHOT_DIR/current" - local snapshot want have generations + local snapshot want have generations changed local -a touched=() if [ ! -s "$pointer" ]; then @@ -860,7 +684,7 @@ stage_verify_k8s() { return 1 fi snapshot="$(head -1 "$pointer")" - if [ ! -d "$snapshot" ]; then + if [ ! -d "$snapshot" ] || [ ! -f "$snapshot/ready" ]; then echo "ERROR: snapshot pointer refers to a missing directory: $snapshot" return 1 fi @@ -883,6 +707,13 @@ stage_verify_k8s() { fi echo " snapshot: $snapshot (commit ${have:0:12})" + local entry release chart namespace version values marker + for entry in "${HELM_RELEASES[@]}"; do + IFS='|' read -r release chart namespace version values marker <<<"$entry" + jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null || continue + recover_pending_release "$release" "$namespace" || return 1 + done + generations="$snapshot/generations.before" if [ ! -s "$generations" ]; then # Without a baseline we cannot tell which workloads the apply touched, so @@ -891,9 +722,10 @@ stage_verify_k8s() { : >"$generations" fi + changed="$(changed_workloads "$generations")" || return 1 while read -r kind ns name; do [ -n "${kind:-}" ] && touched+=("$kind $ns $name") - done < <(changed_workloads "$generations") + done <<<"$changed" log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)" if [ "${#touched[@]}" -eq 0 ]; then @@ -932,21 +764,19 @@ stage_verify_k8s() { verify_compose_stack() { local cf="$1" local expected running missing=() - expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)" - running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)" + expected="$(compose "$cf" config --format json | jq -r ' .services | to_entries[] | select(.value.restart != "no") | .key' | sort)" || return 1 + running="$(compose "$cf" ps --status running --services | sort)" || return 1 [ -n "$expected" ] || return 0 while IFS= read -r svc; do [ -n "$svc" ] || continue # restart:"no" services are allowed to have exited. - if ! printf '%s\n' "$running" | grep -qx "$svc" \ - && ! docker compose -f "$cf" config 2>/dev/null \ - | grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then + if ! printf '%s\n' "$running" | grep -qx "$svc"; then missing+=("$svc") fi done <<<"$expected" if [ "${#missing[@]}" -gt 0 ]; then echo " NOT RUNNING: ${missing[*]}" - docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true + compose "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true return 1 fi echo " all ${#expected} service(s) running" @@ -1030,10 +860,11 @@ traefik_routed_hosts() { # cases, so ask Traefik which routes it built and fail on the difference. stage_smoke() { cd "$REPO" + if [ -n "${DEPLOY_PLAN:-}" ] && jq -e '.full_smoke' "$DEPLOY_PLAN" >/dev/null; then + DEPLOY_SMOKE_ALL=true + fi select_manifests >/dev/null local -a hosts=() - # Not named failed: an array of that name already exists in restart_stale_images - # above, and a scalar shadowing an array is a trap rather than a shadow. local h code rc bad=0 while IFS= read -r h; do [ -n "$h" ] && hosts+=("$h") @@ -1041,8 +872,8 @@ stage_smoke() { if [ "${#hosts[@]}" -eq 0 ]; then # Nothing to probe means the extraction broke, not that the cluster is empty. - echo "ERROR: no public hostnames found in active manifests, refusing to report success" - return 1 + echo "No public routes in the selected components" + return 0 fi log "Probing ${#hosts[@]} public route(s)" @@ -1126,37 +957,17 @@ stage_apply_compose() { cd "$REPO" select_manifests >/dev/null local cf - log "Redeploying docker compose stacks (${#COMPOSE_STACKS[@]} stacks)" - for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do - echo " compose: $cf" - if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then - docker compose -f "$cf" build - docker compose -f "$cf" push - fi - docker compose -f "$cf" up -d --pull always --remove-orphans + for cf in "${COMPOSE_STACKS[@]}"; do + log "Applying Compose ${cf#"$REPO"/}" + compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans + verify_compose_stack "$cf" done - - local -a broken=() - for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do - echo " verifying: $cf" - if ! verify_compose_stack "$cf"; then - broken+=("$cf") - fi - done - if [ "${#broken[@]}" -gt 0 ]; then - echo - echo "ERROR: ${#broken[@]} compose stack(s) did not come up:" - printf ' - %s\n' "${broken[@]}" - echo "Compose stacks are not rolled back automatically: their images use mutable" - echo "':latest' tags, so there is no previous version to return to. Check the logs" - echo "above, then re-run the deploy once the cause is fixed." - return 1 - fi + echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)" } run_stage() { case "${1:?stage required}" in - preflight) stage_preflight ;; + doctor) stage_doctor ;; validate) stage_validate ;; apply-k8s) stage_apply_k8s ;; verify-k8s) stage_verify_k8s ;; diff --git a/.gitea/workflows/deploy-plan.py b/.gitea/workflows/deploy-plan.py new file mode 100644 index 0000000..f794ad8 --- /dev/null +++ b/.gitea/workflows/deploy-plan.py @@ -0,0 +1,124 @@ +#!/usr/bin/env python3 +"""Calculate selected components against the last fully successful deploy.""" + +import hashlib +import json +import re +import subprocess +from pathlib import Path + + +def output(*args, **kwargs): + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603 + + +def tracked(repo): + return output('git', '-C', str(repo), 'ls-files').splitlines() + + +def helm_releases(repo): + text = (repo / '.gitea/workflows/deploy-lib.sh').read_text() + return [line.split('|') for line in re.findall(r'^ "([^"\n]+\|[^"\n]+)"$', text, re.MULTILINE)] + + +def inventory(repo): + files = tracked(repo) + k8s = sorted( + {f.split('/k8s/')[0] for f in files if '/k8s/' in f and (repo / f.split('/k8s/')[0] / 'k8s/active').is_file()} + ) + compose = sorted( + { + str(Path(f).parent) + for f in files + if Path(f).name in ('compose.yaml', 'compose.yml') and (repo / Path(f).parent / 'active').is_file() + } + ) + return {'k8s': k8s, 'compose': compose} + + +def file_hash(path): + return hashlib.sha256(path.read_bytes()).hexdigest() if path.is_file() else 'missing' + + +def make_plan(repo, config_repo, release, previous, mode, live_helm): + active = inventory(repo) + all_services = set(active['k8s'] + active['compose']) + helm_inputs = {} + helm_selected = [] + for name, chart, namespace, version, values, marker in helm_releases(repo): + if not (repo / marker).is_file(): + continue + value_path = repo / values if (repo / values).is_file() else config_repo / values + if not value_path.is_file(): + raise ValueError(f'Missing Helm values: {values}') + stamp = hashlib.sha256(f'{chart}|{version}|{file_hash(value_path)}'.encode()).hexdigest() + helm_inputs[name] = stamp + live = next((h for h in live_helm if h['name'] == name and h['namespace'] == namespace), None) + if ( + mode == 'full' + or previous is None + or previous.get('helm_inputs', {}).get(name) != stamp + or live is None + or live.get('status') != 'deployed' + or live.get('chart') != f'{chart.split("/")[-1]}-{version}' + ): + helm_selected.append(name) + local_inputs = {} + for service in all_services: + candidates = [config_repo / service / '.env'] + if service in active['compose']: + candidates.append(config_repo / '.env') + cfg = config_repo / service / 'config' + if cfg.is_dir(): + candidates.extend( + p for p in cfg.rglob('*') if p.is_file() and p.suffix in ('.yaml', '.yml', '.json', '.conf') + ) + local_inputs[service] = hashlib.sha256( + '\n'.join(f'{p.relative_to(config_repo)}:{file_hash(p)}' for p in sorted(candidates)).encode() + ).hexdigest() + if previous is None: + if mode == 'changed': + raise ValueError('No successful baseline; run deploy in full mode first') + changed = set(all_services) + removed = [] + else: + paths = output('git', '-C', str(repo), 'diff', '--name-only', previous['sha'], release['sha']).splitlines() + changed = {path.split('/')[0] for path in paths} + if any(path.startswith('.gitea/') for path in paths): + changed |= all_services + changed |= {s for s in all_services if previous.get('local_inputs', {}).get(s) != local_inputs[s]} + for file in tracked(repo): + service = file.split('/')[0] + if service not in all_services or not file.endswith(('.yaml', '.yml')): + continue + text = (repo / file).read_text() + if any( + image in text and previous.get('images', {}).get(image) != digest + for image, digest in release['images'].items() + ): + changed.add(service) + removed = sorted( + set(previous.get('active', {}).get('k8s', []) + previous.get('active', {}).get('compose', [])) + - all_services + ) + removed += [path for path in paths if '/k8s/' in path and not (repo / path).exists()] + if mode == 'full': + changed = set(all_services) + dependencies = json.loads((repo / '.gitea/deploy-dependencies.json').read_text()) + while True: + expanded = changed | {dependent for service in changed for dependent in dependencies.get(service, [])} + if expanded == changed: + break + changed = expanded + return { + 'version': 1, + 'sha': release['sha'], + 'images': release['images'], + 'active': active, + 'selected': {kind: sorted(set(services) & changed) for kind, services in active.items()}, + 'helm': helm_selected, + 'helm_inputs': helm_inputs, + 'local_inputs': local_inputs, + 'removed': sorted(set(removed)), + 'full_smoke': mode == 'full' or 'traefik' in changed, + } diff --git a/.gitea/workflows/deploy-stage.sh b/.gitea/workflows/deploy-stage.sh new file mode 100755 index 0000000..6d630f3 --- /dev/null +++ b/.gitea/workflows/deploy-stage.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +set -euo pipefail +source "${REPO:?}/.gitea/workflows/deploy-lib.sh" +case "${1:?stage required}" in + workload-count) + select_manifests >/dev/null + selected_workload_refs | sort -u | wc -l + ;; + *) run_stage "$1" ;; +esac diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 6c455b8..2ac1f60 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -1,202 +1,104 @@ name: deploy on: - # Deploy only what CI already validated. workflow_run is used instead of - # workflow_dispatch so a red lint/validate run can never reach the cluster. workflow_run: workflows: [ci] branches: [main] types: [completed] workflow_dispatch: + inputs: + deploy_ref: + description: "Commit already checked by successful main CI (main or SHA)" + default: main + required: true + deploy_mode: + description: "First deploy requires full; plan changes no production resources" + type: choice + options: [changed, full, plan] + default: changed + refresh_images: + description: "Explicitly refresh mutable third-party Compose tags" + type: boolean + default: false -# The deploy jobs read the tree, then reach the cluster over SSH with the -# deploy key. The Actions token itself is not part of that path, so it gets -# read-only contents and no more. permissions: contents: read + actions: read concurrency: group: deploy-main - # Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and - # takes the verify job down with it, so a superseded deploy would leave the - # cluster half-applied and unchecked — the exact failure the verify job exists - # to catch. kubectl apply and docker compose up are both idempotent, so letting - # the older run finish and then deploying the newer commit costs little. cancel-in-progress: false env: - DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} - DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }} - DEPLOY_USER: ${{ secrets.DEPLOY_USER }} - DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }} + DEPLOY_HOST: ${{ vars.DEPLOY_HOST || secrets.DEPLOY_HOST }} + DEPLOY_PORT: ${{ vars.DEPLOY_PORT || secrets.DEPLOY_PORT }} + DEPLOY_USER: ${{ vars.DEPLOY_USER || secrets.DEPLOY_USER }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} - APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} - # workflow_run's own GITHUB_SHA points at the branch head, not at the commit the - # finished ci run checked. Pin the exact validated commit instead, so a push - # landing mid-deploy cannot make the workstation deploy something else. Also - # what the verify job checks the snapshot against. Empty for workflow_dispatch, - # which falls back to the current origin/main. - DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }} + DEPLOY_KNOWN_HOSTS: ${{ vars.DEPLOY_KNOWN_HOSTS }} + DEPLOY_RUN_ID: ${{ github.run_id }}-${{ github.run_attempt || 1 }} + DEPLOY_MODE: ${{ inputs.deploy_mode || 'changed' }} + REFRESH_IMAGES: ${{ inputs.refresh_images && 'true' || 'false' }} jobs: - preflight: - # Autodeploy defaults to OFF: pushes deploy only when the AUTODEPLOY repo - # variable is set to 'true' (Settings -> Actions -> Variables). A manual - # Run workflow always bypasses the switch: dispatching it is the explicit - # intent to deploy. + gate: if: >- + github.ref == 'refs/heads/main' && (vars.AUTODEPLOY == 'true' || github.event_name == 'workflow_dispatch') && (github.event_name != 'workflow_run' || - (github.event.workflow_run.conclusion == 'success' && - github.event.workflow_run.head_branch == 'main')) - runs-on: [self-hosted, linux, arch, homelab, prod] + (github.event.workflow_run.conclusion == 'success' && github.event.workflow_run.head_branch == 'main')) + runs-on: homelab timeout-minutes: 10 + outputs: + sha: ${{ steps.release.outputs.sha }} steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + fetch-depth: 0 + - name: Check successful CI and download the exact commit release + id: release + env: + GITEA_TOKEN: ${{ github.token }} + DEPLOY_REF: ${{ inputs.deploy_ref || 'main' }} + EVENT_SHA: ${{ github.event.workflow_run.head_sha }} + run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA" + - name: Submit durable deploy to workstation + run: bash .gitea/workflows/ssh-run.sh start - - name: Fetch and reset workstation - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh preflight - - validate: - needs: [preflight] - runs-on: [self-hosted, linux, arch, homelab, prod] - timeout-minutes: 20 + apply: + needs: [gate] + runs-on: homelab + timeout-minutes: 100 steps: - - name: Checkout repository + - name: Checkout checked commit uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + ref: ${{ needs.gate.outputs.sha }} + - name: Follow validation and sequential Kubernetes / Compose apply + run: bash .gitea/workflows/ssh-run.sh apply - - name: Dry-run manifests and check Secrets - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh validate - - apply-k8s: - needs: [validate] - runs-on: [self-hosted, linux, arch, homelab, prod] - # Apply only, no verification, so this is just the work itself: snapshot, - # then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop. - # Verification has its own job and its own budget. - # - # 45 is roughly four times the measured cost of the stage, which is - # deliberately not raised on a theory: - # - # helm, healthy 3 no-op upgrades ~3-5 min - # helm, one release bad rollback-on-failure spends its 10m, ~10-15 min - # then rolls that one back - # apply loop ~40 manifests, 4 of which ~1 min - # resolve an image digest - # restart_stale_images 7.6s to find 8 workloads, ~0.5 min - # 9.8s to resolve their digests - # - # The helm figure is one release, not three: `set -e` aborts - # upgrade_helm_releases on the first failure, so a broken release costs - # 10m and the other two are never attempted. Multiplying 10m by three - # overstates the worst case by 20 minutes. - # - # The 45 minutes this was last raised to 45 were still not enough, and the - # job logs for those runs no longer exist, so what actually consumed the - # budget is not known - the two measurable candidates above account for - # ~15 of it. The unbounded `docker manifest inspect` against the registry's - # known hang mode is now bounded inside registry_digest (25s timeout, 3 - # attempts): a dead registry fails each owned image after ~85s instead of - # hanging the stage, and a blinking one is retried instead of failing the - # whole apply file. Still open: make the stage announce which manifest it - # is working on, so a killed run leaves a diagnosable last line. - timeout-minutes: 45 + verify: + needs: [gate, apply] + if: always() && needs.gate.result == 'success' + runs-on: homelab + timeout-minutes: 130 steps: - - name: Checkout repository + - name: Checkout checked commit uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + ref: ${{ needs.gate.outputs.sha }} + - name: Follow workload verification and recovery + run: bash .gitea/workflows/ssh-run.sh verify - - name: Apply Kubernetes manifests - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh apply-k8s - - apply-compose: - needs: [validate] - runs-on: [self-hosted, linux, arch, homelab, prod] - timeout-minutes: 30 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Redeploy docker compose stacks - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh apply-compose - - # Watches the workloads this deploy changed and rolls back the ones that never - # became healthy. Runs even when the apply jobs failed, timed out or were - # cancelled — that is the whole point of splitting it out. `always()` is what - # lets it start after a failed dependency; the needs on apply-compose are a - # barrier, so verification begins only once both applies are done. - verify-k8s: - needs: [preflight, apply-k8s, apply-compose] - if: >- - always() && - needs.preflight.result == 'success' && - needs.apply-k8s.result != 'skipped' && - needs.apply-compose.result != 'skipped' - runs-on: [self-hosted, linux, arch, homelab, prod] - # Not raised, because the arithmetic does not close. - # - # 32 workloads are under management and the wave width is 8, so the verify - # itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when - # every rollout times out rather than converging. That is already 20 of 30. - # - # The other 10 would have to absorb rollback, and rollback_workloads is a - # serial `while read` loop at 300s per failed workload. 10 minutes buys two. - # Any larger number is buying a bigger multiple of an unbounded term rather - # than covering a known cost: 60 minutes buys eight, and 60 minutes is - # therefore not a bound, it is a guess with two digits. - # - # The number becomes derivable the moment rollback uses the same wave width - # as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160, - # and 45 covers verify plus rollback at full width. That change is to the - # recovery path and is not folded into a timeout edit. - timeout-minutes: 30 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Verify workloads and roll back on failure - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh verify-k8s - - # Asks the public route of every active service whether it is actually - # serving, which the rollout check above structurally cannot: a pod can - # converge and still be crash-looping, or be listening on a port no Service - # points at, or answer 500. - # - # `always()` for the same reason verify-k8s has it, and it runs after that job - # specifically because a rollback is when a route most needs re-checking. The - # needs is a barrier, not a filter: whether verify-k8s passed, failed or was - # cancelled, the probes are what say whether the cluster is serving, and - # suppressing them on a rollback would hide the one run where the answer - # matters most. smoke: - needs: [preflight, verify-k8s] - if: >- - always() && - needs.preflight.result == 'success' && - needs.verify-k8s.result != 'skipped' - runs-on: [self-hosted, linux, arch, homelab, prod] - timeout-minutes: 10 + needs: [gate, verify] + if: always() && needs.gate.result == 'success' + runs-on: homelab + timeout-minutes: 15 steps: - - name: Checkout repository + - name: Checkout checked commit uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Probe the public route of every active service - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh smoke + with: + ref: ${{ needs.gate.outputs.sha }} + - name: Follow public route checks + run: bash .gitea/workflows/ssh-run.sh smoke diff --git a/.gitea/workflows/install-ci-tools.sh b/.gitea/workflows/install-ci-tools.sh index 2fcb3b3..60c2d87 100755 --- a/.gitea/workflows/install-ci-tools.sh +++ b/.gitea/workflows/install-ci-tools.sh @@ -13,9 +13,14 @@ here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=tool-versions.env . "$here/tool-versions.env" -TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}" +TOOLS_DIR="${TOOLS_DIR:-${XDG_CACHE_HOME:-$HOME/.cache}/homelab-ci}" BIN_DIR="$TOOLS_DIR/bin" mkdir -p "$BIN_DIR" +# A runner may accept overlapping workflows even though each workflow is sequential. +exec 9>"$TOOLS_DIR/install.lock" +flock -w 300 9 +export UV_TOOL_DIR="$TOOLS_DIR/uv-tools" +export UV_CACHE_DIR="$TOOLS_DIR/uv-cache" # The just-installed tools must resolve inside this script too: callers only # prepend BIN_DIR to PATH after the script exits, so a bare `uv` below would # miss the binary install_uv just placed (exit 127 on a clean runner). @@ -48,7 +53,7 @@ esac fetch() { # fetch if command -v curl >/dev/null 2>&1; then - curl -sSLf --retry 3 -o "$2" "$1" + curl -sSLf --connect-timeout 15 --max-time 120 --retry 3 -o "$2" "$1" elif command -v wget >/dev/null 2>&1; then wget -q -O "$2" "$1" else @@ -88,10 +93,13 @@ installed_version() { # at_version at_version() { - case "$(installed_version "$1")" in - *"$2"*) return 0 ;; - *) return 1 ;; - esac + local version expected="${2#v}" + version="$(installed_version "$1")" + if [[ "$version" =~ (^|[^0-9.])v?([0-9]+(\.[0-9]+)+) ]]; then + [ "${BASH_REMATCH[2]}" = "$expected" ] + else + return 1 + fi } install_kubeconform() { @@ -176,6 +184,7 @@ install_pip_audit() { } install_prettier() { + install_node if at_version prettier "${PRETTIER_VERSION}"; then return 0 fi @@ -236,29 +245,41 @@ install_actionlint() { rm -rf "$tmp" } -wanted=("$@") -if [ "${#wanted[@]}" -eq 0 ]; then - wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint) -fi +main() { + wanted=("$@") + if [ "${#wanted[@]}" -eq 0 ]; then + wanted=(node jq kubeconform shellcheck actionlint prettier ruff yamllint hadolint) + fi -for tool in "${wanted[@]}"; do - case "$tool" in - kubeconform) install_kubeconform ;; - shellcheck) install_shellcheck ;; - jq) install_jq ;; - actionlint) install_actionlint ;; - prettier) install_prettier ;; - ruff) install_ruff ;; - yamllint) install_yamllint ;; - pip-audit) install_pip_audit ;; - hadolint) install_hadolint ;; - node) install_node ;; - uv) install_uv ;; - *) - echo "install-ci-tools: unknown tool: $tool" >&2 - exit 1 - ;; - esac -done + for tool in "${wanted[@]}"; do + case "$tool" in + kubeconform) install_kubeconform ;; + shellcheck) install_shellcheck ;; + jq) install_jq ;; + actionlint) install_actionlint ;; + prettier) install_prettier ;; + ruff) install_ruff ;; + yamllint) install_yamllint ;; + pip-audit) install_pip_audit ;; + hadolint) install_hadolint ;; + node) install_node ;; + uv) install_uv ;; + *) + echo "install-ci-tools: unknown tool: $tool" >&2 + exit 1 + ;; + esac + done -printf '%s\n' "$BIN_DIR" + for old in "$BIN_DIR"/node-* "$BIN_DIR"/prettier-*; do + [ -d "$old" ] || continue + case "$(basename "$old")" in + "node-$NODE_VERSION"|"prettier-$PRETTIER_VERSION") ;; + *) rm -rf "$old" ;; + esac + done + if [ -x "$BIN_DIR/uv" ]; then "$BIN_DIR/uv" cache prune >/dev/null; fi + printf '%s\n' "$BIN_DIR" +} + +if [ "${BASH_SOURCE[0]}" = "$0" ]; then main "$@"; fi diff --git a/.gitea/workflows/release.py b/.gitea/workflows/release.py new file mode 100644 index 0000000..daff5aa --- /dev/null +++ b/.gitea/workflows/release.py @@ -0,0 +1,350 @@ +#!/usr/bin/env python3 +"""CI release artifacts and the SHA-specific Gitea deployment gate (stdlib only).""" + +import argparse +import hashlib +import io +import itertools +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +import urllib.error +import urllib.parse +import urllib.request +import zipfile +from pathlib import Path + +SHA = re.compile(r'[0-9a-f]{40}') +DIGEST = re.compile(r'sha256:[0-9a-f]{64}') +IMAGES = { + 'error-pages': ('errorpages', 'errorpages/Dockerfile'), + 'forust-homepage': ('homepages', 'homepages/Dockerfile.forust'), + 'xdfnx-homepage': ('homepages', 'homepages/Dockerfile.xdfnx'), +} + +# These images are released by the EDU application repository. +EXTERNAL_IMAGES = {'gcr.forust.xyz/forust/session-keeper', 'gcr.forust.xyz/forust/webinar-checker'} + + +def command(*args, **kwargs): + """Arguments are passed directly to the executable, never to a shell.""" + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603, S607 + + +def validate_release(data, sha=None): + if data.get('version') != 1 or not SHA.fullmatch(data.get('sha', '')): + raise ValueError('Invalid release version or SHA') + if sha is not None and data['sha'] != sha: + raise ValueError('Release SHA does not match the checked CI commit') + expected = {f'gcr.forust.xyz/forust/{name}' for name in IMAGES} + if set(data.get('images', {})) != expected: + raise ValueError('Release must contain all owned images') + if not all(DIGEST.fullmatch(value) for value in data['images'].values()): + raise ValueError('Release has an invalid image digest') + if set(data.get('inputs', {})) != expected or not all( + re.fullmatch(r'[0-9a-f]{64}', value) for value in data['inputs'].values() + ): + raise ValueError('Release has invalid build input fingerprints') + return data + + +class NoRedirect(urllib.request.HTTPRedirectHandler): + def redirect_request(self, _req, _fp, _code, _msg, _headers, _newurl): + return None + + +class Gitea: + def __init__(self): + self.origin = os.environ['GITHUB_SERVER_URL'].rstrip('/') + if urllib.parse.urlsplit(self.origin).scheme != 'https': + raise ValueError('Gitea API must use HTTPS') + self.repository = os.environ['GITHUB_REPOSITORY'] + if not re.fullmatch(r'[\w.-]+/[\w.-]+', self.repository): + raise ValueError('Invalid Gitea repository') + self.token = os.environ['GITEA_TOKEN'] + self.base = f'{self.origin}/api/v1/repos/{self.repository}' + + def request(self, url, *, archive=False): + if not url.startswith(self.base + '/'): + raise ValueError('Refusing to send the Actions token to another origin') + req = urllib.request.Request(url, headers={'Authorization': f'token {self.token}'}) # noqa: S310 -- HTTPS origin validated above + opener = urllib.request.build_opener(NoRedirect()) + try: + response = opener.open(req, timeout=30) # noqa: S310 + except urllib.error.HTTPError as error: + if not archive or error.code not in (301, 302, 303, 307, 308): + raise RuntimeError(f'Gitea API returned HTTP {error.code}') from None + target = urllib.parse.urljoin(url, error.headers['Location']) + if urllib.parse.urlsplit(target).scheme != 'https': + raise ValueError('Artifact redirect must use HTTPS') from None + # Signed storage redirects must never receive the Gitea token. + response = urllib.request.urlopen(target, timeout=30) # noqa: S310 + with response: + payload = response.read(8 * 1024 * 1024 + 1) + if len(payload) > 8 * 1024 * 1024: + raise ValueError('Gitea response exceeds 8 MiB') + return payload if archive else json.loads(payload) + + def pages(self, path, key, **params): + for page in range(1, 101): + query = urllib.parse.urlencode({**params, 'page': page, 'limit': 50}) + data = self.request(f'{self.base}/{path}?{query}') + entries = data[key] + yield from entries + if len(entries) < 50: + return + raise RuntimeError('Gitea pagination limit exceeded') + + def successful_runs(self, sha=None): + params = {'branch': 'main', 'status': 'success', 'exclude_pull_requests': 'true'} + if sha: + params['head_sha'] = sha + for run in self.pages('actions/workflows/ci.yaml/runs', 'workflow_runs', **params): + if ( + run.get('status') == 'completed' + and run.get('conclusion') == 'success' + and run.get('head_branch') == 'main' + and run.get('event') in ('push', 'workflow_dispatch') + and (run.get('repository') or {}).get('full_name') == self.repository + and (run.get('head_repository') or run.get('repository') or {}).get('full_name') == self.repository + and (sha is None or run.get('head_sha') == sha) + ): + yield run + + def release(self, run): + sha = run['head_sha'] + jobs = list(self.pages(f'actions/runs/{run["id"]}/jobs', 'jobs')) + # A green workflow with a skipped build must not authorize a deploy. + if not any(job.get('name') == 'build' and job.get('conclusion') == 'success' for job in jobs): + raise ValueError('CI build job did not succeed') + artifacts = self.request(f'{self.base}/actions/runs/{run["id"]}/artifacts')['artifacts'] + matching = [a for a in artifacts if a['name'] == f'release-{sha}' and not a.get('expired')] + if len(matching) != 1: + raise ValueError('CI release artifact is missing, expired or ambiguous; rerun CI') + blob = self.request(f'{self.base}/actions/artifacts/{matching[0]["id"]}/zip', archive=True) + with zipfile.ZipFile(io.BytesIO(blob)) as archive: + files = [entry for entry in archive.infolist() if not entry.is_dir()] + if len(files) != 1 or files[0].filename != 'release.json' or files[0].file_size > 256 * 1024: + raise ValueError('Unexpected release archive contents') + return validate_release(json.loads(archive.read(files[0])), sha) + + +def fingerprint(context, dockerfile): + tree = command('git', 'ls-tree', '-r', 'HEAD', '--', context, dockerfile, '.gitea/workflows/release.py') + return hashlib.sha256(tree.encode()).hexdigest() + + +def gate(output, requested_ref, event_sha): + command('git', 'fetch', '--quiet', 'origin', 'main') + if event_sha: + if not SHA.fullmatch(event_sha): + raise ValueError('Invalid workflow_run SHA') + sha = event_sha + else: + if requested_ref == 'main': + requested_ref = 'origin/main' + sha = command('git', 'rev-parse', '--verify', '--end-of-options', f'{requested_ref}^{{commit}}') + if not SHA.fullmatch(sha): + raise ValueError('Invalid deploy SHA') + command('git', 'merge-base', '--is-ancestor', sha, 'origin/main') + api = Gitea() + runs = list(api.successful_runs(sha)) + if not runs: + raise ValueError(f'No successful main CI for {sha}; run CI before deploying') + release = api.release(max(runs, key=lambda run: run['id'])) + output.write_text(json.dumps(release, indent=2) + '\n') + if os.environ.get('GITHUB_OUTPUT'): + with Path(os.environ['GITHUB_OUTPUT']).open('a') as stream: + stream.write(f'sha={sha}\n') + print(f'CI gate accepted {sha}') + + +def build(output): + sha = command('git', 'rev-parse', 'HEAD') + if sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha): + raise ValueError('Build checkout does not match GITHUB_SHA') + api = Gitea() + previous = None + for run in sorted(itertools.islice(api.successful_runs(), 50), key=lambda item: item['id'], reverse=True): + if str(run['id']) == os.environ.get('GITHUB_RUN_ID'): + continue + try: + previous = api.release(run) + break + except ValueError: + # Expired artifacts only cost a rebuild; mutable tags are never a fallback. + continue + docker_config = tempfile.mkdtemp(prefix='homelab-registry-') + builder_config = Path.home() / '.cache/homelab-ci/buildx' + builder_config.mkdir(parents=True, exist_ok=True) + env = {**os.environ, 'DOCKER_CONFIG': docker_config, 'BUILDX_CONFIG': str(builder_config)} + try: + subprocess.run( # noqa: S603, S607 + [ + shutil.which('docker') or '/usr/bin/docker', + 'login', + 'gcr.forust.xyz', + '-u', + os.environ['REGISTRY_USERNAME'], + '--password-stdin', + ], + input=os.environ['REGISTRY_PASSWORD'], + text=True, + check=True, + env=env, + ) + builder = 'homelab-ci' + versions = dict( + re.findall(r'^([A-Z_]+)="([^"\n]+)"$', Path('.gitea/workflows/tool-versions.env').read_text(), re.MULTILINE) + ) + image = versions['BUILDKIT_IMAGE'] + signature = builder_config / 'homelab-ci-image' + exists = ( + subprocess.run( # noqa: S603 + [shutil.which('docker') or '/usr/bin/docker', 'buildx', 'inspect', builder], + capture_output=True, + env=env, + ).returncode + == 0 + ) + if exists and (not signature.exists() or signature.read_text().strip() != image): + command('docker', 'buildx', 'rm', '--keep-state', builder, env=env) + exists = False + if not exists: + command( + 'docker', + 'buildx', + 'create', + '--name', + builder, + '--driver', + 'docker-container', + '--driver-opt', + f'image={image}', + '--buildkitd-config', + '.gitea/runner/buildkitd.toml', + env=env, + ) + signature.write_text(image + '\n') + release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}} + for name, (context, dockerfile) in IMAGES.items(): + image = f'gcr.forust.xyz/forust/{name}' + inputs = fingerprint(context, dockerfile) + old_digest = (previous or {}).get('images', {}).get(image) + exists = False + if old_digest and previous['inputs'].get(image) == inputs: + exists = ( + subprocess.run( # noqa: S603, S607 + [ + shutil.which('docker') or '/usr/bin/docker', + 'buildx', + 'imagetools', + 'inspect', + f'{image}@{old_digest}', + ], + capture_output=True, + env=env, + timeout=60, + ).returncode + == 0 + ) + if exists: + print(f'Reuse {name}: inputs unchanged') + digest = old_digest + else: + print(f'Build {name}', flush=True) + metadata = Path(docker_config) / 'metadata.json' + command( + 'docker', + 'buildx', + 'build', + '--builder', + builder, + '--push', + '--platform', + 'linux/amd64', + '--provenance=false', + '--cache-from', + f'type=registry,ref={image}:buildcache', + '--cache-to', + f'type=registry,ref={image}:buildcache,mode=max', + '--tag', + f'{image}:sha-{sha}', + '--metadata-file', + str(metadata), + '--file', + dockerfile, + context, + env=env, + ) + digest = json.loads(metadata.read_text())['containerimage.digest'] + release['images'][image] = digest + release['inputs'][image] = inputs + validate_release(release, sha) + output.write_text(json.dumps(release, indent=2) + '\n') + finally: + # Cleanup errors must neither leak credentials nor mask the original build error. + try: + subprocess.run( # noqa: S603 + [ + shutil.which('docker') or '/usr/bin/docker', + 'buildx', + 'prune', + '--builder', + 'homelab-ci', + '--force', + '--max-used-space', + '1gb', + ], + env=env, + timeout=60, + ) + except (OSError, subprocess.TimeoutExpired): + print('CI builder cache cleanup deferred', flush=True) + finally: + shutil.rmtree(docker_config) + + +def render(stream, destination): + release = validate_release(json.loads(Path(os.environ['RELEASE_FILE']).read_text()), os.environ['DEPLOY_SHA']) + image_line = re.compile( + r"^(\s*(?:-\s*)?image:\s*)(['\"]?)(gcr\.forust\.xyz/forust/[\w.-]+)(?::[\w.-]+|@sha256:[0-9a-f]{64})\2(\s*(?:#.*)?)$" + ) + rendered = [] + for line in stream: + match = image_line.fullmatch(line.rstrip('\n')) + if match: + prefix, quote, image, tail = match.groups() + if image in EXTERNAL_IMAGES and f'{image}@sha256:' in line: + rendered.append(line) + continue + if image not in release['images']: + raise ValueError(f'Owned image missing from checked release: {image}') + line = f'{prefix}{quote}{image}@{release["images"][image]}{quote}{tail}\n' + elif re.match(r'\s*(?:-\s*)?image:', line) and 'gcr.forust.xyz/forust/' in line: + raise ValueError('Unsupported owned image syntax; refusing to apply a mutable tag') + rendered.append(line) + destination.writelines(rendered) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('action', choices=('build', 'gate', 'render')) + parser.add_argument('--output', type=Path, default=Path('release.json')) + parser.add_argument('--ref', default='main') + parser.add_argument('--event-sha', default='') + args = parser.parse_args() + if args.action == 'render': + render(sys.stdin, sys.stdout) + elif args.action == 'gate': + gate(args.output, args.ref, args.event_sha) + else: + build(args.output) + + +if __name__ == '__main__': + main() diff --git a/.gitea/workflows/renovate-ci.yaml b/.gitea/workflows/renovate-ci.yaml index 0886d50..5592dc7 100644 --- a/.gitea/workflows/renovate-ci.yaml +++ b/.gitea/workflows/renovate-ci.yaml @@ -26,7 +26,7 @@ permissions: jobs: validate-renovate: - runs-on: [self-hosted, linux, arch, homelab] + runs-on: homelab timeout-minutes: 20 steps: - name: Checkout repository diff --git a/.gitea/workflows/renovate-run.yaml b/.gitea/workflows/renovate-run.yaml index 658a603..05d221a 100644 --- a/.gitea/workflows/renovate-run.yaml +++ b/.gitea/workflows/renovate-run.yaml @@ -32,7 +32,7 @@ concurrency: jobs: run-renovate: - runs-on: [self-hosted, linux, arch, homelab] + runs-on: homelab timeout-minutes: 60 steps: - name: Checkout repository diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index b2d8441..cfc94dc 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -1,71 +1,56 @@ #!/usr/bin/env bash -# usage: ssh-run.sh -# Runs one deploy-lib.sh stage on the workstation over SSH. +# The SSH client submits once and follows durable stages on workstation. set -euo pipefail - : "${DEPLOY_HOST:?missing DEPLOY_HOST}" : "${DEPLOY_USER:?missing DEPLOY_USER}" : "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}" - -deploy_port="${DEPLOY_PORT:-22}" -deploy_path="${DEPLOY_PATH:-/srv/homelab}" -deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)" - -# The private key is written to a per-run directory that is removed on exit, so a -# failed or cancelled job cannot leave deploy credentials in the runner's temp -# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel. +: "${DEPLOY_KNOWN_HOSTS:?configure pinned DEPLOY_KNOWN_HOSTS}" +: "${DEPLOY_RUN_ID:?missing DEPLOY_RUN_ID}" +[[ "$DEPLOY_USER" =~ ^[A-Za-z_][A-Za-z0-9_.-]*$ ]] || exit 1 +[[ "$DEPLOY_HOST" =~ ^[A-Za-z0-9_.:-]+$ ]] || exit 1 +[[ "$DEPLOY_RUN_ID" =~ ^[0-9]+-[0-9]+$ ]] || exit 1 +[[ "${DEPLOY_PORT:-22}" =~ ^[0-9]+$ ]] || exit 1 key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")" -trap 'rm -rf "$key_dir"' EXIT INT TERM - -ssh_key="$key_dir/deploy_key" -printf '%s\n' "$DEPLOY_KEY" > "$ssh_key" -chmod 600 "$ssh_key" - -# A connection that died silently used to hang until the job timeout, and the -# stage was never re-run: one flaky TCP session cost a whole 45-minute apply. -# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds -# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that -# fails on its own merits exits with the remote's status, so a real failure -# still surfaces its own log instead of burning three attempts. The stages are -# declarative applies, so re-running one that had already committed is harmless. -ssh_opts=( - -i "$ssh_key" -p "$deploy_port" - -o BatchMode=yes -o StrictHostKeyChecking=accept-new - -o ConnectTimeout=15 - -o ServerAliveInterval=15 -o ServerAliveCountMax=4 -) - -rc=0 -# apply-k8s and apply-compose are separate workflow jobs so the graph stays -# intact for the verify job, but on a single node they must not run at once: -# host docker churn on top of cluster churn is what melts the node (load 40+, -# netbird/ssh die, helm is left pending-*). Serialize them on the workstation -# with a shared lock; whoever arrives second waits. -remote_cmd=(bash -se) -case "$1" in - apply-k8s | apply-compose) - remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se) +trap 'rm -rf "$key_dir"' EXIT +chmod 700 "$key_dir" +printf '%s\n' "$DEPLOY_KEY" >"$key_dir/key" +printf '%s\n' "$DEPLOY_KNOWN_HOSTS" >"$key_dir/known_hosts" +chmod 600 "$key_dir/key" "$key_dir/known_hosts" +ssh_opts=(-i "$key_dir/key" -p "${DEPLOY_PORT:-22}" -o BatchMode=yes -o StrictHostKeyChecking=yes + -o "UserKnownHostsFile=$key_dir/known_hosts" -o ConnectTimeout=15 + -o ServerAliveInterval=15 -o ServerAliveCountMax=4) +controller=.local/lib/homelab-deploy/controller.py +case "${1:?start, apply, verify or smoke required}" in + start) + python3 - <<'PY' >"$key_dir/request.json" +import json +import os +from pathlib import Path +release = json.loads(Path('release.json').read_text()) +print(json.dumps({'release': release, 'mode': os.environ.get('DEPLOY_MODE', 'changed'), + 'refresh_images': os.environ.get('REFRESH_IMAGES', 'false') == 'true'})) +PY + for attempt in 1 2 3; do + rc=0 + # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" start "$DEPLOY_RUN_ID" <"$key_dir/request.json" || rc=$? + [ "$rc" -eq 0 ] && exit 0 + [ "$rc" -eq 255 ] || exit "$rc" + sleep 5 + done + exit "$rc" ;; + apply|verify|smoke) + for attempt in 1 2 3; do + rc=0 + # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$? + [ "$rc" -eq 0 ] && exit 0 + [ "$rc" -eq 255 ] || exit "$rc" + echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" + sleep 5 + done + exit "$rc" + ;; + *) echo "Unknown SSH operation: $1" >&2; exit 1 ;; esac -for attempt in 1 2 3; do - if [ "$attempt" -gt 1 ]; then - echo ":: warning::ssh transport failed, retrying (${attempt}/3)" - sleep $((attempt * 5)) - fi - rc=0 - # shellcheck disable=SC2029 # remote_cmd/ssh_opts expand on the client on purpose: they select the local ssh invocation, only the heredoc runs remotely. - ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \ - env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \ - "DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \ - "STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$? -source "$REPO/.gitea/workflows/deploy-lib.sh" -run_stage "$STAGE" -EOF - [ "$rc" -eq 0 ] && break - [ "$rc" -ne 255 ] && break -done - -if [ "$rc" -ne 0 ]; then - echo ":: error::stage $1 failed over ssh (exit $rc)" -fi -exit "$rc" diff --git a/.gitea/workflows/tool-versions.env b/.gitea/workflows/tool-versions.env index cf8edb9..6d47ad5 100644 --- a/.gitea/workflows/tool-versions.env +++ b/.gitea/workflows/tool-versions.env @@ -34,3 +34,6 @@ NODE_VERSION="22.23.3" # Secret-reference regression tests parse rendered Kubernetes objects. JQ_VERSION="1.8.1" + +# BuildKit is the only auxiliary CI container; jobs themselves stay on the host. +BUILDKIT_IMAGE="moby/buildkit:v0.33.1" diff --git a/renovate/k8s/configmap.yaml b/renovate/k8s/configmap.yaml index 76baa8b..0b75b93 100644 --- a/renovate/k8s/configmap.yaml +++ b/renovate/k8s/configmap.yaml @@ -178,6 +178,13 @@ data: "datasourceTemplate": "helm", "depNameTemplate": "reloader", "registryUrlTemplate": "https://stakater.github.io/stakater-charts" + }, + { + "customType": "regex", + "description": "Pinned CI BuildKit helper image", + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], + "matchStrings": ["BUILDKIT_IMAGE=\"(?moby/buildkit):(?[^\"\\n]+)\""], + "datasourceTemplate": "docker" } ], "packageRules": [ diff --git a/renovate/renovate.json b/renovate/renovate.json index e9eae92..bd29619 100644 --- a/renovate/renovate.json +++ b/renovate/renovate.json @@ -167,6 +167,13 @@ "datasourceTemplate": "helm", "depNameTemplate": "reloader", "registryUrlTemplate": "https://stakater.github.io/stakater-charts" + }, + { + "customType": "regex", + "description": "Pinned CI BuildKit helper image", + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], + "matchStrings": ["BUILDKIT_IMAGE=\"(?moby/buildkit):(?[^\"\\n]+)\""], + "datasourceTemplate": "docker" } ], "packageRules": [ diff --git a/tests/test_cicd.py b/tests/test_cicd.py new file mode 100644 index 0000000..367c82f --- /dev/null +++ b/tests/test_cicd.py @@ -0,0 +1,267 @@ +"""CI gate, selection, persistent configuration and recovery regression tests.""" + +import importlib.util +import json +import os +import subprocess +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[1] + + +def module(name, filename): + spec = importlib.util.spec_from_file_location(name, ROOT / '.gitea/workflows' / filename) + loaded = importlib.util.module_from_spec(spec) + spec.loader.exec_module(loaded) + return loaded + + +release_module = module('release_test', 'release.py') +planner = module('plan_test', 'deploy-plan.py') +compose_module = module('compose_test', 'compose-release.py') +controller = module('controller_test', 'deploy-controller.py') + + +def release(sha='a' * 40): + return { + 'version': 1, + 'sha': sha, + 'images': {f'gcr.forust.xyz/forust/{name}': 'sha256:' + 'b' * 64 for name in release_module.IMAGES}, + 'inputs': {f'gcr.forust.xyz/forust/{name}': 'c' * 64 for name in release_module.IMAGES}, + } + + +class ReleaseGateTests(unittest.TestCase): + def test_release_rejects_wrong_sha_missing_images_and_mutable_tags(self): + for mutation in ('sha', 'missing', 'tag'): + data = release() + if mutation == 'sha': + data['sha'] = 'd' * 40 + elif mutation == 'missing': + data['images'].pop(next(iter(data['images']))) + else: + data['images'][next(iter(data['images']))] = 'prod' + with self.assertRaises(ValueError): + release_module.validate_release(data, 'a' * 40) + + def test_gate_excludes_pr_wrong_branch_and_failed_runs(self): + api = object.__new__(release_module.Gitea) + api.repository = 'forust/homelab' + good = { + 'id': 1, + 'status': 'completed', + 'conclusion': 'success', + 'head_branch': 'main', + 'event': 'push', + 'head_sha': 'a' * 40, + 'repository': {'full_name': api.repository}, + } + entries = [ + good, + {**good, 'event': 'pull_request'}, + {**good, 'head_branch': 'dev'}, + {**good, 'conclusion': 'failure'}, + {**good, 'head_sha': 'b' * 40}, + {**good, 'head_repository': {'full_name': 'attacker/fork'}}, + ] + with patch.object(api, 'pages', return_value=iter(entries)): + self.assertEqual(list(api.successful_runs('a' * 40)), [good]) + + def test_green_ci_with_skipped_build_is_rejected(self): + api = object.__new__(release_module.Gitea) + with ( + patch.object(api, 'pages', return_value=iter([{'name': 'build', 'conclusion': 'skipped'}])), + self.assertRaisesRegex(ValueError, 'build job'), + ): + api.release({'id': 1, 'head_sha': 'a' * 40}) + + def test_expired_or_ambiguous_artifacts_are_rejected(self): + api = object.__new__(release_module.Gitea) + api.base = 'https://example.test/api/v1/repos/a/b' + artifact = {'id': 1, 'name': 'release-' + 'a' * 40} + for artifacts in ([{**artifact, 'expired': True}], [artifact, artifact], []): + with ( + patch.object(api, 'pages', return_value=iter([{'name': 'build', 'conclusion': 'success'}])), + patch.object(api, 'request', return_value={'artifacts': artifacts}), + self.assertRaisesRegex(ValueError, 'artifact'), + ): + api.release({'id': 1, 'head_sha': 'a' * 40}) + + +class SelectionTests(unittest.TestCase): + def setUp(self): + self.scratch = tempfile.TemporaryDirectory() + self.addCleanup(self.scratch.cleanup) + self.repo = Path(self.scratch.name) + self.git('init', '-q') + self.git('config', 'user.email', 'test@example.test') + self.git('config', 'user.name', 'CI Test') + for service in ('one', 'two', 'postgres'): + directory = self.repo / service / 'k8s' + directory.mkdir(parents=True) + (directory / 'active').touch() + (directory / 'app.yaml').write_text('kind: Deployment\n') + (self.repo / '.gitea/workflows').mkdir(parents=True) + (self.repo / '.gitea/workflows/deploy-lib.sh').write_text('HELM_RELEASES=(\n)\n') + (self.repo / '.gitea/deploy-dependencies.json').write_text('{"postgres": ["one", "two"]}') + self.sha = self.commit() + self.initial = planner.make_plan(self.repo, self.repo, release(self.sha), None, 'full', []) + + def git(self, *args): + return planner.output('git', '-C', str(self.repo), *args) + + def commit(self): + self.git('add', '.') + self.git('commit', '-qm', 'Test state') + return self.git('rev-parse', 'HEAD') + + def test_first_changed_deploy_requires_explicit_full(self): + with self.assertRaisesRegex(ValueError, 'full'): + planner.make_plan(self.repo, self.repo, release(self.sha), None, 'changed', []) + + def test_only_changed_service_is_selected(self): + (self.repo / 'one/k8s/app.yaml').write_text('kind: StatefulSet\n') + result = planner.make_plan(self.repo, self.repo, release(self.commit()), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one']) + self.assertEqual(result['helm'], []) + + def test_failed_intermediate_deploy_does_not_lose_changes(self): + (self.repo / 'one/k8s/app.yaml').write_text('kind: StatefulSet\n') + self.commit() # This commit failed deploy: baseline must remain initial. + (self.repo / 'two/k8s/app.yaml').write_text('kind: StatefulSet\n') + result = planner.make_plan(self.repo, self.repo, release(self.commit()), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one', 'two']) + + def test_dependencies_and_removals_are_reported(self): + (self.repo / 'postgres/k8s/app.yaml').write_text('kind: StatefulSet\n') + (self.repo / 'two/k8s/active').unlink() + result = planner.make_plan(self.repo, self.repo, release(self.commit()), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one', 'postgres']) + self.assertIn('two', result['removed']) + + def test_local_configuration_change_selects_service(self): + (self.repo / 'one/.env').write_text('TEST_VALUE=changed\n') + result = planner.make_plan(self.repo, self.repo, release(self.sha), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one']) + + def test_redeploy_is_noop_and_full_includes_all(self): + result = planner.make_plan(self.repo, self.repo, release(self.sha), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], []) + result = planner.make_plan(self.repo, self.repo, release(self.sha), self.initial, 'full', []) + self.assertEqual(result['selected']['k8s'], ['one', 'postgres', 'two']) + + +class ComposeConfigurationTests(unittest.TestCase): + def test_pin_preserves_project_volumes_paths_and_previous_image(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + run = root / 'run' + source = run / 'source' + config_repo = root / 'persistent' + (source / 'headscale').mkdir(parents=True) + config_repo.mkdir() + (run / 'release.json').write_text(json.dumps(release())) + old = 'busybox@sha256:' + 'd' * 64 + new = 'busybox@sha256:' + 'e' * 64 + config = { + 'name': 'headscale', + 'services': { + 'app': { + 'image': 'busybox:latest', + 'volumes': [ + {'type': 'bind', 'source': str(config_repo / 'headscale/config.yaml'), 'target': '/config'}, + {'type': 'volume', 'source': 'data', 'target': '/data'}, + ], + } + }, + 'volumes': {'data': {'name': 'headscale_data'}}, + } + + def fake_output(*args, **kwargs): + if args[:2] == ('docker', 'compose'): + self.assertEqual(kwargs['cwd'], config_repo) + self.assertIn(str(config_repo / 'headscale'), args) + return json.dumps(config) + if args[:2] == ('docker', 'ps'): + return 'container' + if args[:2] == ('docker', 'inspect'): + return 'sha256:' + 'f' * 64 + return json.dumps([old]) + + with ( + patch.dict(os.environ, {'CONFIG_REPO': str(config_repo), 'REPO': str(source), 'RUN_DIR': str(run)}), + patch.object(compose_module, 'output', side_effect=fake_output), + patch.object(compose_module, 'resolve', return_value=new), + ): + compose_module.prepare(source / 'headscale/compose.yaml') + pinned = json.loads((run / 'compose/headscale.json').read_text()) + before = json.loads((run / 'compose-before/headscale.json').read_text()) + self.assertEqual(pinned['name'], 'headscale') + self.assertEqual(pinned['volumes'], config['volumes']) + self.assertEqual(pinned['services']['app']['volumes'], config['services']['app']['volumes']) + self.assertEqual(pinned['services']['app']['image'], new) + self.assertEqual(before['services']['app']['image'], old) + self.assertEqual((run / 'compose/headscale.json').stat().st_mode & 0o777, 0o600) + + def test_registry_index_and_single_image_descriptors(self): + for digest in ('a' * 64, 'b' * 64): + with patch.object(compose_module, 'output', return_value=json.dumps({'digest': 'sha256:' + digest})): + self.assertEqual( + compose_module.resolve('registry.test:5000/repo:latest'), f'registry.test:5000/repo@sha256:{digest}' + ) + + +class ControllerTests(unittest.TestCase): + def test_completed_stage_cannot_apply_again(self): + with tempfile.TemporaryDirectory() as scratch: + directory = Path(scratch) + controller.atomic_json( + directory / 'status.json', {'state': 'success', 'stages': {'apply-k8s': {'result': 'success'}}} + ) + with patch.object(subprocess, 'run') as execute: + self.assertTrue(controller.stage(directory, 'apply-k8s', 60)) + execute.assert_not_called() + + def test_run_id_is_not_shell_or_path_input(self): + for invalid in ('../123', '-1', '1;touch bad', 'abc', '1/2'): + with self.assertRaises(ValueError): + controller.run_directory(invalid) + + def test_exact_previous_revision_is_used_for_rollback(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + (root / 'current').write_text(str(root)) + (root / 'revisions.json').write_text( + json.dumps([{'kind': 'deployment', 'namespace': 'app', 'name': 'web', 'uid': 'same', 'revision': 7}]) + ) + (root / 'failed').write_text('deployment app web\n') + script = """set -euo pipefail +source "$LIB" +kubectl() { + case "$*" in + *metadata.annotations*) printf '{}' ;; + *metadata.uid*) printf same ;; + 'rollout undo'*) printf '%s\\n' "$*" >>"$CALLS" ;; + 'rollout status'*) return 0 ;; + *) return 1 ;; + esac +} +rollback_workloads "$FAILED" +""" + env = { + **os.environ, + 'REPO': str(ROOT), + 'LIB': str(ROOT / '.gitea/workflows/deploy-lib.sh'), + 'DEPLOY_SNAPSHOT_DIR': str(root), + 'CALLS': str(root / 'calls'), + 'FAILED': str(root / 'failed'), + } + subprocess.run(['/usr/bin/bash', '-c', script], env=env, check=True) # noqa: S603 + self.assertIn('--to-revision=7', (root / 'calls').read_text()) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_cicd_lifecycle.py b/tests/test_cicd_lifecycle.py new file mode 100644 index 0000000..f14caab --- /dev/null +++ b/tests/test_cicd_lifecycle.py @@ -0,0 +1,237 @@ +"""Release publication and controller recovery tests without a live server.""" + +import io +import json +import os +import subprocess +import tempfile +import unittest +import zipfile +from pathlib import Path +from unittest.mock import Mock, patch + +from test_cicd import ROOT, controller, release, release_module + + +class ArtifactTests(unittest.TestCase): + def test_archive_rejects_nested_or_extra_files(self): + api = object.__new__(release_module.Gitea) + api.base = 'https://example.test/api/v1/repos/a/b' + for names in (['../release.json'], ['release.json', 'credentials']): + blob = io.BytesIO() + with zipfile.ZipFile(blob, 'w') as archive: + for name in names: + archive.writestr(name, json.dumps(release())) + replies = [{'artifacts': [{'id': 1, 'name': 'release-' + 'a' * 40}]}, blob.getvalue()] + with ( + patch.object(api, 'pages', return_value=iter([{'name': 'build', 'conclusion': 'success'}])), + patch.object(api, 'request', side_effect=replies), + self.assertRaisesRegex(ValueError, 'archive'), + ): + api.release({'id': 1, 'head_sha': 'a' * 40}) + + def test_quoted_and_single_platform_images_render_from_checked_release(self): + with tempfile.TemporaryDirectory() as scratch: + path = Path(scratch) / 'release.json' + path.write_text(json.dumps(release())) + result = io.StringIO() + image = 'gcr.forust.xyz/forust/error-pages' + with patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}): + release_module.render(io.StringIO(f' image: "{image}:prod" # note\n'), result) + self.assertEqual(result.getvalue(), f' image: "{image}@sha256:{"b" * 64}" # note\n') + + def test_edu_release_digest_is_preserved(self): + with tempfile.TemporaryDirectory() as scratch: + path = Path(scratch) / 'release.json' + path.write_text(json.dumps(release())) + image = 'gcr.forust.xyz/forust/session-keeper' + line = f'image: {image}@sha256:{"e" * 64}\n' + result = io.StringIO() + with patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}): + release_module.render(io.StringIO(line), result) + self.assertEqual(result.getvalue(), line) + with self.assertRaises(ValueError): + release_module.render(io.StringIO(f'image: {image}:prod\n'), io.StringIO()) + + def test_unknown_image_cannot_emit_partial_manifest(self): + with tempfile.TemporaryDirectory() as scratch: + path = Path(scratch) / 'release.json' + path.write_text(json.dumps(release())) + result = io.StringIO() + with ( + patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}), + self.assertRaisesRegex(ValueError, 'missing'), + ): + release_module.render( + io.StringIO('kind: Deployment\nimage: gcr.forust.xyz/forust/unknown:prod\n'), result + ) + self.assertEqual(result.getvalue(), '') + + def test_only_changed_image_is_built_and_credentials_are_removed(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + built = [] + auth_directories = [] + old = release() + + def fake_command(*args, **kwargs): + if args[:2] == ('git', 'rev-parse'): + return 'e' * 40 + if args[:3] == ('docker', 'buildx', 'build'): + built.append(args[args.index('--file') + 1]) + metadata = Path(args[args.index('--metadata-file') + 1]) + metadata.write_text(json.dumps({'containerimage.digest': 'sha256:' + 'f' * 64})) + auth_directories.append(Path(kwargs['env']['DOCKER_CONFIG'])) + return '' + + api = Mock() + api.successful_runs.return_value = iter([{'id': 1}]) + api.release.return_value = old + with ( + patch.dict( + os.environ, + { + 'GITHUB_SHA': 'e' * 40, + 'GITHUB_RUN_ID': '2', + 'REGISTRY_USERNAME': 'test', + 'REGISTRY_PASSWORD': 'placeholder', + }, + ), + patch.object(release_module.Path, 'home', return_value=root), + patch.object(release_module, 'Gitea', return_value=api), + patch.object( + release_module, + 'fingerprint', + side_effect=lambda context, _file: ('d' if context == 'errorpages' else 'c') * 64, + ), + patch.object(release_module, 'command', side_effect=fake_command), + patch.object(subprocess, 'run', return_value=subprocess.CompletedProcess([], 0)), + ): + release_module.build(root / 'release.json') + self.assertEqual(built, ['errorpages/Dockerfile']) + self.assertTrue(all(not directory.exists() for directory in auth_directories)) + self.assertEqual(json.loads((root / 'release.json').read_text())['sha'], 'e' * 40) + + +class DurableRunTests(unittest.TestCase): + def test_duplicate_start_only_reattaches(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + directory.mkdir(parents=True) + request = {'release': release(), 'mode': 'full', 'refresh_images': False} + controller.atomic_json(directory / 'request.json', request) + controller.atomic_json(directory / 'status.json', {'state': 'running', 'stages': {}}) + with ( + patch.object(controller, 'STATE', state), + patch.object(controller.sys, 'stdin', io.TextIOWrapper(io.BytesIO(json.dumps(request).encode()))), + patch.object(controller, 'command') as execute, + ): + controller.start('123-1') + execute.assert_not_called() + + def test_failed_apply_still_verifies_and_does_not_advance_baseline(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + directory.mkdir(parents=True) + controller.atomic_json( + directory / 'request.json', {'release': release(), 'mode': 'full', 'refresh_images': False} + ) + controller.atomic_json(directory / 'status.json', {'state': 'queued', 'stages': {}}) + called = [] + + def fake_stage(folder, name, _budget): + called.append(name) + status = json.loads((folder / 'status.json').read_text()) + status['stages'][name] = {'result': 'failure' if name == 'apply-k8s' else 'success'} + controller.atomic_json(folder / 'status.json', status) + return name != 'apply-k8s' + + with ( + patch.object(controller, 'STATE', state), + patch.object(controller, 'make_plan', return_value={'selected': {}, 'helm': [], 'removed': []}), + patch.object(controller, 'stage', side_effect=fake_stage), + patch.object(controller, 'command', return_value='1'), + self.assertRaises(RuntimeError), + ): + controller.execute('123-1') + self.assertIn('verify-k8s', called) + self.assertIn('smoke', called) + self.assertNotIn('apply-compose', called) + self.assertFalse((state / 'last-success.json').exists()) + self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure') + + def test_recovery_finishes_baseline_after_all_stages_completed(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + directory.mkdir(parents=True) + names = ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke') + controller.atomic_json( + directory / 'status.json', + {'state': 'running', 'stages': {name: {'result': 'success'} for name in names}}, + ) + controller.atomic_json(directory / 'plan.json', {'sha': 'a' * 40}) + with patch.object(controller, 'STATE', state), patch.object(controller, 'stage') as execute: + controller.recover(directory) + execute.assert_not_called() + self.assertEqual(json.loads((state / 'last-success.json').read_text())['run_id'], '123-1') + self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'success') + + def test_manual_recovery_retries_checks_without_repeating_apply(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + (directory / 'snapshot').mkdir(parents=True) + (directory / 'snapshot/current').write_text('snapshot') + controller.atomic_json( + directory / 'status.json', + { + 'state': 'failure', + 'stages': { + 'apply-k8s': {'result': 'failure'}, + 'verify-k8s': {'result': 'failure'}, + 'smoke': {'result': 'failure'}, + }, + }, + ) + called = [] + + def checks(folder, name, _budget): + status = json.loads((folder / 'status.json').read_text()) + self.assertNotIn(name, status['stages']) + called.append(name) + status['stages'][name] = {'result': 'success'} + controller.atomic_json(folder / 'status.json', status) + return True + + with patch.object(controller, 'STATE', state), patch.object(controller, 'stage', side_effect=checks): + controller.recover(directory, retry=True) + self.assertEqual(called, ['verify-k8s', 'smoke']) + self.assertFalse((state / 'last-success.json').exists()) + self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure') + + +class InstallerTests(unittest.TestCase): + def test_version_comparison_is_exact_without_network_or_host_packages(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + binary = root / 'bin/fake' + binary.parent.mkdir() + binary.write_text('#!/bin/sh\necho fake-v1.7.70\n') + binary.chmod(0o755) + script = """set -euo pipefail +source "$LIB" +if at_version fake 1.7.7; then exit 1; fi +at_version fake 1.7.70 +""" + subprocess.run( # noqa: S603 + ['/usr/bin/bash', '-c', script], + check=True, + env={**os.environ, 'TOOLS_DIR': str(root), 'LIB': str(ROOT / '.gitea/workflows/install-ci-tools.sh')}, + ) + + +if __name__ == '__main__': + unittest.main()