diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md index 91b9046..bd203a2 100644 --- a/.gitea/runner/README.md +++ b/.gitea/runner/README.md @@ -5,7 +5,12 @@ Compose, workflow, shell, Python, formatting, YAML, Dockerfile and Kubernetes checks appear as separate jobs. Jobs run on `homelab:host`, one at a time; the build waits for every check to pass. CI and deploy runs also show a summary with the release SHA, image build or reuse results, deploy mode, selected services, -and image digests. No job images or Kubernetes credentials +and image digests. Failed runs keep a summary of completed image builds, stage +results, apply results, and recorded Kubernetes recovery. The final deploy +summary is in the smoke job; earlier jobs show the state observed at that time. +Apply success is separate from health and recovery. Update the installed +workstation controller with `setup-workstation.sh` when no deploy is running. +No job images or Kubernetes credentials are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows. ## Runner installation diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 101778e..84f63d3 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -19,6 +19,7 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Validate Compose files shell: bash run: | @@ -47,6 +48,22 @@ jobs: exit 1 fi echo "checked ${#files[@]} Compose file(s)" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Compose + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.source.conclusion == 'failure' + && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi workflows: name: Workflows runs-on: homelab @@ -54,17 +71,35 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint shellcheck)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint Gitea Actions workflows with actionlint shell: bash run: | set -euo pipefail actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Workflows + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi shell: name: Shell runs-on: homelab @@ -72,12 +107,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint shell scripts with ShellCheck shell: bash run: | @@ -91,6 +128,22 @@ jobs: fi shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" bash .gitea/tests/deploy-validation.sh + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Shell + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi formatting: name: Formatting runs-on: homelab @@ -98,12 +151,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Check formatting with Prettier shell: bash run: | @@ -121,6 +176,22 @@ jobs: fi prettier --check --ignore-unknown "${prettier_files[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Formatting + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi python: name: Python and tests runs-on: homelab @@ -128,12 +199,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff jq)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint and format-check Python with Ruff shell: bash run: | @@ -141,6 +214,22 @@ jobs: ruff check . .gitea/workflows ruff format --check . .gitea/workflows python3 -m unittest discover -s tests -v + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Python and tests + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi yaml: name: YAML runs-on: homelab @@ -148,12 +237,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint YAML syntax shell: bash run: | @@ -171,6 +262,22 @@ jobs: fi yamllint -c .yamllint "${yaml_files[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: YAML + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi dockerfiles: name: Dockerfiles runs-on: homelab @@ -178,12 +285,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint Dockerfiles shell: bash run: | @@ -199,6 +308,22 @@ jobs: fi hadolint -c .hadolint.yaml "${dockerfiles[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Dockerfiles + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi kubernetes: name: Kubernetes runs-on: homelab @@ -206,12 +331,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Validate Kubernetes manifests against JSON schemas shell: bash run: | @@ -232,6 +359,22 @@ jobs: -ignore-missing-schemas \ -summary \ "${manifests[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Kubernetes + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi build: needs: - compose @@ -250,12 +393,14 @@ jobs: uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 with: fetch-depth: 0 + id: source - name: Build changed images and write release env: GITEA_TOKEN: ${{ github.token }} REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }} REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }} run: python3 .gitea/workflows/release.py build + id: check - name: Store commit release uses: actions/upload-artifact@c6a366c94c3e0affe28c06c8df20a878f24da3cf with: @@ -263,3 +408,19 @@ jobs: path: release.json if-no-files-found: error retention-days: 30 + id: artifact + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Image build and release artifact + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.artifact.conclusion == 'failure' + && 'Release artifact upload' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi diff --git a/.gitea/workflows/deploy-controller.py b/.gitea/workflows/deploy-controller.py index 4bf80f5..be51277 100644 --- a/.gitea/workflows/deploy-controller.py +++ b/.gitea/workflows/deploy-controller.py @@ -212,14 +212,17 @@ def execute(run_id): raise ValueError(f'Interrupted deploy {other.name}; run recover first') status['state'] = 'running' atomic_json(directory / 'status.json', status) + phase = 'plan' try: plan = make_plan(directory) print( json.dumps({'selected': plan['selected'], 'helm': plan['helm'], 'manual_removals': plan['removed']}), flush=True, ) + phase = 'doctor' if not stage(directory, 'doctor', 600): raise RuntimeError('Preflight failed') + phase = 'validate' if not stage(directory, 'validate', 1200): raise RuntimeError('Validation failed') if json.loads((directory / 'request.json').read_text())['mode'] == 'plan': @@ -228,6 +231,7 @@ def execute(run_id): atomic_json(directory / 'status.json', status) return # Budget includes both rollout checks and rollback waves, plus API overhead. + phase = 'Recovery budget' count = int( command( 'bash', @@ -239,14 +243,24 @@ def execute(run_id): verify_budget = max(600, 2 * math.ceil(count / 4) * 300 + 120) if verify_budget > 7200: raise ValueError('More than two hours of recovery required; split this deploy') + phase = 'apply-k8s' k8s_ok = stage(directory, 'apply-k8s', 2700) + phase = 'apply-compose' compose_ok = stage(directory, 'apply-compose', 1800) if k8s_ok else False + phase = 'verify-k8s' verify_ok = stage(directory, 'verify-k8s', verify_budget) + phase = 'smoke' smoke_ok = stage(directory, 'smoke', 600) if not all((k8s_ok, compose_ok, verify_ok, smoke_ok)): raise RuntimeError('Deploy failed; inspect stage logs and recovery report') + phase = 'Save the successful baseline' finish_success(directory, plan) except Exception as error: + status = json.loads((directory / 'status.json').read_text()) + status['failure_stage'] = next( + (name for name, result in status['stages'].items() if result.get('result') == 'failure'), phase + ) + atomic_json(directory / 'status.json', status) with (directory / 'controller.log').open('a') as stream: stream.write(f'{error}\n') recover(directory) @@ -305,6 +319,52 @@ def summary(run_id): f'- Mode: `{request["mode"]}`', f'- Refresh third-party images: `{request["refresh_images"]}`', ] + status = json.loads((directory / 'status.json').read_text()) + if status.get('failure_stage'): + lines.append(f'- Failed stage: **{status["failure_stage"]}**') + lines.extend( + [ + '', + f'- Observed run state: **{status["state"]}**', + '', + '### Stage results', + '| Stage | Result | Exit code |', + '| --- | --- | --- |', + ] + ) + for name in ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke'): + stage_result = status['stages'].get(name, {}) + lines.append(f'| {name} | {stage_result.get("result", "not started")} | {stage_result.get("exit_code", "—")} |') + lines.extend(['', '### Apply and Helm recovery results']) + events_file = directory / 'apply-events.jsonl' + events = [] + if events_file.exists(): + for line in events_file.read_text().splitlines(): + try: + events.append(json.loads(line)) + except json.JSONDecodeError: + lines.append('- An operation record is incomplete. Check the stage log.') + latest = {(event['action'], event['target']): event['result'] for event in events} + lines.extend(f'- `{action}` `{target}`: **{result}**' for (action, target), result in latest.items()) + if not latest: + lines.append('- No apply results were recorded.') + lines.append('- A completed apply does not confirm health. See verification and smoke results.') + lines.extend(['', '### Kubernetes recovery']) + pointer = directory / 'snapshot/current' + failed = Path(pointer.read_text().strip()) / 'failed-workloads' if pointer.exists() else None + if failed and failed.exists(): + contents = failed.read_text() + counts = dict(re.findall(r'^(ROLLED_BACK|UNRECOVERED)=([0-9]+)$', contents, re.MULTILINE)) + if not contents.strip(): + lines.append('- No failed workloads were recorded. See the verification result above.') + elif counts: + lines.append(f'- Workloads restored: **{counts.get("ROLLED_BACK", "unknown")}**') + lines.append(f'- Workloads that need manual recovery: **{counts.get("UNRECOVERED", "unknown")}**') + else: + lines.append('- Rollback has no recorded result yet. Check the verification log.') + else: + lines.append('- No workload rollback was recorded. This does not confirm health.') + lines.append('- Compose requires manual recovery. Use the saved command in the apply log.') if not plan_file.exists(): lines.extend(['', 'Plan was not created. Check the controller log.']) print('\n'.join(lines)) diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 8731b90..58f04e7 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -27,6 +27,15 @@ log() { echo "== $* ==" } +# Store operation results without command output or local configuration values. +record_apply() { + [ -n "${RUN_DIR:-}" ] || return 0 + jq -cn --arg action "$1" --arg target "$2" --arg result "$3" \ + '{action: $action, target: $target, result: $result}' >>"$RUN_DIR/apply-events.jsonl" \ + || echo 'WARNING: cannot record an apply result' >&2 + return 0 +} + warn() { echo "WARNING: $*" >&2 } @@ -376,15 +385,19 @@ recover_pending_release() { echo "ERROR: no captured Helm revision for $release; manual recovery required" return 1 fi + record_apply helm-rollback "$namespace/$release" started if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then + record_apply helm-rollback "$namespace/$release" failure echo "WARN: helm rollback of $release did not complete" return 1 fi status="$(helm_release_status "$release" "$namespace")" || return 1 if [ "$status" != "deployed" ]; then + record_apply helm-rollback "$namespace/$release" failure echo "WARN: $release is $status after rollback" return 1 fi + record_apply helm-rollback "$namespace/$release" success ;; esac return 0 @@ -448,11 +461,13 @@ upgrade_helm_releases() { # --rollback-on-failure (+ --wait) rolls the release back when the upgrade # times out or the workloads it touches never become ready, so a bad chart # bump is not left half applied. (--atomic was this combo; deprecated.) + record_apply helm-upgrade "$namespace/$release" started if ! helm upgrade --install "$release" "$chart" \ --namespace "$namespace" \ --version "$version" \ --values "$values" \ --wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then + record_apply helm-upgrade "$namespace/$release" failure echo "WARN: upgrade of $release failed, checking release state" # --rollback-on-failure already attempted its own rollback; finish the job when that # rollback never completed, otherwise the release stays pending-* and @@ -462,8 +477,10 @@ upgrade_helm_releases() { else echo "ERROR: upgrade of $release failed (release is back on its previous revision)." fi + record_apply helm-recovery-state "$namespace/$release" "$(helm_release_status "$release" "$namespace" || echo unknown)" return 1 fi + record_apply helm-upgrade "$namespace/$release" success done } @@ -631,7 +648,12 @@ stage_apply_k8s() { if [ "${#ns_files[@]}" -gt 0 ]; then log "Applying namespaces (${#ns_files[@]} files)" for m in "${ns_files[@]}"; do - kubectl apply -f "$m" + record_apply kubectl "${m#"$REPO"/}" started + if ! kubectl apply -f "$m"; then + record_apply kubectl "${m#"$REPO"/}" failure + return 1 + fi + record_apply kubectl "${m#"$REPO"/}" success done fi if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then @@ -646,18 +668,24 @@ stage_apply_k8s() { log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do log "Applying ${m#"$REPO"/}" + record_apply kubectl "${m#"$REPO"/}" started if ! render_pinned <"$m" | kubectl apply -f -; then + record_apply kubectl "${m#"$REPO"/}" failure echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 exit 1 fi + record_apply kubectl "${m#"$REPO"/}" success done fi for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)" + record_apply kustomize "${k#"$REPO"/}" started if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then + record_apply kustomize "${k#"$REPO"/}" failure echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2 exit 1 fi + record_apply kustomize "${k#"$REPO"/}" success done # No verification here on purpose. This stage may be killed at any point by @@ -959,7 +987,12 @@ stage_apply_compose() { local cf for cf in "${COMPOSE_STACKS[@]}"; do log "Applying Compose ${cf#"$REPO"/}" - compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans + record_apply compose "${cf#"$REPO"/}" started + if ! compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans; then + record_apply compose "${cf#"$REPO"/}" failure + return 1 + fi + record_apply compose "${cf#"$REPO"/}" success verify_compose_stack "$cf" done echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)" diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 2ac1f60..900280c 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -64,6 +64,18 @@ jobs: run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA" - name: Submit durable deploy to workstation run: bash .gitea/workflows/ssh-run.sh start + - name: Write the request result + if: always() + env: + REQUEST_RESULT: ${{ job.status }} + CHECKED_SHA: ${{ steps.release.outputs.sha }} + run: | + if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## Deploy request\n\n- Result: **%s**\n- Checked commit: `%s`\n- Mode: `%s`\n' "$REQUEST_RESULT" "${CHECKED_SHA:-not checked}" "$DEPLOY_MODE" >>"$GITHUB_STEP_SUMMARY" + if [ "$REQUEST_RESULT" != success ]; then + echo 'Open the failed step log. If SSH submission failed, check the remote controller state.' >>"$GITHUB_STEP_SUMMARY" + fi + fi apply: needs: [gate] @@ -76,6 +88,14 @@ jobs: ref: ${{ needs.gate.outputs.sha }} - name: Follow validation and sequential Kubernetes / Compose apply run: bash .gitea/workflows/ssh-run.sh apply + - name: Write the deploy result + if: always() + run: | + if [ -f .gitea/workflows/ssh-run.sh ]; then + bash .gitea/workflows/ssh-run.sh summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY" + fi verify: needs: [gate, apply] @@ -89,6 +109,14 @@ jobs: ref: ${{ needs.gate.outputs.sha }} - name: Follow workload verification and recovery run: bash .gitea/workflows/ssh-run.sh verify + - name: Write the deploy result + if: always() + run: | + if [ -f .gitea/workflows/ssh-run.sh ]; then + bash .gitea/workflows/ssh-run.sh summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY" + fi smoke: needs: [gate, verify] @@ -102,3 +130,11 @@ jobs: ref: ${{ needs.gate.outputs.sha }} - name: Follow public route checks run: bash .gitea/workflows/ssh-run.sh smoke + - name: Write the deploy result + if: always() + run: | + if [ -f .gitea/workflows/ssh-run.sh ]; then + bash .gitea/workflows/ssh-run.sh summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY" + fi diff --git a/.gitea/workflows/release.py b/.gitea/workflows/release.py index cc9e9bd..00d2a79 100644 --- a/.gitea/workflows/release.py +++ b/.gitea/workflows/release.py @@ -163,10 +163,11 @@ def gate(output, requested_ref, event_sha): print(f'CI gate accepted {sha}') -def build(output): +def build_images(output, report): sha = command('git', 'rev-parse', 'HEAD') if sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha): raise ValueError('Build checkout does not match GITHUB_SHA') + report['phase'] = 'Find a successful CI release' api = Gitea() previous = None for run in sorted(itertools.islice(api.successful_runs(), 50), key=lambda item: item['id'], reverse=True): @@ -183,6 +184,7 @@ def build(output): builder_config.mkdir(parents=True, exist_ok=True) env = {**os.environ, 'DOCKER_CONFIG': docker_config, 'BUILDX_CONFIG': str(builder_config)} try: + report['phase'] = 'Registry login' subprocess.run( # noqa: S603, S607 [ shutil.which('docker') or '/usr/bin/docker', @@ -197,6 +199,7 @@ def build(output): check=True, env=env, ) + report['phase'] = 'Prepare the builder' builder = 'homelab-ci' versions = dict( re.findall(r'^([A-Z_]+)="([^"\n]+)"$', Path('.gitea/workflows/tool-versions.env').read_text(), re.MULTILINE) @@ -231,9 +234,10 @@ def build(output): ) signature.write_text(image + '\n') release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}} - built = [] - reused = [] + report['images'] = release['images'] for name, (context, dockerfile) in IMAGES.items(): + report['phase'] = f'Build or reuse {name}' + report['current'] = name image = f'gcr.forust.xyz/forust/{name}' inputs = fingerprint(context, dockerfile) old_digest = (previous or {}).get('images', {}).get(image) @@ -257,10 +261,9 @@ def build(output): if exists: print(f'Reuse {name}: inputs unchanged') digest = old_digest - reused.append((name, image, digest)) + report['reused'].append(name) else: print(f'Build {name}', flush=True) - built.append((name, image)) metadata = Path(docker_config) / 'metadata.json' command( 'docker', @@ -286,27 +289,13 @@ def build(output): env=env, ) digest = json.loads(metadata.read_text())['containerimage.digest'] + report['built'].append(name) release['images'][image] = digest release['inputs'][image] = inputs validate_release(release, sha) output.write_text(json.dumps(release, indent=2) + '\n') - summary = os.environ.get('GITHUB_STEP_SUMMARY') - if summary: - lines = [f'## Image release for `{sha}`', '', '### Built'] - lines.extend(f'- `{name}` — `{image}`' for name, image in built) - if not built: - lines.append('- None') - lines.extend(['', '### Reused from successful CI']) - lines.extend(f'- `{name}` — `{image}@{digest}`' for name, image, digest in reused) - if not reused: - lines.append('- None') - lines.extend(['', '### Release digests']) - lines.extend( - f'- `{name}` — `{image}@{release["images"][image]}`' - for name in IMAGES - for image in [f'gcr.forust.xyz/forust/{name}'] - ) - Path(summary).write_text('\n'.join(lines) + '\n') + report['current'] = None + report['phase'] = 'Release file saved' finally: # Cleanup errors must neither leak credentials nor mask the original build error. try: @@ -330,6 +319,59 @@ def build(output): shutil.rmtree(docker_config) +def write_summary(lines): + path = os.environ.get('GITHUB_STEP_SUMMARY') + if path: + try: + with Path(path).open('a') as stream: + stream.write('\n'.join(lines) + '\n\n') + except OSError: + print('WARNING: cannot write the job summary') + + +def check_summary(): + lines = [ + f'## {os.environ["SUMMARY_CHECK"]}', + '', + f'- Commit: `{os.environ.get("GITHUB_SHA", "unknown")}`', + f'- Result: **{os.environ["SUMMARY_RESULT"]}**', + ] + if os.environ.get('SUMMARY_FAILED_STEP'): + lines.append(f'- Failed step: {os.environ["SUMMARY_FAILED_STEP"]}') + if os.environ['SUMMARY_RESULT'] != 'success': + lines.append('- Open the failed step log for the error details.') + write_summary(lines) + + +def build(output): + report = {'phase': 'Check the source commit', 'current': None, 'built': [], 'reused': [], 'images': {}} + result = 'failure' + try: + build_images(output, report) + result = 'success' + finally: + lines = [ + f'## Image release `{os.environ.get("GITHUB_SHA", "unknown")}`', + '', + f'- Result: **{result}**', + f'- Last stage: {report["phase"]}', + ] + if result == 'failure': + lines.append('- No release from this build can be deployed. Open the failed step log.') + if report['current']: + lines.append(f'- Image at the failure: `{report["current"]}`') + for title, key in (('Built', 'built'), ('Reused from successful CI', 'reused')): + lines.extend(['', f'### {title}']) + lines.extend(f'- `{name}`' for name in report[key]) + if not report[key]: + lines.append('- None') + lines.extend(['', '### Completed image digests']) + lines.extend(f'- `{image}@{digest}`' for image, digest in report['images'].items()) + if not report['images']: + lines.append('- None') + write_summary(lines) + + def render(stream, destination): release = validate_release(json.loads(Path(os.environ['RELEASE_FILE']).read_text()), os.environ['DEPLOY_SHA']) image_line = re.compile( @@ -354,12 +396,14 @@ def render(stream, destination): def main(): parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument('action', choices=('build', 'gate', 'render')) + parser.add_argument('action', choices=('build', 'gate', 'render', 'check-summary')) parser.add_argument('--output', type=Path, default=Path('release.json')) parser.add_argument('--ref', default='main') parser.add_argument('--event-sha', default='') args = parser.parse_args() - if args.action == 'render': + if args.action == 'check-summary': + check_summary() + elif args.action == 'render': render(sys.stdin, sys.stdout) elif args.action == 'gate': gate(args.output, args.ref, args.event_sha) diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index 8caecfa..7f29be4 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -20,7 +20,7 @@ ssh_opts=(-i "$key_dir/key" -p "${DEPLOY_PORT:-22}" -o BatchMode=yes -o StrictHo -o "UserKnownHostsFile=$key_dir/known_hosts" -o ConnectTimeout=15 -o ServerAliveInterval=15 -o ServerAliveCountMax=4) controller=.local/lib/homelab-deploy/controller.py -case "${1:?start, apply, verify or smoke required}" in +case "${1:?start, apply, verify, smoke or summary required}" in start) python3 - <<'PY' >"$key_dir/request.json" import json @@ -40,41 +40,31 @@ PY done exit "$rc" ;; - apply) + apply|verify|smoke) result=0 for attempt in 1 2 3; do rc=0 # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. - ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" apply || rc=$? + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$? [ "$rc" -eq 0 ] && break [ "$rc" -eq 255 ] || { result="$rc"; break; } echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" if [ "$attempt" -eq 3 ]; then result=255; break; fi sleep 5 done + exit "$result" + ;; + summary) if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then rc=0 # shellcheck disable=SC2029 # The run ID is validated above. ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" summary "$DEPLOY_RUN_ID" >"$key_dir/deploy-summary.md" || rc=$? if [ "$rc" -eq 0 ]; then - cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" + cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" || echo "WARNING: cannot write the deploy summary" else - echo 'Deploy summary is unavailable. Check the controller log.' >>"$GITHUB_STEP_SUMMARY" + echo 'Deploy summary is unavailable. The SSH connection failed or the controller did not respond. Check the job log.' >>"$GITHUB_STEP_SUMMARY" || true fi fi - exit "$result" - ;; - verify|smoke) - for attempt in 1 2 3; do - rc=0 - # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. - ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$? - [ "$rc" -eq 0 ] && exit 0 - [ "$rc" -eq 255 ] || exit "$rc" - echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" - sleep 5 - done - exit "$rc" ;; *) echo "Unknown SSH operation: $1" >&2; exit 1 ;; esac diff --git a/tests/test_cicd_lifecycle.py b/tests/test_cicd_lifecycle.py index f14caab..8f332c2 100644 --- a/tests/test_cicd_lifecycle.py +++ b/tests/test_cicd_lifecycle.py @@ -213,6 +213,64 @@ class DurableRunTests(unittest.TestCase): self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure') +class FailureSummaryTests(unittest.TestCase): + def test_build_failure_keeps_progress_and_does_not_expose_exception_text(self): + with tempfile.TemporaryDirectory() as scratch: + summary = Path(scratch) / 'summary.md' + + def failed_build(_output, report): + report.update(phase='Build or reuse xdfnx-homepage', current='xdfnx-homepage', built=['error-pages']) + report['images']['gcr.forust.xyz/forust/error-pages'] = 'sha256:' + 'b' * 64 + raise RuntimeError('private value must not appear in the summary') + + with ( + patch.dict(os.environ, {'GITHUB_STEP_SUMMARY': str(summary), 'GITHUB_SHA': 'a' * 40}), + patch.object(release_module, 'build_images', side_effect=failed_build), + self.assertRaises(RuntimeError), + ): + release_module.build(Path(scratch) / 'release.json') + content = summary.read_text() + self.assertIn('**failure**', content) + self.assertIn('error-pages', content) + self.assertIn('xdfnx-homepage', content) + self.assertNotIn('private value', content) + + def test_deploy_failure_reports_completed_apply_and_rollback_result(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + snapshot = directory / 'snapshot/before' + snapshot.mkdir(parents=True) + (directory / 'snapshot/current').write_text(str(snapshot)) + (snapshot / 'failed-workloads').write_text('deployment app api\nROLLED_BACK=1\nUNRECOVERED=0\n') + controller.atomic_json( + directory / 'request.json', {'release': release(), 'mode': 'changed', 'refresh_images': False} + ) + controller.atomic_json( + directory / 'status.json', + { + 'state': 'failure', + 'stages': { + 'apply-k8s': {'result': 'success', 'exit_code': 0}, + 'verify-k8s': {'result': 'failure', 'exit_code': 1}, + }, + }, + ) + controller.atomic_json(directory / 'plan.json', {'selected': {'k8s': ['app'], 'compose': []}}) + (directory / 'apply-events.jsonl').write_text( + json.dumps({'action': 'kubectl', 'target': 'app/k8s/api.yaml', 'result': 'success'}) + '\n' + ) + output = io.StringIO() + with patch.object(controller, 'STATE', state), patch('sys.stdout', output): + controller.summary('123-1') + content = output.getvalue() + self.assertIn('verify-k8s | failure | 1', content) + self.assertIn('app/k8s/api.yaml', content) + self.assertIn('Workloads restored: **1**', content) + self.assertIn('manual recovery: **0**', content) + self.assertIn('Compose requires manual recovery', content) + + class InstallerTests(unittest.TestCase): def test_version_comparison_is_exact_without_network_or_host_packages(self): with tempfile.TemporaryDirectory() as scratch: