diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 97f57ce..f32cbe3 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -361,7 +361,7 @@ jobs: case "$service" in dtek_notif) image="${REGISTRY}/forust/dtek-notif" - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -384,7 +384,7 @@ jobs: ;; errorpages) image="${REGISTRY}/forust/error-pages" - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -406,7 +406,7 @@ jobs: done ;; userbot) - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -449,7 +449,7 @@ jobs: image="${REGISTRY}/forust/xdfnx-homepage" ;; esac - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -483,7 +483,7 @@ jobs: image="${REGISTRY}/forust/webinar-checker" ;; esac - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index a815900..39d5113 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -233,21 +233,95 @@ registry_digest() { | head -1 } +# Rewrites our own images to immutable digests on the way into the cluster. +# Reads a manifest stream on stdin, writes the pinned stream to stdout. +# +# A digest is not knowable when a manifest is written, so it is resolved here, at +# apply time, and never committed: git keeps a readable `:prod` tag. That is what +# makes rollback mean something. `kubectl rollout undo` restores the previous +# ReplicaSet's pod template verbatim, and a template naming a digest restores the +# exact bytes that were serving before. A template naming a moving tag does not — +# the tag has already moved by the time the rollback runs, so the "rollback" +# re-pulls the very image that just failed and the cluster stays broken. +# +# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a +# reference that is not `:latest` defaults to IfNotPresent, which is what the +# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot +# change, so pulling again buys nothing. +# +# An image that cannot be resolved is fatal. Carrying on would quietly apply a +# mutable tag again, which is the exact failure this function exists to remove. +render_pinned() { + local src refs map ref digest missing=0 + src="$(mktemp)" + refs="$(mktemp)" + map="$(mktemp)" + + cat >"$src" + grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true + + while read -r ref; do + [ -n "$ref" ] || continue + digest="$(registry_digest "$ref")" + if [ -z "$digest" ]; then + echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2 + echo " The build job has to push that tag before the deploy resolves it." >&2 + missing=$((missing + 1)) + continue + fi + printf '%s\t%s\n' "$ref" "$digest" >>"$map" + done <"$refs" + if [ "$missing" -gt 0 ]; then + rm -f "$src" "$refs" "$map" + return 1 + fi + + awk -v mapfile="$map" ' + BEGIN { + while ((getline line < mapfile) > 0) { + i = index(line, "\t") + d[substr(line, 1, i - 1)] = substr(line, i + 1) + } + } + { + if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) { + name = $0 + sub(/^[[:space:]]*image:[[:space:]]*/, "", name) + sub(/[[:space:]]*$/, "", name) + if (name in d) { + pad = $0 + sub(/image:.*/, "", pad) + # Drop the tag: the canonical form used in the docs is repo@sha256:..., + # and leaving :prod next to the digest reads like it still matters. + repo = name + sub(/:[A-Za-z0-9._-]+$/, "", repo) + print pad "image: " repo "@" d[name] + next + } + } + print + } + ' "$src" + rm -f "$src" "$refs" "$map" +} + # Restarts every owned workload whose running image is not the one its tag # resolves to now. # -# Our manifests pin images to `:latest`, so a rebuild leaves the pod template -# byte-identical, `kubectl apply` decides there is nothing to do, no ReplicaSet is -# created and nothing is pulled. imagePullPolicy: Always does not help here: it -# only decides whether a pod that *is* starting pulls, and no pod ever starts. The -# cluster keeps serving the previous build indefinitely. +# This used to be how a rebuild reached the cluster at all: the manifests pinned +# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply` +# decided there was nothing to do, and the cluster served the previous build +# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves +# the pod template and rolls out on its own. # -# Comparing the running imageID against the registry is what makes this converge, -# and it is idempotent: when the tag still points at the digest a pod is already -# running, nothing is restarted, so a redeploy that changed no image does not -# bounce healthy services. When the tag *has* moved, the restart bumps the -# generation, which is what makes the change visible to changed_workloads and -# therefore watchable and revertible by the verify stage. +# What is left is the drift check: a hand-run `kubectl set image`, or anything +# else that edits a live workload behind the deploy's back, is the only way to end +# up serving a digest the tag has moved past. It stays idempotent, so a redeploy +# that changed no image still does not bounce healthy services. +# +# The container is matched on its repository rather than on the exact reference: +# once render_pinned has run, a pod's status reports `repo@sha256:...` while this +# still reads the repository's `:prod` tag out of the manifest. restart_stale_images() { local ns target image want selector running entry one local unchecked=0 @@ -272,9 +346,12 @@ restart_stale_images() { continue fi running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \ - | jq -r --arg img "$image" ' + | jq -r --arg repo "${image%%:*}" ' .items[] | .status.containerStatuses[]? - | select(.image == $img) | .imageID + | select(.image == $repo + or (.image | startswith($repo + ":")) + or (.image | startswith($repo + "@"))) + | .imageID ' 2>/dev/null)" if [ -z "$running" ]; then # Scaled to zero. Nothing is serving stale code, and imagePullPolicy @@ -560,14 +637,14 @@ stage_apply_k8s() { fi upgrade_helm_releases if [ "${#other_files[@]}" -gt 0 ]; then - log "Applying resources (${#other_files[@]} files)" + log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do - kubectl apply "${prune_opts[@]}" -f "$m" + render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f - done fi for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do - log "Applying kustomize app: ${k#"$REPO"/}" - kubectl apply -k "$k" + log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)" + kubectl kustomize "$k" | render_pinned | kubectl apply -f - done if [ -f "$REPO/userbot/k8s/active" ]; then log "userbot panel hook" diff --git a/edu_master/k8s/session-keeper.yaml b/edu_master/k8s/session-keeper.yaml index 0f54963..73df466 100644 --- a/edu_master/k8s/session-keeper.yaml +++ b/edu_master/k8s/session-keeper.yaml @@ -31,8 +31,7 @@ spec: echo "redis is ready" containers: - name: session-keeper - image: gcr.forust.xyz/forust/session-keeper:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/session-keeper:prod envFrom: - secretRef: name: edu-master-secrets diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml index e90e3e1..c028aad 100644 --- a/edu_master/k8s/webinar-checker.yaml +++ b/edu_master/k8s/webinar-checker.yaml @@ -45,8 +45,7 @@ spec: echo "playwright ok" containers: - name: webinar-checker - image: gcr.forust.xyz/forust/webinar-checker:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/webinar-checker:prod ports: - name: metrics containerPort: 8000 diff --git a/errorpages/k8s/error-pages.yaml b/errorpages/k8s/error-pages.yaml index 601a37d..0f7b993 100644 --- a/errorpages/k8s/error-pages.yaml +++ b/errorpages/k8s/error-pages.yaml @@ -27,8 +27,7 @@ spec: spec: containers: - name: error-pages - image: gcr.forust.xyz/forust/error-pages:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/error-pages:prod ports: - containerPort: 80 readinessProbe: diff --git a/homepages/k8s/homepages.yaml b/homepages/k8s/homepages.yaml index b0c30f8..96314b3 100644 --- a/homepages/k8s/homepages.yaml +++ b/homepages/k8s/homepages.yaml @@ -27,8 +27,7 @@ spec: spec: containers: - name: forust-homepage - image: gcr.forust.xyz/forust/forust-homepage:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/forust-homepage:prod ports: - containerPort: 80 readinessProbe: @@ -75,8 +74,7 @@ spec: spec: containers: - name: xdfnx-homepage - image: gcr.forust.xyz/forust/xdfnx-homepage:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/xdfnx-homepage:prod ports: - containerPort: 80 readinessProbe: diff --git a/userbot/k8s/base/panel.yaml b/userbot/k8s/base/panel.yaml index 2de09f6..0afe69b 100644 --- a/userbot/k8s/base/panel.yaml +++ b/userbot/k8s/base/panel.yaml @@ -198,8 +198,7 @@ spec: serviceAccountName: userbot-panel containers: - name: userbot-panel - image: gcr.forust.xyz/forust/userbot-panel:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/userbot-panel:prod ports: - name: http containerPort: 8080 diff --git a/userbot/k8s/base/userbots.yaml b/userbot/k8s/base/userbots.yaml index 95664c0..1ab7792 100644 --- a/userbot/k8s/base/userbots.yaml +++ b/userbot/k8s/base/userbots.yaml @@ -24,8 +24,7 @@ spec: spec: containers: - name: forust-userbot - image: gcr.forust.xyz/forust/userbot:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/userbot:prod resources: limits: memory: "1.5Gi" @@ -96,8 +95,7 @@ spec: spec: containers: - name: anna-userbot - image: gcr.forust.xyz/forust/userbot:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/userbot:prod resources: limits: memory: "1.5Gi"