fix(deploy): roll out our images by digest instead of a moving tag

`kubectl rollout undo` restores the previous ReplicaSet's pod template
verbatim. While that template names a tag, the rollback does not roll back
the image: the tag has already moved, so the reverted pod pulls the very
build that just failed and the cluster stays broken. The safety net added
in 1505b63 therefore could not recover from a bad image.

Pin the digest at apply time. A digest is not knowable when a manifest is
written, so render_pinned resolves it on the way into the cluster and the
digest is never committed. Git keeps a readable `:prod`, Renovate keeps
seeing exactly the manifests it saw before, and the previous revision of
each workload now holds the digest that was actually serving, so undo
restores those exact bytes.

imagePullPolicy is dropped from the manifests rather than set to
IfNotPresent: a reference that is not `:latest` already defaults to it, and
that is what the Kubernetes docs ask for alongside a digest.

An unresolvable image is fatal instead of a warning, because carrying on
would quietly apply a mutable tag again.

restart_stale_images keeps its comparison but is no longer how a rebuild
reaches the cluster -- the pinned template rolls out on its own now. What
is left is a drift check for hand-run `kubectl set image`, so it matches
the container by repository: a pod's status now reports `repo@sha256:...`
while the manifest still says `:prod`.

The build job stops pushing `:latest` altogether, which removes the tag
that a dev branch could otherwise move under a prod deploy.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
forustandClaude Opus 4.8 committed 2026-09-27 10:09:06 +02:00
1 parent 2b9e34ba4a
commit 0691536f28
8 files changed
+107 -38

No files matched your search

+5 -5
View File
@@ -361,7 +361,7 @@ jobs:
case "$service" in case "$service" in
dtek_notif) dtek_notif)
image="${REGISTRY}/forust/dtek-notif" image="${REGISTRY}/forust/dtek-notif"
tags=("latest") tags=()
case "${GITHUB_REF_NAME}" in case "${GITHUB_REF_NAME}" in
main) main)
tags+=("main" "prod") tags+=("main" "prod")
@@ -384,7 +384,7 @@ jobs:
;; ;;
errorpages) errorpages)
image="${REGISTRY}/forust/error-pages" image="${REGISTRY}/forust/error-pages"
tags=("latest") tags=()
case "${GITHUB_REF_NAME}" in case "${GITHUB_REF_NAME}" in
main) main)
tags+=("main" "prod") tags+=("main" "prod")
@@ -406,7 +406,7 @@ jobs:
done done
;; ;;
userbot) userbot)
tags=("latest") tags=()
case "${GITHUB_REF_NAME}" in case "${GITHUB_REF_NAME}" in
main) main)
tags+=("main" "prod") tags+=("main" "prod")
@@ -449,7 +449,7 @@ jobs:
image="${REGISTRY}/forust/xdfnx-homepage" image="${REGISTRY}/forust/xdfnx-homepage"
;; ;;
esac esac
tags=("latest") tags=()
case "${GITHUB_REF_NAME}" in case "${GITHUB_REF_NAME}" in
main) main)
tags+=("main" "prod") tags+=("main" "prod")
@@ -483,7 +483,7 @@ jobs:
image="${REGISTRY}/forust/webinar-checker" image="${REGISTRY}/forust/webinar-checker"
;; ;;
esac esac
tags=("latest") tags=()
case "${GITHUB_REF_NAME}" in case "${GITHUB_REF_NAME}" in
main) main)
tags+=("main" "prod") tags+=("main" "prod")
+94 -17
View File
@@ -233,21 +233,95 @@ registry_digest() {
| head -1 | head -1
} }
# Rewrites our own images to immutable digests on the way into the cluster.
# Reads a manifest stream on stdin, writes the pinned stream to stdout.
#
# A digest is not knowable when a manifest is written, so it is resolved here, at
# apply time, and never committed: git keeps a readable `:prod` tag. That is what
# makes rollback mean something. `kubectl rollout undo` restores the previous
# ReplicaSet's pod template verbatim, and a template naming a digest restores the
# exact bytes that were serving before. A template naming a moving tag does not —
# the tag has already moved by the time the rollback runs, so the "rollback"
# re-pulls the very image that just failed and the cluster stays broken.
#
# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a
# reference that is not `:latest` defaults to IfNotPresent, which is what the
# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot
# change, so pulling again buys nothing.
#
# An image that cannot be resolved is fatal. Carrying on would quietly apply a
# mutable tag again, which is the exact failure this function exists to remove.
render_pinned() {
local src refs map ref digest missing=0
src="$(mktemp)"
refs="$(mktemp)"
map="$(mktemp)"
cat >"$src"
grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true
while read -r ref; do
[ -n "$ref" ] || continue
digest="$(registry_digest "$ref")"
if [ -z "$digest" ]; then
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
echo " The build job has to push that tag before the deploy resolves it." >&2
missing=$((missing + 1))
continue
fi
printf '%s\t%s\n' "$ref" "$digest" >>"$map"
done <"$refs"
if [ "$missing" -gt 0 ]; then
rm -f "$src" "$refs" "$map"
return 1
fi
awk -v mapfile="$map" '
BEGIN {
while ((getline line < mapfile) > 0) {
i = index(line, "\t")
d[substr(line, 1, i - 1)] = substr(line, i + 1)
}
}
{
if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) {
name = $0
sub(/^[[:space:]]*image:[[:space:]]*/, "", name)
sub(/[[:space:]]*$/, "", name)
if (name in d) {
pad = $0
sub(/image:.*/, "", pad)
# Drop the tag: the canonical form used in the docs is repo@sha256:...,
# and leaving :prod next to the digest reads like it still matters.
repo = name
sub(/:[A-Za-z0-9._-]+$/, "", repo)
print pad "image: " repo "@" d[name]
next
}
}
print
}
' "$src"
rm -f "$src" "$refs" "$map"
}
# Restarts every owned workload whose running image is not the one its tag # Restarts every owned workload whose running image is not the one its tag
# resolves to now. # resolves to now.
# #
# Our manifests pin images to `:latest`, so a rebuild leaves the pod template # This used to be how a rebuild reached the cluster at all: the manifests pinned
# byte-identical, `kubectl apply` decides there is nothing to do, no ReplicaSet is # `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply`
# created and nothing is pulled. imagePullPolicy: Always does not help here: it # decided there was nothing to do, and the cluster served the previous build
# only decides whether a pod that *is* starting pulls, and no pod ever starts. The # indefinitely. The apply now pins digests via render_pinned, so a rebuild moves
# cluster keeps serving the previous build indefinitely. # the pod template and rolls out on its own.
# #
# Comparing the running imageID against the registry is what makes this converge, # What is left is the drift check: a hand-run `kubectl set image`, or anything
# and it is idempotent: when the tag still points at the digest a pod is already # else that edits a live workload behind the deploy's back, is the only way to end
# running, nothing is restarted, so a redeploy that changed no image does not # up serving a digest the tag has moved past. It stays idempotent, so a redeploy
# bounce healthy services. When the tag *has* moved, the restart bumps the # that changed no image still does not bounce healthy services.
# generation, which is what makes the change visible to changed_workloads and #
# therefore watchable and revertible by the verify stage. # The container is matched on its repository rather than on the exact reference:
# once render_pinned has run, a pod's status reports `repo@sha256:...` while this
# still reads the repository's `:prod` tag out of the manifest.
restart_stale_images() { restart_stale_images() {
local ns target image want selector running entry one local ns target image want selector running entry one
local unchecked=0 local unchecked=0
@@ -272,9 +346,12 @@ restart_stale_images() {
continue continue
fi fi
running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \ running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \
| jq -r --arg img "$image" ' | jq -r --arg repo "${image%%:*}" '
.items[] | .status.containerStatuses[]? .items[] | .status.containerStatuses[]?
| select(.image == $img) | .imageID | select(.image == $repo
or (.image | startswith($repo + ":"))
or (.image | startswith($repo + "@")))
| .imageID
' 2>/dev/null)" ' 2>/dev/null)"
if [ -z "$running" ]; then if [ -z "$running" ]; then
# Scaled to zero. Nothing is serving stale code, and imagePullPolicy # Scaled to zero. Nothing is serving stale code, and imagePullPolicy
@@ -560,14 +637,14 @@ stage_apply_k8s() {
fi fi
upgrade_helm_releases upgrade_helm_releases
if [ "${#other_files[@]}" -gt 0 ]; then if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files)" log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
for m in "${other_files[@]}"; do for m in "${other_files[@]}"; do
kubectl apply "${prune_opts[@]}" -f "$m" render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -
done done
fi fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/}" log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
kubectl apply -k "$k" kubectl kustomize "$k" | render_pinned | kubectl apply -f -
done done
if [ -f "$REPO/userbot/k8s/active" ]; then if [ -f "$REPO/userbot/k8s/active" ]; then
log "userbot panel hook" log "userbot panel hook"
+1 -2
View File
@@ -31,8 +31,7 @@ spec:
echo "redis is ready" echo "redis is ready"
containers: containers:
- name: session-keeper - name: session-keeper
image: gcr.forust.xyz/forust/session-keeper:latest image: gcr.forust.xyz/forust/session-keeper:prod
imagePullPolicy: Always
envFrom: envFrom:
- secretRef: - secretRef:
name: edu-master-secrets name: edu-master-secrets
+1 -2
View File
@@ -45,8 +45,7 @@ spec:
echo "playwright ok" echo "playwright ok"
containers: containers:
- name: webinar-checker - name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:latest image: gcr.forust.xyz/forust/webinar-checker:prod
imagePullPolicy: Always
ports: ports:
- name: metrics - name: metrics
containerPort: 8000 containerPort: 8000
+1 -2
View File
@@ -27,8 +27,7 @@ spec:
spec: spec:
containers: containers:
- name: error-pages - name: error-pages
image: gcr.forust.xyz/forust/error-pages:latest image: gcr.forust.xyz/forust/error-pages:prod
imagePullPolicy: Always
ports: ports:
- containerPort: 80 - containerPort: 80
readinessProbe: readinessProbe:
+2 -4
View File
@@ -27,8 +27,7 @@ spec:
spec: spec:
containers: containers:
- name: forust-homepage - name: forust-homepage
image: gcr.forust.xyz/forust/forust-homepage:latest image: gcr.forust.xyz/forust/forust-homepage:prod
imagePullPolicy: Always
ports: ports:
- containerPort: 80 - containerPort: 80
readinessProbe: readinessProbe:
@@ -75,8 +74,7 @@ spec:
spec: spec:
containers: containers:
- name: xdfnx-homepage - name: xdfnx-homepage
image: gcr.forust.xyz/forust/xdfnx-homepage:latest image: gcr.forust.xyz/forust/xdfnx-homepage:prod
imagePullPolicy: Always
ports: ports:
- containerPort: 80 - containerPort: 80
readinessProbe: readinessProbe:
+1 -2
View File
@@ -198,8 +198,7 @@ spec:
serviceAccountName: userbot-panel serviceAccountName: userbot-panel
containers: containers:
- name: userbot-panel - name: userbot-panel
image: gcr.forust.xyz/forust/userbot-panel:latest image: gcr.forust.xyz/forust/userbot-panel:prod
imagePullPolicy: Always
ports: ports:
- name: http - name: http
containerPort: 8080 containerPort: 8080
+2 -4
View File
@@ -24,8 +24,7 @@ spec:
spec: spec:
containers: containers:
- name: forust-userbot - name: forust-userbot
image: gcr.forust.xyz/forust/userbot:latest image: gcr.forust.xyz/forust/userbot:prod
imagePullPolicy: Always
resources: resources:
limits: limits:
memory: "1.5Gi" memory: "1.5Gi"
@@ -96,8 +95,7 @@ spec:
spec: spec:
containers: containers:
- name: anna-userbot - name: anna-userbot
image: gcr.forust.xyz/forust/userbot:latest image: gcr.forust.xyz/forust/userbot:prod
imagePullPolicy: Always
resources: resources:
limits: limits:
memory: "1.5Gi" memory: "1.5Gi"