fix(deploy): roll out our images by digest instead of a moving tag

`kubectl rollout undo` restores the previous ReplicaSet's pod template
verbatim. While that template names a tag, the rollback does not roll back
the image: the tag has already moved, so the reverted pod pulls the very
build that just failed and the cluster stays broken. The safety net added
in 1505b63 therefore could not recover from a bad image.

Pin the digest at apply time. A digest is not knowable when a manifest is
written, so render_pinned resolves it on the way into the cluster and the
digest is never committed. Git keeps a readable `:prod`, Renovate keeps
seeing exactly the manifests it saw before, and the previous revision of
each workload now holds the digest that was actually serving, so undo
restores those exact bytes.

imagePullPolicy is dropped from the manifests rather than set to
IfNotPresent: a reference that is not `:latest` already defaults to it, and
that is what the Kubernetes docs ask for alongside a digest.

An unresolvable image is fatal instead of a warning, because carrying on
would quietly apply a mutable tag again.

restart_stale_images keeps its comparison but is no longer how a rebuild
reaches the cluster -- the pinned template rolls out on its own now. What
is left is a drift check for hand-run `kubectl set image`, so it matches
the container by repository: a pod's status now reports `repo@sha256:...`
while the manifest still says `:prod`.

The build job stops pushing `:latest` altogether, which removes the tag
that a dev branch could otherwise move under a prod deploy.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
forustandClaude Opus 4.8 committed 2026-09-27 10:09:06 +02:00
1 parent 2b9e34ba4a
commit 0691536f28
8 files changed
+107 -38

No files matched your search

+5 -5
View File
@@ -361,7 +361,7 @@ jobs:
case "$service" in
dtek_notif)
image="${REGISTRY}/forust/dtek-notif"
tags=("latest")
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -384,7 +384,7 @@ jobs:
;;
errorpages)
image="${REGISTRY}/forust/error-pages"
tags=("latest")
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -406,7 +406,7 @@ jobs:
done
;;
userbot)
tags=("latest")
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -449,7 +449,7 @@ jobs:
image="${REGISTRY}/forust/xdfnx-homepage"
;;
esac
tags=("latest")
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -483,7 +483,7 @@ jobs:
image="${REGISTRY}/forust/webinar-checker"
;;
esac
tags=("latest")
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
+94 -17
View File
@@ -233,21 +233,95 @@ registry_digest() {
| head -1
}
# Rewrites our own images to immutable digests on the way into the cluster.
# Reads a manifest stream on stdin, writes the pinned stream to stdout.
#
# A digest is not knowable when a manifest is written, so it is resolved here, at
# apply time, and never committed: git keeps a readable `:prod` tag. That is what
# makes rollback mean something. `kubectl rollout undo` restores the previous
# ReplicaSet's pod template verbatim, and a template naming a digest restores the
# exact bytes that were serving before. A template naming a moving tag does not —
# the tag has already moved by the time the rollback runs, so the "rollback"
# re-pulls the very image that just failed and the cluster stays broken.
#
# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a
# reference that is not `:latest` defaults to IfNotPresent, which is what the
# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot
# change, so pulling again buys nothing.
#
# An image that cannot be resolved is fatal. Carrying on would quietly apply a
# mutable tag again, which is the exact failure this function exists to remove.
render_pinned() {
local src refs map ref digest missing=0
src="$(mktemp)"
refs="$(mktemp)"
map="$(mktemp)"
cat >"$src"
grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true
while read -r ref; do
[ -n "$ref" ] || continue
digest="$(registry_digest "$ref")"
if [ -z "$digest" ]; then
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
echo " The build job has to push that tag before the deploy resolves it." >&2
missing=$((missing + 1))
continue
fi
printf '%s\t%s\n' "$ref" "$digest" >>"$map"
done <"$refs"
if [ "$missing" -gt 0 ]; then
rm -f "$src" "$refs" "$map"
return 1
fi
awk -v mapfile="$map" '
BEGIN {
while ((getline line < mapfile) > 0) {
i = index(line, "\t")
d[substr(line, 1, i - 1)] = substr(line, i + 1)
}
}
{
if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) {
name = $0
sub(/^[[:space:]]*image:[[:space:]]*/, "", name)
sub(/[[:space:]]*$/, "", name)
if (name in d) {
pad = $0
sub(/image:.*/, "", pad)
# Drop the tag: the canonical form used in the docs is repo@sha256:...,
# and leaving :prod next to the digest reads like it still matters.
repo = name
sub(/:[A-Za-z0-9._-]+$/, "", repo)
print pad "image: " repo "@" d[name]
next
}
}
print
}
' "$src"
rm -f "$src" "$refs" "$map"
}
# Restarts every owned workload whose running image is not the one its tag
# resolves to now.
#
# Our manifests pin images to `:latest`, so a rebuild leaves the pod template
# byte-identical, `kubectl apply` decides there is nothing to do, no ReplicaSet is
# created and nothing is pulled. imagePullPolicy: Always does not help here: it
# only decides whether a pod that *is* starting pulls, and no pod ever starts. The
# cluster keeps serving the previous build indefinitely.
# This used to be how a rebuild reached the cluster at all: the manifests pinned
# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply`
# decided there was nothing to do, and the cluster served the previous build
# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves
# the pod template and rolls out on its own.
#
# Comparing the running imageID against the registry is what makes this converge,
# and it is idempotent: when the tag still points at the digest a pod is already
# running, nothing is restarted, so a redeploy that changed no image does not
# bounce healthy services. When the tag *has* moved, the restart bumps the
# generation, which is what makes the change visible to changed_workloads and
# therefore watchable and revertible by the verify stage.
# What is left is the drift check: a hand-run `kubectl set image`, or anything
# else that edits a live workload behind the deploy's back, is the only way to end
# up serving a digest the tag has moved past. It stays idempotent, so a redeploy
# that changed no image still does not bounce healthy services.
#
# The container is matched on its repository rather than on the exact reference:
# once render_pinned has run, a pod's status reports `repo@sha256:...` while this
# still reads the repository's `:prod` tag out of the manifest.
restart_stale_images() {
local ns target image want selector running entry one
local unchecked=0
@@ -272,9 +346,12 @@ restart_stale_images() {
continue
fi
running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \
| jq -r --arg img "$image" '
| jq -r --arg repo "${image%%:*}" '
.items[] | .status.containerStatuses[]?
| select(.image == $img) | .imageID
| select(.image == $repo
or (.image | startswith($repo + ":"))
or (.image | startswith($repo + "@")))
| .imageID
' 2>/dev/null)"
if [ -z "$running" ]; then
# Scaled to zero. Nothing is serving stale code, and imagePullPolicy
@@ -560,14 +637,14 @@ stage_apply_k8s() {
fi
upgrade_helm_releases
if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files)"
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
for m in "${other_files[@]}"; do
kubectl apply "${prune_opts[@]}" -f "$m"
render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/}"
kubectl apply -k "$k"
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
kubectl kustomize "$k" | render_pinned | kubectl apply -f -
done
if [ -f "$REPO/userbot/k8s/active" ]; then
log "userbot panel hook"