From 0691536f286bbdba29682a1180028390ea1902b5 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Sun, 27 Sep 2026 10:09:06 +0200 Subject: [PATCH] fix(deploy): roll out our images by digest instead of a moving tag `kubectl rollout undo` restores the previous ReplicaSet's pod template verbatim. While that template names a tag, the rollback does not roll back the image: the tag has already moved, so the reverted pod pulls the very build that just failed and the cluster stays broken. The safety net added in 1505b63 therefore could not recover from a bad image. Pin the digest at apply time. A digest is not knowable when a manifest is written, so render_pinned resolves it on the way into the cluster and the digest is never committed. Git keeps a readable `:prod`, Renovate keeps seeing exactly the manifests it saw before, and the previous revision of each workload now holds the digest that was actually serving, so undo restores those exact bytes. imagePullPolicy is dropped from the manifests rather than set to IfNotPresent: a reference that is not `:latest` already defaults to it, and that is what the Kubernetes docs ask for alongside a digest. An unresolvable image is fatal instead of a warning, because carrying on would quietly apply a mutable tag again. restart_stale_images keeps its comparison but is no longer how a rebuild reaches the cluster -- the pinned template rolls out on its own now. What is left is a drift check for hand-run `kubectl set image`, so it matches the container by repository: a pod's status now reports `repo@sha256:...` while the manifest still says `:prod`. The build job stops pushing `:latest` altogether, which removes the tag that a dev branch could otherwise move under a prod deploy. Co-Authored-By: Claude Opus 4.8 (1M context) --- .gitea/workflows/ci.yaml | 10 +-- .gitea/workflows/deploy-lib.sh | 111 +++++++++++++++++++++++----- edu_master/k8s/session-keeper.yaml | 3 +- edu_master/k8s/webinar-checker.yaml | 3 +- errorpages/k8s/error-pages.yaml | 3 +- homepages/k8s/homepages.yaml | 6 +- userbot/k8s/base/panel.yaml | 3 +- userbot/k8s/base/userbots.yaml | 6 +- 8 files changed, 107 insertions(+), 38 deletions(-) diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 97f57ce..f32cbe3 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -361,7 +361,7 @@ jobs: case "$service" in dtek_notif) image="${REGISTRY}/forust/dtek-notif" - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -384,7 +384,7 @@ jobs: ;; errorpages) image="${REGISTRY}/forust/error-pages" - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -406,7 +406,7 @@ jobs: done ;; userbot) - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -449,7 +449,7 @@ jobs: image="${REGISTRY}/forust/xdfnx-homepage" ;; esac - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") @@ -483,7 +483,7 @@ jobs: image="${REGISTRY}/forust/webinar-checker" ;; esac - tags=("latest") + tags=() case "${GITHUB_REF_NAME}" in main) tags+=("main" "prod") diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index a815900..39d5113 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -233,21 +233,95 @@ registry_digest() { | head -1 } +# Rewrites our own images to immutable digests on the way into the cluster. +# Reads a manifest stream on stdin, writes the pinned stream to stdout. +# +# A digest is not knowable when a manifest is written, so it is resolved here, at +# apply time, and never committed: git keeps a readable `:prod` tag. That is what +# makes rollback mean something. `kubectl rollout undo` restores the previous +# ReplicaSet's pod template verbatim, and a template naming a digest restores the +# exact bytes that were serving before. A template naming a moving tag does not — +# the tag has already moved by the time the rollback runs, so the "rollback" +# re-pulls the very image that just failed and the cluster stays broken. +# +# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a +# reference that is not `:latest` defaults to IfNotPresent, which is what the +# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot +# change, so pulling again buys nothing. +# +# An image that cannot be resolved is fatal. Carrying on would quietly apply a +# mutable tag again, which is the exact failure this function exists to remove. +render_pinned() { + local src refs map ref digest missing=0 + src="$(mktemp)" + refs="$(mktemp)" + map="$(mktemp)" + + cat >"$src" + grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true + + while read -r ref; do + [ -n "$ref" ] || continue + digest="$(registry_digest "$ref")" + if [ -z "$digest" ]; then + echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2 + echo " The build job has to push that tag before the deploy resolves it." >&2 + missing=$((missing + 1)) + continue + fi + printf '%s\t%s\n' "$ref" "$digest" >>"$map" + done <"$refs" + if [ "$missing" -gt 0 ]; then + rm -f "$src" "$refs" "$map" + return 1 + fi + + awk -v mapfile="$map" ' + BEGIN { + while ((getline line < mapfile) > 0) { + i = index(line, "\t") + d[substr(line, 1, i - 1)] = substr(line, i + 1) + } + } + { + if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) { + name = $0 + sub(/^[[:space:]]*image:[[:space:]]*/, "", name) + sub(/[[:space:]]*$/, "", name) + if (name in d) { + pad = $0 + sub(/image:.*/, "", pad) + # Drop the tag: the canonical form used in the docs is repo@sha256:..., + # and leaving :prod next to the digest reads like it still matters. + repo = name + sub(/:[A-Za-z0-9._-]+$/, "", repo) + print pad "image: " repo "@" d[name] + next + } + } + print + } + ' "$src" + rm -f "$src" "$refs" "$map" +} + # Restarts every owned workload whose running image is not the one its tag # resolves to now. # -# Our manifests pin images to `:latest`, so a rebuild leaves the pod template -# byte-identical, `kubectl apply` decides there is nothing to do, no ReplicaSet is -# created and nothing is pulled. imagePullPolicy: Always does not help here: it -# only decides whether a pod that *is* starting pulls, and no pod ever starts. The -# cluster keeps serving the previous build indefinitely. +# This used to be how a rebuild reached the cluster at all: the manifests pinned +# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply` +# decided there was nothing to do, and the cluster served the previous build +# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves +# the pod template and rolls out on its own. # -# Comparing the running imageID against the registry is what makes this converge, -# and it is idempotent: when the tag still points at the digest a pod is already -# running, nothing is restarted, so a redeploy that changed no image does not -# bounce healthy services. When the tag *has* moved, the restart bumps the -# generation, which is what makes the change visible to changed_workloads and -# therefore watchable and revertible by the verify stage. +# What is left is the drift check: a hand-run `kubectl set image`, or anything +# else that edits a live workload behind the deploy's back, is the only way to end +# up serving a digest the tag has moved past. It stays idempotent, so a redeploy +# that changed no image still does not bounce healthy services. +# +# The container is matched on its repository rather than on the exact reference: +# once render_pinned has run, a pod's status reports `repo@sha256:...` while this +# still reads the repository's `:prod` tag out of the manifest. restart_stale_images() { local ns target image want selector running entry one local unchecked=0 @@ -272,9 +346,12 @@ restart_stale_images() { continue fi running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \ - | jq -r --arg img "$image" ' + | jq -r --arg repo "${image%%:*}" ' .items[] | .status.containerStatuses[]? - | select(.image == $img) | .imageID + | select(.image == $repo + or (.image | startswith($repo + ":")) + or (.image | startswith($repo + "@"))) + | .imageID ' 2>/dev/null)" if [ -z "$running" ]; then # Scaled to zero. Nothing is serving stale code, and imagePullPolicy @@ -560,14 +637,14 @@ stage_apply_k8s() { fi upgrade_helm_releases if [ "${#other_files[@]}" -gt 0 ]; then - log "Applying resources (${#other_files[@]} files)" + log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do - kubectl apply "${prune_opts[@]}" -f "$m" + render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f - done fi for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do - log "Applying kustomize app: ${k#"$REPO"/}" - kubectl apply -k "$k" + log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)" + kubectl kustomize "$k" | render_pinned | kubectl apply -f - done if [ -f "$REPO/userbot/k8s/active" ]; then log "userbot panel hook" diff --git a/edu_master/k8s/session-keeper.yaml b/edu_master/k8s/session-keeper.yaml index 0f54963..73df466 100644 --- a/edu_master/k8s/session-keeper.yaml +++ b/edu_master/k8s/session-keeper.yaml @@ -31,8 +31,7 @@ spec: echo "redis is ready" containers: - name: session-keeper - image: gcr.forust.xyz/forust/session-keeper:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/session-keeper:prod envFrom: - secretRef: name: edu-master-secrets diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml index e90e3e1..c028aad 100644 --- a/edu_master/k8s/webinar-checker.yaml +++ b/edu_master/k8s/webinar-checker.yaml @@ -45,8 +45,7 @@ spec: echo "playwright ok" containers: - name: webinar-checker - image: gcr.forust.xyz/forust/webinar-checker:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/webinar-checker:prod ports: - name: metrics containerPort: 8000 diff --git a/errorpages/k8s/error-pages.yaml b/errorpages/k8s/error-pages.yaml index 601a37d..0f7b993 100644 --- a/errorpages/k8s/error-pages.yaml +++ b/errorpages/k8s/error-pages.yaml @@ -27,8 +27,7 @@ spec: spec: containers: - name: error-pages - image: gcr.forust.xyz/forust/error-pages:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/error-pages:prod ports: - containerPort: 80 readinessProbe: diff --git a/homepages/k8s/homepages.yaml b/homepages/k8s/homepages.yaml index b0c30f8..96314b3 100644 --- a/homepages/k8s/homepages.yaml +++ b/homepages/k8s/homepages.yaml @@ -27,8 +27,7 @@ spec: spec: containers: - name: forust-homepage - image: gcr.forust.xyz/forust/forust-homepage:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/forust-homepage:prod ports: - containerPort: 80 readinessProbe: @@ -75,8 +74,7 @@ spec: spec: containers: - name: xdfnx-homepage - image: gcr.forust.xyz/forust/xdfnx-homepage:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/xdfnx-homepage:prod ports: - containerPort: 80 readinessProbe: diff --git a/userbot/k8s/base/panel.yaml b/userbot/k8s/base/panel.yaml index 2de09f6..0afe69b 100644 --- a/userbot/k8s/base/panel.yaml +++ b/userbot/k8s/base/panel.yaml @@ -198,8 +198,7 @@ spec: serviceAccountName: userbot-panel containers: - name: userbot-panel - image: gcr.forust.xyz/forust/userbot-panel:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/userbot-panel:prod ports: - name: http containerPort: 8080 diff --git a/userbot/k8s/base/userbots.yaml b/userbot/k8s/base/userbots.yaml index 95664c0..1ab7792 100644 --- a/userbot/k8s/base/userbots.yaml +++ b/userbot/k8s/base/userbots.yaml @@ -24,8 +24,7 @@ spec: spec: containers: - name: forust-userbot - image: gcr.forust.xyz/forust/userbot:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/userbot:prod resources: limits: memory: "1.5Gi" @@ -96,8 +95,7 @@ spec: spec: containers: - name: anna-userbot - image: gcr.forust.xyz/forust/userbot:latest - imagePullPolicy: Always + image: gcr.forust.xyz/forust/userbot:prod resources: limits: memory: "1.5Gi"