fix(deploy): restart workloads whose image tag moved past what they run
Our manifests pin images to `:latest`, so a rebuild leaves the pod template byte-identical. kubectl apply sees no change, creates no ReplicaSet and pulls nothing, and the cluster keeps serving the previous build. imagePullPolicy: Always does not help, because it only decides whether a pod that *is* starting pulls, and no pod ever starts. All eight workloads that consume an image from our own registry were affected. Three of them had been running code from 23 September, and the single hardcoded `rollout restart deployment/userbot-panel` covered one of the eight. Restarting everything unconditionally was not the answer either: that bounces healthy services on every deploy, error-pages included, and the brief window where nothing answers is exactly what error-pages exists to prevent. So compare what each workload actually runs against what the tag resolves to now, and restart only the ones that differ. When the tag still points at the running digest nothing happens, so a redeploy that changed no image is a no-op. Scope is the repository, deliberately. Five more workloads run our images but have no manifest here, and they are applied out of band. Walking the manifests rather than the cluster means this can never reach them. The digest is resolved for the node architecture. A multi-arch tag also carries `unknown/unknown` entries for the build attestation, and a pod's imageID is always the per-platform digest, so comparing the wrong entry would mark everything stale forever. Once a restart happens it bumps the generation, which is what makes the change visible to changed_workloads and therefore watchable and revertible by the verify stage.
This commit is contained in:
1 parent
1505b638ce
commit
db7bccfd89
2 files changed
+147
-1
No files matched your search
@@ -174,6 +174,151 @@ changed_workloads() {
|
||||
done < <(workload_generations)
|
||||
}
|
||||
|
||||
# Prints "<ns> <kind>/<name> <image>" for every workload this repository owns that
|
||||
# runs an image from our own registry.
|
||||
#
|
||||
# The repository is the scope, deliberately. The cluster also holds workloads on
|
||||
# our registry that no manifest here declares (they are applied out of band), and
|
||||
# those are somebody else's to deploy. Walking the manifests rather than the
|
||||
# cluster means those can never be restarted by this pipeline, now or later.
|
||||
owned_registry_workloads() {
|
||||
local kd_rel f
|
||||
while IFS= read -r kd_rel; do
|
||||
[ -f "$REPO/$kd_rel/active" ] || continue
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] || continue
|
||||
# A file that does not mention the registry cannot declare a workload on it,
|
||||
# and parsing costs ~2.5s per file against a millisecond for the grep. The
|
||||
# filter keeps this at a handful of parses instead of one per manifest.
|
||||
grep -q 'gcr\.forust\.xyz/forust/' "$REPO/$f" 2>/dev/null || continue
|
||||
# kubectl prints a bare object for a single-document file and a List for a
|
||||
# multi-document one, so normalise both shapes before filtering.
|
||||
kubectl apply --dry-run=client -f "$REPO/$f" -o json 2>/dev/null \
|
||||
| jq -r '
|
||||
(if .items then .items[] else . end)
|
||||
| select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
|
||||
| select(any((.spec.template.spec.containers // [])[]?;
|
||||
(.image // "") | test("^gcr\\.forust\\.xyz/forust/")))
|
||||
| (.metadata.namespace // "default") as $ns
|
||||
| ([.spec.template.spec.containers[].image
|
||||
| select(test("^gcr\\.forust\\.xyz/forust/"))][0]) as $img
|
||||
| "\($ns) \(.kind | ascii_downcase)/\(.metadata.name) \($img)"
|
||||
' 2>/dev/null || true
|
||||
done < <(collect_k8s "$kd_rel" || true)
|
||||
done < <(
|
||||
git -C "$REPO" ls-files '*.yaml' '*.yml' \
|
||||
| grep -E '(^|/)k8s/' \
|
||||
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
|
||||
| sort -u
|
||||
)
|
||||
}
|
||||
|
||||
# Prints the digest an image tag resolves to for this cluster's architecture, or
|
||||
# nothing when it cannot be resolved.
|
||||
#
|
||||
# Only the manifest entry matching the node architecture counts. A multi-arch tag
|
||||
# also carries `unknown/unknown` entries for the build attestation, and a pod's
|
||||
# imageID is always the per-platform digest, so comparing the wrong entry would
|
||||
# mark every workload stale forever and restart the whole cluster on every deploy.
|
||||
registry_digest() {
|
||||
local arch
|
||||
arch="$(kubectl get nodes -o jsonpath='{.items[0].status.nodeInfo.architecture}' 2>/dev/null)"
|
||||
[ -n "$arch" ] || arch=amd64
|
||||
docker manifest inspect "$1" 2>/dev/null \
|
||||
| jq -r --arg arch "$arch" '
|
||||
.manifests[]?
|
||||
| select(.platform.os == "linux" and .platform.architecture == $arch)
|
||||
| .digest
|
||||
' 2>/dev/null \
|
||||
| head -1
|
||||
}
|
||||
|
||||
# Restarts every owned workload whose running image is not the one its tag
|
||||
# resolves to now.
|
||||
#
|
||||
# Our manifests pin images to `:latest`, so a rebuild leaves the pod template
|
||||
# byte-identical, `kubectl apply` decides there is nothing to do, no ReplicaSet is
|
||||
# created and nothing is pulled. imagePullPolicy: Always does not help here: it
|
||||
# only decides whether a pod that *is* starting pulls, and no pod ever starts. The
|
||||
# cluster keeps serving the previous build indefinitely.
|
||||
#
|
||||
# Comparing the running imageID against the registry is what makes this converge,
|
||||
# and it is idempotent: when the tag still points at the digest a pod is already
|
||||
# running, nothing is restarted, so a redeploy that changed no image does not
|
||||
# bounce healthy services. When the tag *has* moved, the restart bumps the
|
||||
# generation, which is what makes the change visible to changed_workloads and
|
||||
# therefore watchable and revertible by the verify stage.
|
||||
restart_stale_images() {
|
||||
local ns target image want selector running entry one
|
||||
local unchecked=0
|
||||
local -A digests=()
|
||||
local -a stale=()
|
||||
while read -r ns target image; do
|
||||
[ -n "${target:-}" ] || continue
|
||||
if [ -z "${digests[$image]:-}" ]; then
|
||||
digests[$image]="$(registry_digest "$image")"
|
||||
fi
|
||||
want="${digests[$image]}"
|
||||
if [ -z "$want" ]; then
|
||||
warn "cannot resolve ${image##*/} in the registry, leaving $target alone"
|
||||
unchecked=$((unchecked + 1))
|
||||
continue
|
||||
fi
|
||||
selector="$(kubectl get "$target" -n "$ns" -o jsonpath='{.spec.selector.matchLabels}' 2>/dev/null \
|
||||
| jq -r 'to_entries | map("\(.key)=\(.value)") | join(",")' 2>/dev/null)"
|
||||
if [ -z "$selector" ]; then
|
||||
warn "cannot read the pod selector of $target, skipping"
|
||||
unchecked=$((unchecked + 1))
|
||||
continue
|
||||
fi
|
||||
running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \
|
||||
| jq -r --arg img "$image" '
|
||||
.items[] | .status.containerStatuses[]?
|
||||
| select(.image == $img) | .imageID
|
||||
' 2>/dev/null)"
|
||||
if [ -z "$running" ]; then
|
||||
# Scaled to zero. Nothing is serving stale code, and imagePullPolicy
|
||||
# resolves the tag when it is scaled back up.
|
||||
continue
|
||||
fi
|
||||
entry=""
|
||||
while IFS= read -r one; do
|
||||
[ -n "$one" ] || continue
|
||||
entry="${one##*@}"
|
||||
if [ "$entry" != "$want" ]; then
|
||||
stale+=("$ns $target")
|
||||
break
|
||||
fi
|
||||
done <<<"$running"
|
||||
done < <(owned_registry_workloads)
|
||||
if [ "${#stale[@]}" -eq 0 ]; then
|
||||
if [ "$unchecked" -gt 0 ]; then
|
||||
# Say so plainly. Reporting "everything is current" after checking nothing
|
||||
# would tell the operator the deploy is fine when it may not be.
|
||||
warn "No workload needed a restart, but $unchecked could not be checked"
|
||||
else
|
||||
log "All owned workloads already run the image their tag points at"
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
log "Restarting ${#stale[@]} workload(s) running an image their tag has moved past"
|
||||
for ref in "${stale[@]}"; do
|
||||
log " $ref"
|
||||
done
|
||||
local failed=()
|
||||
for ref in "${stale[@]}"; do
|
||||
ns="${ref%% *}"
|
||||
target="${ref#* }"
|
||||
if ! kubectl rollout restart "$target" -n "$ns" >/dev/null 2>&1; then
|
||||
failed+=("$ref")
|
||||
fi
|
||||
done
|
||||
if [ "${#failed[@]}" -gt 0 ]; then
|
||||
warn "could not restart: ${failed[*]}"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# verify_workloads <failed-file> <kind> <ns> <name> ...
|
||||
# Watches every workload in parallel and records the ones that never became
|
||||
# healthy. Returns non-zero if any of them failed.
|
||||
@@ -434,8 +579,8 @@ stage_apply_k8s() {
|
||||
else
|
||||
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
|
||||
fi
|
||||
kubectl rollout restart deployment/userbot-panel -n userbot
|
||||
fi
|
||||
restart_stale_images
|
||||
|
||||
# No verification here on purpose. This stage may be killed at any point by
|
||||
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
|
||||
|
||||
Reference in new issue
Block a user