fix(deploy): recover helm releases from pending-* and skip helm-owned rollbacks
An --atomic upgrade whose own rollback never finishes leaves the release in pending-*, blocking every future run until a human rolls back (loki rev 18/21). Recover automatically before and after each upgrade, and fail loud when recovery does not land on deployed. Also skip helm-managed workloads in rollback_workloads: rollout undo there would step back to the revision --atomic just escaped.
This commit is contained in:
1 parent
7c4843c88c
commit
dde6b1c743
1 file changed
+66
-2
@@ -489,6 +489,14 @@ rollback_workloads() {
|
||||
local -a recovered=()
|
||||
while read -r kind ns name; do
|
||||
[ -n "${kind:-}" ] || continue
|
||||
# Helm-owned workloads are already rolled back by the release's --atomic
|
||||
# upgrade. `rollout undo` here would step back to the revision Helm just
|
||||
# escaped (the failed one), so leave them for the operator instead.
|
||||
if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then
|
||||
echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}"
|
||||
unrecovered+=("${kind}/${ns}/${name} (helm-managed)")
|
||||
continue
|
||||
fi
|
||||
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
|
||||
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
||||
echo " rolled back: ${kind}/${ns}/${name}"
|
||||
@@ -527,6 +535,44 @@ helm_repo_for() {
|
||||
esac
|
||||
}
|
||||
|
||||
# helm_release_status <release> <namespace>
|
||||
# Prints the release status in lowercase (deployed, failed, pending-rollback,
|
||||
# ...) or "not-found" when the release does not exist yet.
|
||||
helm_release_status() {
|
||||
local out
|
||||
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
|
||||
echo "not-found"
|
||||
return 0
|
||||
fi
|
||||
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
|
||||
}
|
||||
|
||||
# recover_pending_release <release> <namespace>
|
||||
# Rolls a release out of a pending-* state left by a failed --atomic upgrade
|
||||
# whose own rollback never completed. Without this every future upgrade errors
|
||||
# out until a human runs `helm rollback`. Passes through releases that are not
|
||||
# pending (deployed, failed, not-found). Returns non-zero when the release is
|
||||
# still not recoverable, so the pipeline fails loud instead of wedging.
|
||||
recover_pending_release() {
|
||||
local release="$1" namespace="$2" status
|
||||
status="$(helm_release_status "$release" "$namespace")"
|
||||
case "$status" in
|
||||
pending-upgrade|pending-rollback|pending-install)
|
||||
log "Release $release is $status, rolling back to the last deployed revision"
|
||||
if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then
|
||||
echo "WARN: helm rollback of $release did not complete"
|
||||
return 1
|
||||
fi
|
||||
status="$(helm_release_status "$release" "$namespace")"
|
||||
if [ "$status" != "deployed" ]; then
|
||||
echo "WARN: $release is $status after rollback"
|
||||
return 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
return 0
|
||||
}
|
||||
|
||||
upgrade_helm_releases() {
|
||||
local entry release chart namespace version values marker repo
|
||||
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||
@@ -547,13 +593,31 @@ upgrade_helm_releases() {
|
||||
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||
log "Upgrading $release ($chart $version)"
|
||||
# A previous --atomic run whose own rollback never finished leaves the
|
||||
# release in pending-*, which blocks every future upgrade. Recover first
|
||||
# so one wedged revision cannot wedge the pipeline forever.
|
||||
if ! recover_pending_release "$release" "$namespace"; then
|
||||
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
|
||||
return 1
|
||||
fi
|
||||
# --atomic rolls the release back when the upgrade times out or the workloads
|
||||
# it touches never become ready, so a bad chart bump is not left half applied.
|
||||
helm upgrade --install "$release" "$chart" \
|
||||
if ! helm upgrade --install "$release" "$chart" \
|
||||
--namespace "$namespace" \
|
||||
--version "$version" \
|
||||
--values "$REPO/$values" \
|
||||
--atomic --cleanup-on-fail --timeout 10m
|
||||
--atomic --cleanup-on-fail --timeout 10m; then
|
||||
echo "WARN: upgrade of $release failed, checking release state"
|
||||
# --atomic already attempted its own rollback; finish the job when that
|
||||
# rollback never completed, otherwise the release stays pending-* and
|
||||
# blocks every future run.
|
||||
if ! recover_pending_release "$release" "$namespace"; then
|
||||
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
|
||||
else
|
||||
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
|
||||
fi
|
||||
return 1
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user