From 2cb06debc55e6fa846c689a246e8799331258b21 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Mon, 28 Sep 2026 22:52:54 +0200 Subject: [PATCH] fix(deploy): retry registry lookups with a timeout A single blink of the registry failed render_pinned for the whole file and redded the apply stage. registry_digest now retries 3 times under a 25s timeout with a warning per attempt; empty still means unresolvable and callers report it by name as before. --- .gitea/workflows/deploy-lib.sh | 29 ++++++++++++++++++++++------- .gitea/workflows/deploy.yaml | 12 ++++++------ 2 files changed, 28 insertions(+), 13 deletions(-) diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index bbb640a..7aabbe5 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -233,13 +233,28 @@ registry_digest() { # pipefail reports the rightmost non-zero stage, so a ref the registry does not # have would abort the caller at the assignment instead of yielding an empty # string. The callers check for empty themselves and report it by name. - docker manifest inspect "$1" 2>/dev/null \ - | jq -r --arg arch "$arch" ' - .manifests[]? - | select(.platform.os == "linux" and .platform.architecture == $arch) - | .digest - ' 2>/dev/null \ - | head -1 || true + # + # Retried with a hard timeout because the registry has a known hang mode (and + # a known blink mode: a single failed lookup aborts the whole apply file in + # render_pinned). A short sleep between attempts lets a restarting registry + # come back instead of failing the deploy on one bad second. + local attempt=0 digest="" + while [ "$attempt" -lt 3 ]; do + digest="$(timeout 25s docker manifest inspect "$1" 2>/dev/null \ + | jq -r --arg arch "$arch" ' + .manifests[]? + | select(.platform.os == "linux" and .platform.architecture == $arch) + | .digest + ' 2>/dev/null \ + | head -1 || true)" + [ -n "$digest" ] && break + attempt=$((attempt + 1)) + if [ "$attempt" -lt 3 ]; then + echo "WARNING: registry lookup of $1 failed (attempt $attempt/3), retrying in 5s" >&2 + sleep 5 + fi + done + printf '%s' "$digest" } # The commit this deploy is for: what CI validated, or - on a manual dispatch, diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index ba2e37f..5120d32 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -95,12 +95,12 @@ jobs: # The 45 minutes this was last raised to 45 were still not enough, and the # job logs for those runs no longer exist, so what actually consumed the # budget is not known - the two measurable candidates above account for - # ~15 of it. The one unbounded thing left in this stage is - # `docker manifest inspect` at deploy-lib.sh:236, which has no timeout - # against a registry with a known hang mode. Bound it, and make the stage - # announce what it is working on, before spending any of that on a larger - # ceiling: a stage that is killed with a diagnosable last line is a bug - # report, one that vanishes is not. + # ~15 of it. The unbounded `docker manifest inspect` against the registry's + # known hang mode is now bounded inside registry_digest (25s timeout, 3 + # attempts): a dead registry fails each owned image after ~85s instead of + # hanging the stage, and a blinking one is retried instead of failing the + # whole apply file. Still open: make the stage announce which manifest it + # is working on, so a killed run leaves a diagnosable last line. timeout-minutes: 45 steps: - name: Checkout repository