diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index 6fb1b0a..f74f350 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -21,10 +21,39 @@ ssh_key="$key_dir/deploy_key" printf '%s\n' "$DEPLOY_KEY" > "$ssh_key" chmod 600 "$ssh_key" -ssh -i "$ssh_key" -p "$deploy_port" \ - -o BatchMode=yes -o StrictHostKeyChecking=accept-new \ - "${DEPLOY_USER}@${DEPLOY_HOST}" \ - "REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-} STAGE=$1 bash -se" <<'EOF' +# A connection that died silently used to hang until the job timeout, and the +# stage was never re-run: one flaky TCP session cost a whole 45-minute apply. +# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds +# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that +# fails on its own merits exits with the remote's status, so a real failure +# still surfaces its own log instead of burning three attempts. The stages are +# declarative applies, so re-running one that had already committed is harmless. +ssh_opts=( + -i "$ssh_key" -p "$deploy_port" + -o BatchMode=yes -o StrictHostKeyChecking=accept-new + -o ConnectTimeout=15 + -o ServerAliveInterval=15 -o ServerAliveCountMax=4 +) + +rc=0 +for attempt in 1 2 3; do + if [ "$attempt" -gt 1 ]; then + echo ":: warning::ssh transport failed, retrying (${attempt}/3)" + sleep $((attempt * 5)) + fi + rc=0 + ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \ + env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \ + "DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \ + "STAGE=$1" bash -se <<'EOF' || rc=$? source "$REPO/.gitea/workflows/deploy-lib.sh" run_stage "$STAGE" EOF + [ "$rc" -eq 0 ] && break + [ "$rc" -ne 255 ] && break +done + +if [ "$rc" -ne 0 ]; then + echo ":: error::stage $1 failed over ssh (exit $rc)" +fi +exit "$rc" diff --git a/userbot/k8s/base/panel.yaml b/userbot/k8s/base/panel.yaml index 8632033..c5f566e 100644 --- a/userbot/k8s/base/panel.yaml +++ b/userbot/k8s/base/panel.yaml @@ -210,6 +210,10 @@ spec: - name: USERBOT_IMAGE # The build pushes main/prod only. The deploy resolves every gcr ref in # this file, so a :latest here aborts the whole apply as unresolvable. + # Deliberately left on the tag: render_pinned only rewrites plain + # `image:` lines to a digest, and this ref is what the panel injects + # into the per-instance Deployments it creates. Instances track prod + # rather than the panel's own resolved digest. value: gcr.forust.xyz/forust/userbot:prod - name: USERBOT_STORAGE_CLASS value: local-path-retain