diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 784a4d4..e89b349 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -588,6 +588,28 @@ recover_pending_release() { return 0 } +# wait_for_calm +# The deploy itself is heavy enough to melt this single node (helm churn plus +# apply churn drove load past 40, killed netbird/ssh, left helm pending-*). +# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for +# the 1-minute load average to drop below the ceiling, then proceed anyway +# with a warning so a permanently busy node cannot wedge the pipeline forever. +wait_for_calm() { + local load waited=0 + while [ "$waited" -lt 600 ]; do + load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)" + if [ "$load" -lt 28 ]; then + return 0 + fi + if [ "$((waited % 60))" -eq 0 ]; then + log "$1: load $load, waiting for calm (<28)..." + fi + sleep 15 + waited=$((waited + 15)) + done + echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway" +} + upgrade_helm_releases() { local entry release chart namespace version values marker repo for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do @@ -608,6 +630,7 @@ upgrade_helm_releases() { helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true helm repo update "${repo%% *}" >/dev/null 2>&1 || true log "Upgrading $release ($chart $version)" + wait_for_calm "helm $release" # A previous run with --rollback-on-failure whose own rollback never finished leaves the # release in pending-*, which blocks every future upgrade. Recover first # so one wedged revision cannot wedge the pipeline forever. @@ -763,6 +786,7 @@ stage_apply_k8s() { fi fi upgrade_helm_releases + wait_for_calm "apply resources" if [ "${#other_files[@]}" -gt 0 ]; then log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index f74f350..b8396cb 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -36,6 +36,17 @@ ssh_opts=( ) rc=0 +# apply-k8s and apply-compose are separate workflow jobs so the graph stays +# intact for the verify job, but on a single node they must not run at once: +# host docker churn on top of cluster churn is what melts the node (load 40+, +# netbird/ssh die, helm is left pending-*). Serialize them on the workstation +# with a shared lock; whoever arrives second waits. +remote_cmd=(bash -se) +case "$1" in + apply-k8s | apply-compose) + remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se) + ;; +esac for attempt in 1 2 3; do if [ "$attempt" -gt 1 ]; then echo ":: warning::ssh transport failed, retrying (${attempt}/3)" @@ -45,7 +56,7 @@ for attempt in 1 2 3; do ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \ env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \ "DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \ - "STAGE=$1" bash -se <<'EOF' || rc=$? + "STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$? source "$REPO/.gitea/workflows/deploy-lib.sh" run_stage "$STAGE" EOF