feat(deploy): serialize apply stages and wait for calm node
apply-k8s and apply-compose share a workstation flock so host docker churn never overlaps cluster churn. Each helm upgrade and the apply loop wait up to 10m for load <28 first, so a deploy never piles onto an already-hot node (the load-40/netbird-death/pending-helm spiral).
This commit is contained in:
1 parent
e56662194b
commit
6f4cd03f4b
2 files changed
+36
-1
No files matched your search
@@ -588,6 +588,28 @@ recover_pending_release() {
|
||||
return 0
|
||||
}
|
||||
|
||||
# wait_for_calm <stage>
|
||||
# The deploy itself is heavy enough to melt this single node (helm churn plus
|
||||
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
|
||||
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
|
||||
# the 1-minute load average to drop below the ceiling, then proceed anyway
|
||||
# with a warning so a permanently busy node cannot wedge the pipeline forever.
|
||||
wait_for_calm() {
|
||||
local load waited=0
|
||||
while [ "$waited" -lt 600 ]; do
|
||||
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
|
||||
if [ "$load" -lt 28 ]; then
|
||||
return 0
|
||||
fi
|
||||
if [ "$((waited % 60))" -eq 0 ]; then
|
||||
log "$1: load $load, waiting for calm (<28)..."
|
||||
fi
|
||||
sleep 15
|
||||
waited=$((waited + 15))
|
||||
done
|
||||
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
|
||||
}
|
||||
|
||||
upgrade_helm_releases() {
|
||||
local entry release chart namespace version values marker repo
|
||||
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||
@@ -608,6 +630,7 @@ upgrade_helm_releases() {
|
||||
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||
log "Upgrading $release ($chart $version)"
|
||||
wait_for_calm "helm $release"
|
||||
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
|
||||
# release in pending-*, which blocks every future upgrade. Recover first
|
||||
# so one wedged revision cannot wedge the pipeline forever.
|
||||
@@ -763,6 +786,7 @@ stage_apply_k8s() {
|
||||
fi
|
||||
fi
|
||||
upgrade_helm_releases
|
||||
wait_for_calm "apply resources"
|
||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
|
||||
for m in "${other_files[@]}"; do
|
||||
|
||||
@@ -36,6 +36,17 @@ ssh_opts=(
|
||||
)
|
||||
|
||||
rc=0
|
||||
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
|
||||
# intact for the verify job, but on a single node they must not run at once:
|
||||
# host docker churn on top of cluster churn is what melts the node (load 40+,
|
||||
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
|
||||
# with a shared lock; whoever arrives second waits.
|
||||
remote_cmd=(bash -se)
|
||||
case "$1" in
|
||||
apply-k8s | apply-compose)
|
||||
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
|
||||
;;
|
||||
esac
|
||||
for attempt in 1 2 3; do
|
||||
if [ "$attempt" -gt 1 ]; then
|
||||
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
|
||||
@@ -45,7 +56,7 @@ for attempt in 1 2 3; do
|
||||
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
|
||||
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
|
||||
"STAGE=$1" bash -se <<'EOF' || rc=$?
|
||||
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
|
||||
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||
run_stage "$STAGE"
|
||||
EOF
|
||||
|
||||
Reference in new issue
Block a user