982 lines
40 KiB
Bash
982 lines
40 KiB
Bash
#!/usr/bin/env bash
|
|
# Workstation deploy stages; invoked by the durable controller against pinned source.
|
|
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
|
|
# source "$REPO/.gitea/workflows/deploy-lib.sh"
|
|
# run_stage "$STAGE"
|
|
# EOF
|
|
set -euo pipefail
|
|
|
|
: "${REPO:?REPO must be set}"
|
|
APPLY_PRUNE="${APPLY_PRUNE:-false}"
|
|
CONFIG_REPO="${CONFIG_REPO:-$REPO}"
|
|
# Exact SHA accepted by the CI gate for both manual and automatic deploys.
|
|
DEPLOY_SHA="${DEPLOY_SHA:-}"
|
|
# Handoff point between the apply stage (writes) and the verify stage (reads).
|
|
# Under the deploy user's own XDG state directory rather than /var/backups: the
|
|
# deploy is unprivileged, /var/backups does not exist on a minimal Arch host, and
|
|
# creating it would need root — which is why the first real deploy died here with
|
|
# "is not writable" before touching a single workload. $HOME comes from sshd.
|
|
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state}/homelab-deploy}"
|
|
# Per-workload rollout budget and how many workloads to watch at once. The whole
|
|
# apply job has its own timeout-minutes as a backstop.
|
|
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
|
|
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-4}"
|
|
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
|
|
|
|
log() {
|
|
echo "== $* =="
|
|
}
|
|
|
|
warn() {
|
|
echo "WARNING: $*" >&2
|
|
}
|
|
|
|
# Prune needs the complete desired set in one invocation. Per-file pruning
|
|
# treats resources from the other files as absent and can delete them.
|
|
check_prune_mode() {
|
|
if [ "$APPLY_PRUNE" = "true" ]; then
|
|
echo "ERROR: APPLY_PRUNE=true is unsupported by the per-file deploy loop." >&2
|
|
echo "Disable it; remove obsolete resources explicitly after review." >&2
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
collect_k8s() {
|
|
git -C "$REPO" ls-files -- "$1" \
|
|
| grep -E '\.ya?ml$' \
|
|
| grep -Ev '/overlays/' \
|
|
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
|
|
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
|
|
| sort
|
|
}
|
|
|
|
kustomize_overlay() {
|
|
if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
|
|
echo "$1/overlays/prod"
|
|
elif [ -f "$1/base/kustomization.yaml" ]; then
|
|
echo "$1/base"
|
|
elif [ -f "$1/kustomization.yaml" ]; then
|
|
echo "$1"
|
|
fi
|
|
}
|
|
|
|
selected_service() {
|
|
local kind="$1" service="$2" section=selected
|
|
[ -n "${DEPLOY_PLAN:-}" ] || return 0
|
|
if [ "${DEPLOY_SMOKE_ALL:-false}" = true ]; then section=active; fi
|
|
jq -e --arg kind "$kind" --arg service "$service" --arg section "$section" \
|
|
'.[$section][$kind] | index($service) != null' "$DEPLOY_PLAN" >/dev/null
|
|
}
|
|
|
|
# Resolve .env and relative binds on the persistent workstation tree. Locked
|
|
# JSON configs keep the same Compose project name and volume names.
|
|
compose() {
|
|
local cf="$1" locked project_dir
|
|
project_dir="$CONFIG_REPO/$(basename "$(dirname "$cf")")"
|
|
shift
|
|
locked="${RUN_DIR:-/nonexistent}/compose/$(basename "$(dirname "$cf")").json"
|
|
if [ -f "$locked" ]; then cf="$locked"; fi
|
|
(cd "$CONFIG_REPO" && docker compose --project-directory "$project_dir" -f "$cf" "$@")
|
|
}
|
|
|
|
select_manifests() {
|
|
K8S_MANIFESTS=()
|
|
KUSTOMIZE_APPS=()
|
|
COMPOSE_STACKS=()
|
|
local kd_rel kd overlay cf_rel cf f
|
|
while IFS= read -r kd_rel; do
|
|
kd="$REPO/$kd_rel"
|
|
selected_service k8s "${kd_rel%/k8s}" || continue
|
|
if [ ! -f "$kd/active" ]; then
|
|
echo "skip (no k8s/active): $kd_rel"
|
|
continue
|
|
fi
|
|
overlay="$(kustomize_overlay "$kd" || true)"
|
|
if [ -n "${overlay:-}" ]; then
|
|
echo "kustomize app: ${overlay#"$REPO"/}"
|
|
KUSTOMIZE_APPS+=("$overlay")
|
|
else
|
|
while IFS= read -r f; do
|
|
[ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f")
|
|
done < <(collect_k8s "$kd_rel" || true)
|
|
fi
|
|
done < <(
|
|
git -C "$REPO" ls-files '*.yaml' '*.yml' \
|
|
| grep -E '(^|/)k8s/' \
|
|
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
|
|
| sort -u
|
|
)
|
|
while IFS= read -r cf_rel; do
|
|
cf="$REPO/$cf_rel"
|
|
selected_service compose "$(dirname "$cf_rel")" || continue
|
|
if [ -f "$(dirname "$cf")/active" ]; then
|
|
echo "compose: $cf_rel"
|
|
COMPOSE_STACKS+=("$cf")
|
|
else
|
|
echo "skip (no root active): $cf_rel"
|
|
fi
|
|
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
|
|
}
|
|
|
|
# --- post-apply verification and rollback -------------------------------------
|
|
#
|
|
# A green `kubectl apply` says nothing about the cluster being healthy. These
|
|
# helpers watch exactly the workloads whose spec changed during this apply, and
|
|
# on failure roll them back to the revision that was running before, so a bad
|
|
# push to main cannot leave a service crash-looping.
|
|
#
|
|
# The workstation controller runs apply and verification as separate durable
|
|
# stages. Runner jobs only follow their logs. ExecStopPost recovers interrupted
|
|
# runs using the per-run snapshot, even when the SSH connection has gone away.
|
|
#
|
|
# Creates this run's snapshot directory and publishes it as the handoff point for
|
|
# the verify stage. Fails hard by design: a deploy that cannot record what it is
|
|
# about to change must not start, because then nothing can be rolled back for it
|
|
# automatically. Publishing happens before the first apply, so an apply killed
|
|
# mid-flight still leaves a usable baseline behind.
|
|
snapshot_dir() {
|
|
local stamp dir
|
|
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
|
|
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
|
|
|
|
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
|
|
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
|
|
echo "The verify job needs it to learn which workloads this deploy touches." >&2
|
|
echo "Refusing to deploy without a way to roll back." >&2
|
|
return 1
|
|
fi
|
|
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
|
|
echo "ERROR: cannot create snapshot dir $dir" >&2
|
|
return 1
|
|
fi
|
|
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
|
|
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
|
|
return 1
|
|
fi
|
|
|
|
printf '%s\n' "$dir"
|
|
}
|
|
|
|
save_snapshot() {
|
|
local dir="$1" releases revision status
|
|
log "Saving pre-apply snapshot to $dir"
|
|
workload_generations >"$dir/generations.before" || return 1
|
|
kubectl get "$WORKLOAD_KINDS" -A -o json >"$dir/workloads.json" || return 1
|
|
kubectl get controllerrevisions.apps -A -o json >"$dir/controller-revisions.json" || return 1
|
|
jq --slurpfile revisions "$dir/controller-revisions.json" '
|
|
[.items[] | . as $w | {
|
|
kind: (.kind | ascii_downcase), namespace: .metadata.namespace, name: .metadata.name, uid: .metadata.uid,
|
|
revision: (if .kind == "Deployment" then (.metadata.annotations["deployment.kubernetes.io/revision"] // "0" | tonumber)
|
|
else ([$revisions[0].items[] | select(.metadata.namespace == $w.metadata.namespace)
|
|
| select(any(.metadata.ownerReferences[]?; .uid == $w.metadata.uid))
|
|
| select($w.kind != "StatefulSet" or .metadata.name == $w.status.currentRevision) | .revision] | max // 0) end)
|
|
}]' "$dir/workloads.json" >"$dir/revisions.json" || return 1
|
|
releases="$(helm list --all -A -o json)" || return 1
|
|
for entry in "${HELM_RELEASES[@]}"; do
|
|
IFS='|' read -r release _ namespace _ _ _ <<<"$entry"
|
|
if ! jq -e --arg r "$release" --arg n "$namespace" \
|
|
'any(.[]; .name == $r and .namespace == $n)' <<<"$releases" >/dev/null; then
|
|
continue
|
|
fi
|
|
helm status "$release" -n "$namespace" -o json >"$dir/helm-$release.json" || return 1
|
|
status="$(jq -r '.info.status' "$dir/helm-$release.json")"
|
|
if [ "$status" != deployed ]; then
|
|
# Never capture a pending/failed revision as the recovery target.
|
|
helm history "$release" -n "$namespace" -o json >"$dir/helm-$release.history.json" || return 1
|
|
revision="$(jq '[.[] | select(.status == "deployed" or .status == "superseded") | .revision] | max // 0' \
|
|
"$dir/helm-$release.history.json")"
|
|
jq --argjson revision "$revision" '.version = $revision' "$dir/helm-$release.json" >"$dir/helm-$release.tmp"
|
|
mv "$dir/helm-$release.tmp" "$dir/helm-$release.json"
|
|
fi
|
|
done
|
|
# The verify stage compares this against the commit it is deploying, to refuse
|
|
# rolling back against a baseline left by an earlier run. A snapshot we cannot
|
|
# attribute to a commit is unusable for that, so fail before anything is applied.
|
|
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
|
|
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
|
|
workload_generations() {
|
|
kubectl get "$WORKLOAD_KINDS" -A \
|
|
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
|
|
--no-headers 2>/dev/null \
|
|
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
|
|
}
|
|
|
|
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
|
|
# moved since the snapshot, i.e. the ones this apply actually touched.
|
|
changed_workloads() {
|
|
local before="$1"
|
|
local ns name kind gen old current
|
|
current="$(workload_generations)" || return 1
|
|
while read -r ns name kind gen; do
|
|
[ -n "${gen:-}" ] || continue
|
|
old="$(awk -v want_ns="$ns" -v want_name="$name" -v want_kind="$kind" \
|
|
'$1 == want_ns && $2 == want_name && $3 == want_kind { print $4; exit }' "$before" 2>/dev/null || true)"
|
|
if [ -n "${RUN_DIR:-}" ] && ! grep -qxF "$kind $ns $name" "$RUN_DIR/workload-refs"; then
|
|
continue
|
|
fi
|
|
if [ "$old" != "$gen" ]; then
|
|
printf '%s %s %s\n' "$kind" "$ns" "$name"
|
|
fi
|
|
done <<<"$current"
|
|
}
|
|
|
|
# Resolve owned image references exclusively from the checked CI artifact.
|
|
render_pinned() {
|
|
python3 "$REPO/.gitea/workflows/release.py" render
|
|
}
|
|
|
|
# verify_workloads <failed-file> <kind> <ns> <name> ...
|
|
# Watches every workload in parallel and records the ones that never became
|
|
# healthy. Returns non-zero if any of them failed.
|
|
verify_workloads() {
|
|
local failed_file="$1"
|
|
shift
|
|
[ "$#" -gt 0 ] || return 0
|
|
: >"$failed_file"
|
|
local running=0 pid kind ns name
|
|
local -a pids=()
|
|
for entry in "$@"; do
|
|
read -r kind ns name <<<"$entry"
|
|
(
|
|
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
|
echo " ok: ${kind}/${ns}/${name}"
|
|
else
|
|
echo " FAILED: ${kind}/${ns}/${name}"
|
|
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
|
|
fi
|
|
) &
|
|
pids+=($!)
|
|
running=$((running + 1))
|
|
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
|
|
wait -n 2>/dev/null || true
|
|
running=$((running - 1))
|
|
fi
|
|
done
|
|
for pid in ${pids[@]+"${pids[@]}"}; do
|
|
wait "$pid" || true
|
|
done
|
|
# Non-zero when the file holds at least one failure, i.e. a workload never
|
|
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
|
|
# every rollback this stage is meant to perform.
|
|
[ ! -s "$failed_file" ]
|
|
}
|
|
|
|
# rollback_workloads <failed-file>
|
|
# Restores the previous revision of every failed workload and waits for it to
|
|
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
|
|
# so the operator knows manual recovery is required.
|
|
rollback_workloads() {
|
|
local failed_file="$1" snapshot kind ns name index=0 running=0 pid revision uid
|
|
local -a pids=()
|
|
snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")"
|
|
while read -r kind ns name; do
|
|
[[ "$kind" =~ ^(deployment|statefulset|daemonset)$ ]] || continue
|
|
index=$((index + 1))
|
|
(
|
|
if kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.annotations}' | grep -q 'meta.helm.sh/release-name'; then
|
|
echo " skip (Helm recovery owns this workload): $kind/$ns/$name"
|
|
exit 1
|
|
fi
|
|
revision="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \
|
|
'.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .revision' "$snapshot/revisions.json")"
|
|
uid="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \
|
|
'.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .uid' "$snapshot/revisions.json")"
|
|
if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]] || [ "$uid" != "$(kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.uid}')" ]; then
|
|
echo " no safe previous revision: $kind/$ns/$name (new or replaced workload)"
|
|
exit 1
|
|
fi
|
|
kubectl rollout undo "$kind/$name" -n "$ns" --to-revision="$revision" \
|
|
&& kubectl rollout status "$kind/$name" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s"
|
|
) >"$snapshot/rollback-$index.log" 2>&1 &
|
|
pids+=($!)
|
|
running=$((running + 1))
|
|
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
|
|
wait -n 2>/dev/null || true
|
|
running=$((running - 1))
|
|
fi
|
|
done <"$failed_file"
|
|
local recovered=0 unrecovered=0 i=0
|
|
for pid in "${pids[@]}"; do
|
|
i=$((i + 1))
|
|
if wait "$pid"; then recovered=$((recovered + 1)); else unrecovered=$((unrecovered + 1)); fi
|
|
cat "$snapshot/rollback-$i.log"
|
|
done
|
|
echo "ROLLED_BACK=$recovered" >>"$failed_file"
|
|
echo "UNRECOVERED=$unrecovered" >>"$failed_file"
|
|
[ "$unrecovered" -eq 0 ]
|
|
}
|
|
|
|
# Helm releases owned by this stage, one line each:
|
|
#
|
|
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
|
|
#
|
|
# The chart version is the field Renovate keeps current. The helmv3 manager only
|
|
# understands Chart.yaml and the helm-values manager only values files, so a pin
|
|
# written straight into a `helm upgrade` command would never be updated: these
|
|
# have to be declared as custom.regex managers in renovate/renovate.json.
|
|
HELM_RELEASES=(
|
|
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
|
|
"victoria-operator|victoriametrics/victoria-metrics-operator|prometheus|0.68.1|prometheus-stack/k8s/victoria-operator-values.yaml|prometheus-stack/k8s/active"
|
|
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
|
|
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
|
|
"reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active"
|
|
)
|
|
|
|
# "name url" for the Helm repository hosting a chart, empty if unknown.
|
|
helm_repo_for() {
|
|
case "$1" in
|
|
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
|
|
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
|
|
stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;;
|
|
victoriametrics/*) echo "victoriametrics https://victoriametrics.github.io/helm-charts" ;;
|
|
esac
|
|
}
|
|
|
|
# helm_release_status <release> <namespace>
|
|
# Prints the release status in lowercase (deployed, failed, pending-rollback,
|
|
# ...) or "not-found" when the release does not exist yet.
|
|
helm_release_status() {
|
|
local out
|
|
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
|
|
if [[ "$out" == *"release: not found"* ]]; then
|
|
echo "not-found"
|
|
return 0
|
|
fi
|
|
printf 'ERROR: cannot read Helm status: %s\n' "$out" >&2
|
|
return 1
|
|
fi
|
|
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
|
|
}
|
|
|
|
# recover_pending_release <release> <namespace>
|
|
# Rolls a release out of a pending-* state left by a failed upgrade with --rollback-on-failure
|
|
# whose own rollback never completed. Without this every future upgrade errors
|
|
# out until a human runs `helm rollback`. Passes through releases that are not
|
|
# pending (deployed, failed, not-found). Returns non-zero when the release is
|
|
# still not recoverable, so the pipeline fails loud instead of wedging.
|
|
recover_pending_release() {
|
|
local release="$1" namespace="$2" status revision snapshot
|
|
status="$(helm_release_status "$release" "$namespace")" || return 1
|
|
case "$status" in
|
|
pending-upgrade|pending-rollback|pending-install)
|
|
log "Release $release is $status, rolling back to the last deployed revision"
|
|
revision=""
|
|
if [ -s "$DEPLOY_SNAPSHOT_DIR/current" ]; then
|
|
snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")"
|
|
if [ -s "$snapshot/helm-$release.json" ]; then
|
|
revision="$(jq -r '.version' "$snapshot/helm-$release.json")"
|
|
fi
|
|
fi
|
|
if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]]; then
|
|
echo "ERROR: no captured Helm revision for $release; manual recovery required"
|
|
return 1
|
|
fi
|
|
if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then
|
|
echo "WARN: helm rollback of $release did not complete"
|
|
return 1
|
|
fi
|
|
status="$(helm_release_status "$release" "$namespace")" || return 1
|
|
if [ "$status" != "deployed" ]; then
|
|
echo "WARN: $release is $status after rollback"
|
|
return 1
|
|
fi
|
|
;;
|
|
esac
|
|
return 0
|
|
}
|
|
|
|
# wait_for_calm <stage>
|
|
# The deploy itself is heavy enough to melt this single node (helm churn plus
|
|
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
|
|
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
|
|
# the 1-minute load average to drop below the ceiling, then proceed anyway
|
|
# with a warning so a permanently busy node cannot wedge the pipeline forever.
|
|
wait_for_calm() {
|
|
local load waited=0
|
|
while [ "$waited" -lt 600 ]; do
|
|
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
|
|
if [ "$load" -lt 28 ]; then
|
|
return 0
|
|
fi
|
|
if [ "$((waited % 60))" -eq 0 ]; then
|
|
log "$1: load $load, waiting for calm (<28)..."
|
|
fi
|
|
sleep 15
|
|
waited=$((waited + 15))
|
|
done
|
|
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
|
|
}
|
|
|
|
upgrade_helm_releases() {
|
|
local entry release chart namespace version values marker repo
|
|
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
|
IFS='|' read -r release chart namespace version values marker <<<"$entry"
|
|
if [ -n "${DEPLOY_PLAN:-}" ] && ! jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null; then
|
|
echo "skip (unchanged Helm release): $release"
|
|
continue
|
|
fi
|
|
if [ ! -f "$REPO/$values" ] && [ -f "$CONFIG_REPO/$values" ]; then values="$CONFIG_REPO/$values"; else values="$REPO/$values"; fi
|
|
if [ ! -f "$REPO/$marker" ]; then
|
|
echo "skip (no $marker): $release"
|
|
continue
|
|
fi
|
|
if [ ! -f "$values" ]; then
|
|
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
|
|
return 1
|
|
fi
|
|
repo="$(helm_repo_for "$chart")"
|
|
if [ -z "$repo" ]; then
|
|
echo "ERROR: no Helm repository configured for chart $chart"
|
|
return 1
|
|
fi
|
|
helm repo add "${repo%% *}" "${repo#* }" >/dev/null
|
|
helm repo update "${repo%% *}" >/dev/null
|
|
log "Upgrading $release ($chart $version)"
|
|
wait_for_calm "helm $release"
|
|
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
|
|
# release in pending-*, which blocks every future upgrade. Recover first
|
|
# so one wedged revision cannot wedge the pipeline forever.
|
|
if ! recover_pending_release "$release" "$namespace"; then
|
|
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
|
|
return 1
|
|
fi
|
|
# --rollback-on-failure (+ --wait) rolls the release back when the upgrade
|
|
# times out or the workloads it touches never become ready, so a bad chart
|
|
# bump is not left half applied. (--atomic was this combo; deprecated.)
|
|
if ! helm upgrade --install "$release" "$chart" \
|
|
--namespace "$namespace" \
|
|
--version "$version" \
|
|
--values "$values" \
|
|
--wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then
|
|
echo "WARN: upgrade of $release failed, checking release state"
|
|
# --rollback-on-failure already attempted its own rollback; finish the job when that
|
|
# rollback never completed, otherwise the release stays pending-* and
|
|
# blocks every future run.
|
|
if ! recover_pending_release "$release" "$namespace"; then
|
|
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
|
|
else
|
|
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
|
|
fi
|
|
return 1
|
|
fi
|
|
done
|
|
}
|
|
|
|
stage_doctor() {
|
|
local tool entry release chart namespace version values marker
|
|
for tool in git docker kubectl helm jq curl timeout flock python3; do
|
|
command -v "$tool" >/dev/null || { echo "Missing workstation tool: $tool"; return 1; }
|
|
done
|
|
docker compose version >/dev/null
|
|
docker buildx version >/dev/null
|
|
[ "$(kubectl config current-context)" = "${KUBE_CONTEXT:?configure KUBE_CONTEXT}" ] || { echo "Unexpected Kubernetes context"; return 1; }
|
|
[ "$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" = "${EXPECTED_CLUSTER_UID:?configure EXPECTED_CLUSTER_UID}" ] || { echo "Unexpected Kubernetes cluster"; return 1; }
|
|
kubectl get --raw=/readyz --request-timeout=10s >/dev/null
|
|
[ "$(git -C "$REPO" rev-parse HEAD)" = "$DEPLOY_SHA" ] || return 1
|
|
select_manifests
|
|
for entry in "${HELM_RELEASES[@]}"; do
|
|
IFS='|' read -r release chart namespace version values marker <<<"$entry"
|
|
[ -f "$REPO/$marker" ] || continue
|
|
[ -f "$REPO/$values" ] || [ -f "$CONFIG_REPO/$values" ] || { echo "Missing values: $values"; return 1; }
|
|
done
|
|
jq '{sha, selected, helm, removed}' "$DEPLOY_PLAN"
|
|
local cf
|
|
for cf in "${COMPOSE_STACKS[@]}"; do
|
|
compose "$cf" config --quiet
|
|
while IFS= read -r network; do
|
|
docker network inspect "$network" >/dev/null || return 1
|
|
done < <(compose "$cf" config --format json | jq -r '.networks // {} | to_entries[] | select(.value.external == true) | .value.name')
|
|
python3 "$REPO/.gitea/workflows/compose-release.py" "$cf"
|
|
done
|
|
local image refs m k
|
|
refs="$(
|
|
for m in "${K8S_MANIFESTS[@]}"; do render_pinned <"$m" || return 1; done
|
|
for k in "${KUSTOMIZE_APPS[@]}"; do kubectl kustomize "$k" | render_pinned || return 1; done
|
|
)" || return 1
|
|
refs="$(grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+@sha256:[0-9a-f]{64}' <<<"$refs" | sort -u || true)"
|
|
while IFS= read -r image; do
|
|
[ -n "$image" ] || continue
|
|
timeout 60s docker buildx imagetools inspect "$image" >/dev/null
|
|
done <<<"$refs"
|
|
}
|
|
|
|
# Required pod Secrets, scoped to the resource namespace. TLS route Secrets are
|
|
# created by cert-manager and are not prerequisites for applying a Certificate.
|
|
check_referenced_secrets() {
|
|
local m k objects refs extracted ns name
|
|
local missing=()
|
|
refs=""
|
|
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
|
if skip_uninstalled_vmagent_crd "$m"; then
|
|
continue
|
|
fi
|
|
objects="$(kubectl create --dry-run=client --validate=false -f "$m" -o json)" || return 1
|
|
extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1
|
|
refs+="$extracted"$'\n'
|
|
done
|
|
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
|
objects="$(kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json)" || return 1
|
|
extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1
|
|
refs+="$extracted"$'\n'
|
|
done
|
|
while read -r ns name; do
|
|
[ -n "${name:-}" ] || continue
|
|
if kubectl get secret "$name" -n "$ns" -o name >/dev/null 2>&1; then
|
|
echo " ok: $ns/$name"
|
|
else
|
|
echo " MISSING OR UNREADABLE: $ns/$name"
|
|
missing+=("$ns/$name")
|
|
fi
|
|
done < <(printf '%s' "$refs" | sort -u)
|
|
if [ "${#missing[@]}" -gt 0 ]; then
|
|
echo "ERROR: required pod Secrets are missing or unreadable:"
|
|
printf ' - %s\n' "${missing[@]}"
|
|
echo "Create them in the listed namespaces from the service's secret example."
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
# The VMAgent CRD is installed by the VictoriaMetrics Operator Helm release in
|
|
# stage_apply_k8s, after this preflight stage. Skip only its dry-run until then.
|
|
skip_uninstalled_vmagent_crd() {
|
|
local manifest="$1"
|
|
if [[ "$manifest" == "$REPO/prometheus-stack/k8s/vmagent.yaml" ]] \
|
|
&& ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then
|
|
echo " skip: VMAgent CRD is installed by Helm during apply: ${manifest#"$REPO"/}"
|
|
return 0
|
|
fi
|
|
return 1
|
|
}
|
|
|
|
stage_validate() {
|
|
check_prune_mode || return 1
|
|
cd "$REPO"
|
|
select_manifests
|
|
local m k cf
|
|
# The deploy host has the local .env and secret files. Resolve them here so
|
|
# missing configuration fails before either apply job changes workloads.
|
|
# CI keeps the structure-only check for inactive stacks.
|
|
# shellcheck source=compose-lint.sh
|
|
source "$REPO/.gitea/workflows/compose-lint.sh"
|
|
log "Validate compose stacks"
|
|
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
|
echo " config: $cf"
|
|
compose "$cf" config --quiet
|
|
done
|
|
log "Validate k8s manifests (kubectl dry-run=client)"
|
|
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
|
if skip_uninstalled_vmagent_crd "$m"; then
|
|
continue
|
|
fi
|
|
kubectl apply --dry-run=client -f "$m" >/dev/null
|
|
done
|
|
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
|
kubectl apply -k "$k" --dry-run=client >/dev/null
|
|
done
|
|
log "Validate k8s manifests (kubectl dry-run=server)"
|
|
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
|
if skip_uninstalled_vmagent_crd "$m"; then
|
|
continue
|
|
fi
|
|
kubectl apply --dry-run=server -f "$m" >/dev/null
|
|
done
|
|
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
|
kubectl apply -k "$k" --dry-run=server >/dev/null
|
|
done
|
|
log "Checking referenced Secrets exist"
|
|
echo " (deploy never applies *secret*.yaml; create missing ones manually)"
|
|
check_referenced_secrets
|
|
}
|
|
|
|
selected_workload_refs() {
|
|
local m k
|
|
for m in "${K8S_MANIFESTS[@]}"; do
|
|
if skip_uninstalled_vmagent_crd "$m" >/dev/null; then continue; fi
|
|
kubectl create --dry-run=client --validate=false -f "$m" -o json | jq -r '
|
|
(if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
|
|
| "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"'
|
|
done
|
|
for k in "${KUSTOMIZE_APPS[@]}"; do
|
|
kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json | jq -r '
|
|
(if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
|
|
| "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"'
|
|
done
|
|
}
|
|
|
|
stage_apply_k8s() {
|
|
check_prune_mode || return 1
|
|
cd "$REPO"
|
|
select_manifests >/dev/null
|
|
local ns_files=() other_files=() m k
|
|
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
|
case "$m" in
|
|
*/namespace.yaml|*/namespace.yml) ns_files+=("$m") ;;
|
|
*) other_files+=("$m") ;;
|
|
esac
|
|
done
|
|
|
|
# Record what is about to change, and publish it for the verify job, before
|
|
# the first apply. Both are fatal on failure: see snapshot_dir.
|
|
selected_workload_refs >"$RUN_DIR/workload-refs"
|
|
local snapshot
|
|
snapshot="$(snapshot_dir)" || return 1
|
|
save_snapshot "$snapshot" || return 1
|
|
touch "$snapshot/ready"
|
|
|
|
if [ "${#ns_files[@]}" -gt 0 ]; then
|
|
log "Applying namespaces (${#ns_files[@]} files)"
|
|
for m in "${ns_files[@]}"; do
|
|
kubectl apply -f "$m"
|
|
done
|
|
fi
|
|
if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then
|
|
if [ ! -f "$CONFIG_REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then
|
|
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
|
|
exit 1
|
|
fi
|
|
fi
|
|
upgrade_helm_releases
|
|
wait_for_calm "apply resources"
|
|
if [ "${#other_files[@]}" -gt 0 ]; then
|
|
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
|
|
for m in "${other_files[@]}"; do
|
|
log "Applying ${m#"$REPO"/}"
|
|
if ! render_pinned <"$m" | kubectl apply -f -; then
|
|
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
|
|
exit 1
|
|
fi
|
|
done
|
|
fi
|
|
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
|
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
|
|
if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then
|
|
echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2
|
|
exit 1
|
|
fi
|
|
done
|
|
|
|
# No verification here on purpose. This stage may be killed at any point by
|
|
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
|
|
# connection, and any code below that line would simply not run. stage_verify_k8s
|
|
# picks the work up from the snapshot instead.
|
|
log "Applied. Verification and rollback are the verify job's job, not this one's."
|
|
}
|
|
|
|
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
|
|
# including when they failed, timed out or were cancelled. Reads the baseline the
|
|
# apply stage published and works out what it changed, watches those workloads,
|
|
# and rolls back the ones that never became healthy.
|
|
stage_verify_k8s() {
|
|
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
|
|
local snapshot want have generations changed
|
|
local -a touched=()
|
|
|
|
if [ ! -s "$pointer" ]; then
|
|
echo "ERROR: no snapshot pointer at $pointer."
|
|
echo "The apply stage died before publishing any state, so there is no baseline to"
|
|
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
|
|
echo "inspect the cluster by hand."
|
|
return 1
|
|
fi
|
|
snapshot="$(head -1 "$pointer")"
|
|
if [ ! -d "$snapshot" ] || [ ! -f "$snapshot/ready" ]; then
|
|
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
|
|
return 1
|
|
fi
|
|
|
|
# Never trust the pointer blindly. If the apply stage was killed before it
|
|
# published its own snapshot, `current` still points at the previous deploy's
|
|
# baseline. Verifying against that would watch the wrong workloads and the
|
|
# rollback would revert the wrong revisions, so refuse instead.
|
|
want="${DEPLOY_SHA:-}"
|
|
if [ -z "$want" ]; then
|
|
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
|
|
fi
|
|
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
|
|
if [ -z "$want" ] || [ "$have" != "$want" ]; then
|
|
echo "ERROR: refusing to verify or roll back against a stale snapshot."
|
|
echo " snapshot: $snapshot"
|
|
echo " snapshot commit: ${have:-<missing>}"
|
|
echo " deploy commit: ${want:-<unknown>}"
|
|
return 1
|
|
fi
|
|
echo " snapshot: $snapshot (commit ${have:0:12})"
|
|
|
|
local entry release chart namespace version values marker
|
|
for entry in "${HELM_RELEASES[@]}"; do
|
|
IFS='|' read -r release chart namespace version values marker <<<"$entry"
|
|
jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null || continue
|
|
recover_pending_release "$release" "$namespace" || return 1
|
|
done
|
|
|
|
generations="$snapshot/generations.before"
|
|
if [ ! -s "$generations" ]; then
|
|
# Without a baseline we cannot tell which workloads the apply touched, so
|
|
# fall back to watching everything rather than silently skipping the check.
|
|
warn "no pre-apply baseline, verifying every workload in the cluster"
|
|
: >"$generations"
|
|
fi
|
|
|
|
changed="$(changed_workloads "$generations")" || return 1
|
|
while read -r kind ns name; do
|
|
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
|
|
done <<<"$changed"
|
|
|
|
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
|
|
if [ "${#touched[@]}" -eq 0 ]; then
|
|
echo " nothing to verify"
|
|
return 0
|
|
fi
|
|
printf ' watching: %s\n' "${touched[@]/#/ }"
|
|
|
|
local failed_file="$snapshot/failed-workloads"
|
|
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
|
|
echo
|
|
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
|
|
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
|
|
echo
|
|
log "Rolling back to the previous revision"
|
|
if rollback_workloads "$failed_file"; then
|
|
echo
|
|
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
|
|
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
|
|
echo "externally created resources from this commit are still in place. Review the failed"
|
|
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
|
|
else
|
|
echo
|
|
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
|
|
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
|
|
echo "Pre-apply snapshot: $snapshot"
|
|
fi
|
|
return 1
|
|
fi
|
|
}
|
|
|
|
# verify_compose_stack <compose-file>
|
|
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
|
|
# come back broken with a green pipeline. Require every long-running service to
|
|
# actually be running.
|
|
verify_compose_stack() {
|
|
local cf="$1"
|
|
local expected running missing=()
|
|
expected="$(compose "$cf" config --format json | jq -r ' .services | to_entries[] | select(.value.restart != "no") | .key' | sort)" || return 1
|
|
running="$(compose "$cf" ps --status running --services | sort)" || return 1
|
|
[ -n "$expected" ] || return 0
|
|
while IFS= read -r svc; do
|
|
[ -n "$svc" ] || continue
|
|
# restart:"no" services are allowed to have exited.
|
|
if ! printf '%s\n' "$running" | grep -qx "$svc"; then
|
|
missing+=("$svc")
|
|
fi
|
|
done <<<"$expected"
|
|
if [ "${#missing[@]}" -gt 0 ]; then
|
|
echo " NOT RUNNING: ${missing[*]}"
|
|
compose "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
|
|
return 1
|
|
fi
|
|
echo " all ${#expected} service(s) running"
|
|
return 0
|
|
}
|
|
|
|
# The public hostname of every active service, one per line.
|
|
#
|
|
# Comments are stripped first, and deliberately so: a route that someone
|
|
# disabled by commenting it out is not a service to probe, and naio and xui are
|
|
# both still in the tree that way. A `#` only starts a comment when it is at the
|
|
# start of a line or after whitespace, so `s/#.*//` alone would also cut a
|
|
# legitimate value in half.
|
|
#
|
|
# Only the public names. The *.internal names are the same Traefik and the same
|
|
# Services, reached by a different label, so probing both would double the run
|
|
# to learn the same thing. The public name is also the one a user types.
|
|
smoke_hosts() {
|
|
local m k
|
|
# The backticks below are literal. They are Traefik's Host() delimiter, and the
|
|
# single quotes are precisely what keeps the shell from reading them as a
|
|
# command substitution, so the warning is the opposite of a real problem.
|
|
# shellcheck disable=SC2016
|
|
{
|
|
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
|
[ -f "$m" ] && cat "$m"
|
|
done
|
|
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
|
kubectl kustomize "$k" 2>/dev/null || true
|
|
done
|
|
} | sed -E 's/(^|[[:space:]])#.*$//' \
|
|
| grep -oE 'Host\(`[^`]+`\)' \
|
|
| sed -E 's/^Host\(`//; s/`\)$//' \
|
|
| grep -E '(^|\.)forust\.xyz$' \
|
|
| grep -v '\${' \
|
|
| sort -u
|
|
}
|
|
|
|
# Traefik's own list of the routes it actually built. The Kubernetes CRs are the
|
|
# wrong source for this: when a middleware fails to load, Traefik drops the
|
|
# router that referenced it and leaves the CR behind looking perfectly healthy.
|
|
#
|
|
# api.insecure is already on for the internal `traefik` entrypoint, but the pod
|
|
# IP is not routable from the node, so read it through kubectl exec rather than
|
|
# standing up a port-forward. HTTP only: the TCP routers match on HostSNI(`*`)
|
|
# and the UDP ones carry no rule at all, both selected by entrypoint and port,
|
|
# so neither can answer whether a given host has a route.
|
|
traefik_http_routes() {
|
|
kubectl -n traefik exec deploy/traefik -- \
|
|
wget -qO- --timeout=10 http://127.0.0.1:8080/api/http/routers 2>/dev/null \
|
|
| jq -c '[.[] | {status, rule: (.rule // "")}]'
|
|
}
|
|
|
|
# The hosts Traefik currently routes to, one per line. Every backticked token of
|
|
# an enabled rule counts, which is a superset of the hosts -- PathPrefix values
|
|
# land here too, harmlessly -- but it keeps the host syntax in one place instead
|
|
# of a matcher per host. The scan keeps the delimiters, so strip them: what
|
|
# belongs in a comparison against a hostname is the bare name.
|
|
traefik_routed_hosts() {
|
|
jq -r '[.[] | select(.status == "enabled") | (.rule // "")
|
|
| scan("`[^`]+`") | ltrimstr("`") | rtrimstr("`")]
|
|
| unique | .[]' <<<"$1"
|
|
}
|
|
|
|
# stage_verify_k8s watches the rollout, which reports that the pods converged.
|
|
# It cannot tell a converged pod from a serving one: a route pointing at the
|
|
# wrong port, a Service selector that matches nothing the app listens on, a 500
|
|
# from the app itself, an OOMKill loop that still counts as Available for long
|
|
# enough to pass. All of those are green at the rollout level.
|
|
#
|
|
# So ask the thing users ask. Any HTTP response proves Traefik matched the
|
|
# host, the Service resolved to a pod and the pod answered -- a 302 to a login
|
|
# or a 404 from a path the service does not serve still means the chain is
|
|
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
|
|
# the service is not serving, and only those fail the run.
|
|
#
|
|
# Except that a 404 is not evidence on its own. A router Traefik refused to
|
|
# build answers with the same 404 and nothing behind it, so a middleware that
|
|
# fails to load takes down every route that referenced it while
|
|
# this stage reports `ok` for all of them. No status code separates those two
|
|
# cases, so ask Traefik which routes it built and fail on the difference.
|
|
stage_smoke() {
|
|
cd "$REPO"
|
|
if [ -n "${DEPLOY_PLAN:-}" ] && jq -e '.full_smoke' "$DEPLOY_PLAN" >/dev/null; then
|
|
DEPLOY_SMOKE_ALL=true
|
|
fi
|
|
select_manifests >/dev/null
|
|
local -a hosts=()
|
|
local h code rc bad=0
|
|
while IFS= read -r h; do
|
|
[ -n "$h" ] && hosts+=("$h")
|
|
done < <(smoke_hosts)
|
|
|
|
if [ "${#hosts[@]}" -eq 0 ]; then
|
|
# Nothing to probe means the extraction broke, not that the cluster is empty.
|
|
echo "No public routes in the selected components"
|
|
return 0
|
|
fi
|
|
|
|
log "Probing ${#hosts[@]} public route(s)"
|
|
for h in "${hosts[@]}"; do
|
|
code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$?
|
|
if [ "$rc" -ne 0 ]; then
|
|
echo " UNREACHABLE $h (curl exit $rc)"
|
|
bad=1
|
|
continue
|
|
fi
|
|
# A glob, not a string compare. `case` on the leading digit is the only one
|
|
# of these that survives a three-digit code, and the obvious expansion to
|
|
# try first -- ${code%%[0-9]*} -- is empty for every input, so it silently
|
|
# reports a 500 as healthy.
|
|
case "$code" in
|
|
5*)
|
|
echo " SERVER ERROR $h $code"
|
|
bad=1
|
|
;;
|
|
000)
|
|
# curl exited 0 and still no status, so nothing on the far end replied.
|
|
# Not a pass, whatever the transport thought.
|
|
echo " NO RESPONSE $h"
|
|
bad=1
|
|
;;
|
|
*)
|
|
echo " ok $h $code"
|
|
;;
|
|
esac
|
|
done
|
|
|
|
# Second gate. The probe above only means something if a router matched the
|
|
# host in the first place, so compare the hosts we expect against the hosts
|
|
# Traefik reports and fail on the difference.
|
|
local routes routed
|
|
if ! routes="$(traefik_http_routes)"; then
|
|
echo "ERROR: could not read Traefik's router list, refusing to report success"
|
|
return 1
|
|
fi
|
|
routed="$(traefik_routed_hosts "$routes")"
|
|
|
|
local -a unrouted=()
|
|
local tries=3
|
|
while :; do
|
|
unrouted=()
|
|
for h in "${hosts[@]}"; do
|
|
grep -qxF "$h" <<<"$routed" || unrouted+=("$h")
|
|
done
|
|
if [ "${#unrouted[@]}" -eq 0 ]; then
|
|
break
|
|
fi
|
|
# A router mid-rollout is legitimately absent for a moment. A middleware
|
|
# that failed to load stays absent, so waiting cannot paper over it.
|
|
if [ "$tries" -le 1 ]; then
|
|
break
|
|
fi
|
|
tries=$((tries - 1))
|
|
warn "${#unrouted[@]} host(s) have no enabled route yet, re-checking in 10s"
|
|
sleep 10
|
|
if ! routes="$(traefik_http_routes)"; then
|
|
break
|
|
fi
|
|
routed="$(traefik_routed_hosts "$routes")"
|
|
done
|
|
|
|
if [ "${#unrouted[@]}" -ne 0 ]; then
|
|
for h in "${unrouted[@]}"; do
|
|
echo " NO ROUTE $h (Traefik has no enabled router for this host)"
|
|
done
|
|
bad=1
|
|
fi
|
|
|
|
if [ "$bad" -ne 0 ]; then
|
|
echo "ERROR: at least one active service is not serving over its public route"
|
|
return 1
|
|
fi
|
|
echo "all ${#hosts[@]} route(s) answered and have a router"
|
|
}
|
|
|
|
stage_apply_compose() {
|
|
cd "$REPO"
|
|
select_manifests >/dev/null
|
|
local cf
|
|
for cf in "${COMPOSE_STACKS[@]}"; do
|
|
log "Applying Compose ${cf#"$REPO"/}"
|
|
compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans
|
|
verify_compose_stack "$cf"
|
|
done
|
|
echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)"
|
|
}
|
|
|
|
run_stage() {
|
|
case "${1:?stage required}" in
|
|
doctor) stage_doctor ;;
|
|
validate) stage_validate ;;
|
|
apply-k8s) stage_apply_k8s ;;
|
|
verify-k8s) stage_verify_k8s ;;
|
|
smoke) stage_smoke ;;
|
|
apply-compose) stage_apply_compose ;;
|
|
*)
|
|
echo "ERROR: unknown stage: $1"
|
|
exit 1
|
|
;;
|
|
esac
|
|
}
|