Files
homelab/.gitea/workflows/deploy-lib.sh
T
forust 1505b638ce fix(deploy): verify and roll back in a separate job
verify_workloads ended on `[ -s "$failed_file" ]`, which is the opposite
of what its own contract says. A non-empty file means something failed, so
the function returned success exactly when a workload never came up, and
failure when everything was fine. Every rollback was therefore skipped,
and every deploy that changed anything ended red with an empty failure
list and a bogus "Rolled back successfully".

Worse, the check only ever ran at the end of stage_apply_k8s, inside the
same process as the apply. A job killed by timeout-minutes, cancelled by
a new push, or cut off by a dropped SSH connection never reached it, which
is precisely when a rollback matters. The three helm upgrades alone can
consume the whole 30-minute job budget, so that path was reachable.

Verification now lives in its own job, gated on always(), so it runs
whatever happened to the apply. The apply stage publishes its pre-apply
snapshot through DEPLOY_SNAPSHOT_DIR/current before touching anything,
and the verify stage picks it up from there. A snapshot whose recorded
commit does not match the deploy is refused rather than trusted, so a
stale pointer from an earlier run cannot make the rollback revert the
wrong workloads. An unwritable snapshot directory now fails the deploy up
front instead of silently continuing without a way back.

cancel-in-progress becomes false for the same reason: cancelling a run
kills the apply job and takes the verify job with it, which is the failure
this change exists to prevent. Both applies are idempotent, so queueing
costs little. The SSH key moves to a per-run directory removed on exit,
and the deploy is pinned to the exact commit CI validated.
2026-09-26 20:09:20 +02:00

602 lines
23 KiB
Bash

#!/usr/bin/env bash
# Shared stages for the deploy workflow. Runs on the workstation, invoked as:
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
# source "$REPO/.gitea/workflows/deploy-lib.sh"
# run_stage "$STAGE"
# EOF
set -euo pipefail
: "${REPO:?REPO must be set}"
APPLY_PRUNE="${APPLY_PRUNE:-false}"
# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to
# the current origin/main.
DEPLOY_SHA="${DEPLOY_SHA:-}"
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-/var/backups/homelab-deploy}"
# Per-workload rollout budget and how many workloads to watch at once. The whole
# apply job has its own timeout-minutes as a backstop.
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}"
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
log() {
echo "== $* =="
}
warn() {
echo "WARNING: $*" >&2
}
collect_k8s() {
git -C "$REPO" ls-files -- "$1" \
| grep -E '\.ya?ml$' \
| grep -Ev '/overlays/' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
| sort
}
kustomize_overlay() {
if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
echo "$1/overlays/prod"
elif [ -f "$1/base/kustomization.yaml" ]; then
echo "$1/base"
elif [ -f "$1/kustomization.yaml" ]; then
echo "$1"
fi
}
select_manifests() {
K8S_MANIFESTS=()
KUSTOMIZE_APPS=()
COMPOSE_STACKS=()
local kd_rel kd overlay cf_rel cf f
while IFS= read -r kd_rel; do
kd="$REPO/$kd_rel"
if [ ! -f "$kd/active" ]; then
echo "skip (no k8s/active): $kd_rel"
continue
fi
overlay="$(kustomize_overlay "$kd" || true)"
if [ -n "${overlay:-}" ]; then
echo "kustomize app: ${overlay#"$REPO"/}"
KUSTOMIZE_APPS+=("$overlay")
else
while IFS= read -r f; do
[ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f")
done < <(collect_k8s "$kd_rel" || true)
fi
done < <(
git -C "$REPO" ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
while IFS= read -r cf_rel; do
cf="$REPO/$cf_rel"
if [ -f "$(dirname "$cf")/active" ]; then
echo "compose: $cf_rel"
COMPOSE_STACKS+=("$cf")
else
echo "skip (no root active): $cf_rel"
fi
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
}
# --- post-apply verification and rollback -------------------------------------
#
# A green `kubectl apply` says nothing about the cluster being healthy. These
# helpers watch exactly the workloads whose spec changed during this apply, and
# on failure roll them back to the revision that was running before, so a bad
# push to main cannot leave a service crash-looping.
#
# Verification lives in its own workflow job, not at the end of the apply stage.
# Inside a single process it is worthless exactly when it is needed most: a job
# killed by timeout-minutes or cancelled mid-apply never reaches the rollback
# code, and leaves a half-applied cluster behind. Split out, the apply job can
# die in any way and the verify job still runs.
#
# That split needs a handoff point on the workstation, because the two stages are
# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written
# before anything is applied, read by the verify stage afterwards.
# Creates this run's snapshot directory and publishes it as the handoff point for
# the verify stage. Fails hard by design: a deploy that cannot record what it is
# about to change must not start, because then nothing can be rolled back for it
# automatically. Publishing happens before the first apply, so an apply killed
# mid-flight still leaves a usable baseline behind.
snapshot_dir() {
local stamp dir
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
echo "The verify job needs it to learn which workloads this deploy touches." >&2
echo "Refusing to deploy without a way to roll back." >&2
return 1
fi
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
echo "ERROR: cannot create snapshot dir $dir" >&2
return 1
fi
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
return 1
fi
printf '%s\n' "$dir"
}
save_snapshot() {
local dir="$1"
log "Saving pre-apply snapshot to $dir"
workload_generations >"$dir/generations.before" 2>/dev/null \
|| warn "could not snapshot workload generations"
kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \
|| warn "could not snapshot workloads"
for release in prometheus-stack loki alloy; do
if helm status "$release" -n prometheus >/dev/null 2>&1; then
{
echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)"
helm get values "$release" -n prometheus --all 2>/dev/null
} >"$dir/helm-$release.txt"
fi
done
# The verify stage compares this against the commit it is deploying, to refuse
# rolling back against a baseline left by an earlier run. A snapshot we cannot
# attribute to a commit is unusable for that, so fail before anything is applied.
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
return 1
fi
}
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
workload_generations() {
kubectl get "$WORKLOAD_KINDS" -A \
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
--no-headers 2>/dev/null \
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
}
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
# moved since the snapshot, i.e. the ones this apply actually touched.
changed_workloads() {
local before="$1"
local ns name kind gen old
while read -r ns name kind gen; do
[ -n "${gen:-}" ] || continue
old="$(awk -v want_ns="$ns" -v want_name="$name" \
'$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)"
if [ "$old" != "$gen" ]; then
printf '%s %s %s\n' "$kind" "$ns" "$name"
fi
done < <(workload_generations)
}
# verify_workloads <failed-file> <kind> <ns> <name> ...
# Watches every workload in parallel and records the ones that never became
# healthy. Returns non-zero if any of them failed.
verify_workloads() {
local failed_file="$1"
shift
[ "$#" -gt 0 ] || return 0
: >"$failed_file"
local running=0 pid kind ns name
local -a pids=()
for entry in "$@"; do
read -r kind ns name <<<"$entry"
(
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " ok: ${kind}/${ns}/${name}"
else
echo " FAILED: ${kind}/${ns}/${name}"
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
fi
) &
pids+=($!)
running=$((running + 1))
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
wait -n 2>/dev/null || true
running=$((running - 1))
fi
done
for pid in ${pids[@]+"${pids[@]}"}; do
wait "$pid" || true
done
# Non-zero when the file holds at least one failure, i.e. a workload never
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
# every rollback this stage is meant to perform.
[ ! -s "$failed_file" ]
}
# rollback_workloads <failed-file>
# Restores the previous revision of every failed workload and waits for it to
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
# so the operator knows manual recovery is required.
rollback_workloads() {
local failed_file="$1"
local kind ns name unrecovered=()
local -a recovered=()
while read -r kind ns name; do
[ -n "${kind:-}" ] || continue
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " rolled back: ${kind}/${ns}/${name}"
recovered+=("${kind}/${ns}/${name}")
else
echo " NOT RECOVERED: ${kind}/${ns}/${name}"
unrecovered+=("${kind}/${ns}/${name}")
fi
done <"$failed_file"
echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file"
echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file"
[ "${#unrecovered[@]}" -eq 0 ]
}
# Helm releases owned by this stage, one line each:
#
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
#
# The chart version is the field Renovate keeps current. The helmv3 manager only
# understands Chart.yaml and the helm-values manager only values files, so a pin
# written straight into a `helm upgrade` command would never be updated: these
# have to be declared as custom.regex managers in renovate/renovate.json.
HELM_RELEASES=(
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
)
# "name url" for the Helm repository hosting a chart, empty if unknown.
helm_repo_for() {
case "$1" in
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
esac
}
upgrade_helm_releases() {
local entry release chart namespace version values marker repo
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
IFS='|' read -r release chart namespace version values marker <<<"$entry"
if [ ! -f "$REPO/$marker" ]; then
echo "skip (no $marker): $release"
continue
fi
if [ ! -f "$REPO/$values" ]; then
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
return 1
fi
repo="$(helm_repo_for "$chart")"
if [ -z "$repo" ]; then
echo "ERROR: no Helm repository configured for chart $chart"
return 1
fi
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
log "Upgrading $release ($chart $version)"
# --atomic rolls the release back when the upgrade times out or the workloads
# it touches never become ready, so a bad chart bump is not left half applied.
helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \
--version "$version" \
--values "$REPO/$values" \
--atomic --cleanup-on-fail --timeout 10m
done
}
stage_preflight() {
if [ ! -d "$REPO/.git" ]; then
echo "Repository not found at $REPO"
exit 1
fi
if [ -n "$DEPLOY_SHA" ]; then
log "Checking out the commit CI validated ($DEPLOY_SHA)"
git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \
|| git -C "$REPO" fetch origin main
else
git -C "$REPO" fetch origin main
fi
target="${DEPLOY_SHA:-origin/main}"
log "Workstation state"
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
echo " target: $(git -C "$REPO" rev-parse --short "$target")"
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
echo "ERROR: workstation has local tracked modifications, refusing reset:"
git -C "$REPO" status --porcelain --untracked-files=no
git -C "$REPO" diff --stat
echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy."
exit 1
fi
git -C "$REPO" reset --hard "$target"
}
stage_validate() {
cd "$REPO"
select_manifests
local m k cf
# Compose .env files and secret files are gitignored by design, so the
# workstation never has real values for the inactive stacks. This stage only
# runs the full check on active stacks; the general structure check for every
# committed Compose file (active or not) lives in the ci workflow, which has no
# .env at all.
#
# Active stacks are still validated with interpolation and env-file resolution
# off, so required-variable guards (:?) and missing local files do not fail the
# deploy. Normalization and consistency checks stay enabled.
# shellcheck source=compose-lint.sh
source "$REPO/.gitea/workflows/compose-lint.sh"
local compose_validate_flags=()
mapfile -t compose_validate_flags < <(compose_safe_flags)
log "Validate compose stacks"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " config: $cf"
validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"}
done
log "Validate k8s manifests (kubectl dry-run=client)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
kubectl apply --dry-run=client -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=client >/dev/null
done
log "Validate k8s manifests (kubectl dry-run=server)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
kubectl apply --dry-run=server -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=server >/dev/null
done
log "Checking referenced Secrets exist"
echo " (deploy never applies *secret*.yaml; create missing ones from the laptop)"
local ref_secrets=() missing_secrets=() all_secrets s
if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then
while IFS= read -r s; do
[ -n "$s" ] && ref_secrets+=("$s")
done < <(
{
grep -h -A1 -E 'secretRef:|secretKeyRef:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
grep -h -E 'secretName:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
} | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true
)
fi
all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)"
for s in ${ref_secrets[@]+"${ref_secrets[@]}"}; do
if printf '%s\n' "$all_secrets" | grep -qx "$s"; then
echo " ok: $s"
else
echo " MISSING: $s"
missing_secrets+=("$s")
fi
done
if [ "${#missing_secrets[@]}" -gt 0 ]; then
echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:"
printf ' - %s\n' "${missing_secrets[@]}"
echo "Create them manually from the laptop, e.g.:"
echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example"
exit 1
fi
}
stage_apply_k8s() {
cd "$REPO"
select_manifests >/dev/null
local ns_files=() other_files=() m k prune_opts=()
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
case "$m" in
*/namespace.y?ml) ns_files+=("$m") ;;
*) other_files+=("$m") ;;
esac
done
if [ "$APPLY_PRUNE" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
# Record what is about to change, and publish it for the verify job, before
# the first apply. Both are fatal on failure: see snapshot_dir.
local snapshot
snapshot="$(snapshot_dir)" || return 1
save_snapshot "$snapshot" || return 1
if [ "${#ns_files[@]}" -gt 0 ]; then
log "Applying namespaces (${#ns_files[@]} files)"
for m in "${ns_files[@]}"; do
kubectl apply -f "$m"
done
fi
if [ -f "$REPO/prometheus-stack/k8s/active" ]; then
if [ ! -f "$REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
exit 1
fi
fi
upgrade_helm_releases
if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files)"
for m in "${other_files[@]}"; do
kubectl apply "${prune_opts[@]}" -f "$m"
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/}"
kubectl apply -k "$k"
done
if [ -f "$REPO/userbot/k8s/active" ]; then
log "userbot panel hook"
if kubectl get secret userbot-common-secrets -n userbot >/dev/null 2>&1; then
echo " userbot-common-secrets already present in userbot ns, not touching"
elif kubectl get secret userbot-common-secrets -n default >/dev/null 2>&1; then
echo " bootstrapping userbot-common-secrets into userbot ns"
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
else
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
fi
kubectl rollout restart deployment/userbot-panel -n userbot
fi
# No verification here on purpose. This stage may be killed at any point by
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
# connection, and any code below that line would simply not run. stage_verify_k8s
# picks the work up from the snapshot instead.
log "Applied. Verification and rollback are the verify job's job, not this one's."
}
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
# including when they failed, timed out or were cancelled. Reads the baseline the
# apply stage published and works out what it changed, watches those workloads,
# and rolls back the ones that never became healthy.
stage_verify_k8s() {
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
local snapshot want have generations
local -a touched=()
if [ ! -s "$pointer" ]; then
echo "ERROR: no snapshot pointer at $pointer."
echo "The apply stage died before publishing any state, so there is no baseline to"
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
echo "inspect the cluster by hand."
return 1
fi
snapshot="$(head -1 "$pointer")"
if [ ! -d "$snapshot" ]; then
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
return 1
fi
# Never trust the pointer blindly. If the apply stage was killed before it
# published its own snapshot, `current` still points at the previous deploy's
# baseline. Verifying against that would watch the wrong workloads and the
# rollback would revert the wrong revisions, so refuse instead.
want="${DEPLOY_SHA:-}"
if [ -z "$want" ]; then
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
fi
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
if [ -z "$want" ] || [ "$have" != "$want" ]; then
echo "ERROR: refusing to verify or roll back against a stale snapshot."
echo " snapshot: $snapshot"
echo " snapshot commit: ${have:-<missing>}"
echo " deploy commit: ${want:-<unknown>}"
return 1
fi
echo " snapshot: $snapshot (commit ${have:0:12})"
generations="$snapshot/generations.before"
if [ ! -s "$generations" ]; then
# Without a baseline we cannot tell which workloads the apply touched, so
# fall back to watching everything rather than silently skipping the check.
warn "no pre-apply baseline, verifying every workload in the cluster"
: >"$generations"
fi
while read -r kind ns name; do
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
done < <(changed_workloads "$generations")
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
if [ "${#touched[@]}" -eq 0 ]; then
echo " nothing to verify"
return 0
fi
printf ' watching: %s\n' "${touched[@]/#/ }"
local failed_file="$snapshot/failed-workloads"
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
echo
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
echo
log "Rolling back to the previous revision"
if rollback_workloads "$failed_file"; then
echo
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
echo "externally created resources from this commit are still in place. Review the failed"
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
else
echo
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
echo "Pre-apply snapshot: $snapshot"
fi
return 1
fi
}
# verify_compose_stack <compose-file>
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
# come back broken with a green pipeline. Require every long-running service to
# actually be running.
verify_compose_stack() {
local cf="$1"
local expected running missing=()
expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)"
running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)"
[ -n "$expected" ] || return 0
while IFS= read -r svc; do
[ -n "$svc" ] || continue
# restart:"no" services are allowed to have exited.
if ! printf '%s\n' "$running" | grep -qx "$svc" \
&& ! docker compose -f "$cf" config 2>/dev/null \
| grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then
missing+=("$svc")
fi
done <<<"$expected"
if [ "${#missing[@]}" -gt 0 ]; then
echo " NOT RUNNING: ${missing[*]}"
docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
return 1
fi
echo " all ${#expected} service(s) running"
return 0
}
stage_apply_compose() {
cd "$REPO"
select_manifests >/dev/null
local cf
log "Redeploying docker compose stacks (${#COMPOSE_STACKS[@]} stacks)"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " compose: $cf"
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
docker compose -f "$cf" build
docker compose -f "$cf" push
fi
docker compose -f "$cf" up -d --pull always --remove-orphans
done
local -a broken=()
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " verifying: $cf"
if ! verify_compose_stack "$cf"; then
broken+=("$cf")
fi
done
if [ "${#broken[@]}" -gt 0 ]; then
echo
echo "ERROR: ${#broken[@]} compose stack(s) did not come up:"
printf ' - %s\n' "${broken[@]}"
echo "Compose stacks are not rolled back automatically: their images use mutable"
echo "':latest' tags, so there is no previous version to return to. Check the logs"
echo "above, then re-run the deploy once the cause is fixed."
return 1
fi
}
run_stage() {
case "${1:?stage required}" in
preflight) stage_preflight ;;
validate) stage_validate ;;
apply-k8s) stage_apply_k8s ;;
verify-k8s) stage_verify_k8s ;;
apply-compose) stage_apply_compose ;;
*)
echo "ERROR: unknown stage: $1"
exit 1
;;
esac
}