fix(deploy): verify and roll back in a separate job
verify_workloads ended on `[ -s "$failed_file" ]`, which is the opposite of what its own contract says. A non-empty file means something failed, so the function returned success exactly when a workload never came up, and failure when everything was fine. Every rollback was therefore skipped, and every deploy that changed anything ended red with an empty failure list and a bogus "Rolled back successfully". Worse, the check only ever ran at the end of stage_apply_k8s, inside the same process as the apply. A job killed by timeout-minutes, cancelled by a new push, or cut off by a dropped SSH connection never reached it, which is precisely when a rollback matters. The three helm upgrades alone can consume the whole 30-minute job budget, so that path was reachable. Verification now lives in its own job, gated on always(), so it runs whatever happened to the apply. The apply stage publishes its pre-apply snapshot through DEPLOY_SNAPSHOT_DIR/current before touching anything, and the verify stage picks it up from there. A snapshot whose recorded commit does not match the deploy is refused rather than trusted, so a stale pointer from an earlier run cannot make the rollback revert the wrong workloads. An unwritable snapshot directory now fails the deploy up front instead of silently continuing without a way back. cancel-in-progress becomes false for the same reason: cancelling a run kills the apply job and takes the verify job with it, which is the failure this change exists to prevent. Both applies are idempotent, so queueing costs little. The SSH key moves to a per-run directory removed on exit, and the deploy is pinned to the exact commit CI validated.
This commit is contained in:
1 parent
7ce727bc8a
commit
1505b638ce
3 files changed
+447
-51
No files matched your search
+387
-44
@@ -3,16 +3,29 @@
|
|||||||
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
|
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
|
||||||
# source "$REPO/.gitea/workflows/deploy-lib.sh"
|
# source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||||
# run_stage "$STAGE"
|
# run_stage "$STAGE"
|
||||||
# EOF
|
# EOF
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
: "${REPO:?REPO must be set}"
|
: "${REPO:?REPO must be set}"
|
||||||
APPLY_PRUNE="${APPLY_PRUNE:-false}"
|
APPLY_PRUNE="${APPLY_PRUNE:-false}"
|
||||||
|
# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to
|
||||||
|
# the current origin/main.
|
||||||
|
DEPLOY_SHA="${DEPLOY_SHA:-}"
|
||||||
|
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-/var/backups/homelab-deploy}"
|
||||||
|
# Per-workload rollout budget and how many workloads to watch at once. The whole
|
||||||
|
# apply job has its own timeout-minutes as a backstop.
|
||||||
|
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
|
||||||
|
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}"
|
||||||
|
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
|
||||||
|
|
||||||
log() {
|
log() {
|
||||||
echo "== $* =="
|
echo "== $* =="
|
||||||
}
|
}
|
||||||
|
|
||||||
|
warn() {
|
||||||
|
echo "WARNING: $*" >&2
|
||||||
|
}
|
||||||
|
|
||||||
collect_k8s() {
|
collect_k8s() {
|
||||||
git -C "$REPO" ls-files -- "$1" \
|
git -C "$REPO" ls-files -- "$1" \
|
||||||
| grep -E '\.ya?ml$' \
|
| grep -E '\.ya?ml$' \
|
||||||
@@ -27,6 +40,8 @@ kustomize_overlay() {
|
|||||||
echo "$1/overlays/prod"
|
echo "$1/overlays/prod"
|
||||||
elif [ -f "$1/base/kustomization.yaml" ]; then
|
elif [ -f "$1/base/kustomization.yaml" ]; then
|
||||||
echo "$1/base"
|
echo "$1/base"
|
||||||
|
elif [ -f "$1/kustomization.yaml" ]; then
|
||||||
|
echo "$1"
|
||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -43,7 +58,7 @@ select_manifests() {
|
|||||||
fi
|
fi
|
||||||
overlay="$(kustomize_overlay "$kd" || true)"
|
overlay="$(kustomize_overlay "$kd" || true)"
|
||||||
if [ -n "${overlay:-}" ]; then
|
if [ -n "${overlay:-}" ]; then
|
||||||
echo "kustomize app: ${overlay#$REPO/}"
|
echo "kustomize app: ${overlay#"$REPO"/}"
|
||||||
KUSTOMIZE_APPS+=("$overlay")
|
KUSTOMIZE_APPS+=("$overlay")
|
||||||
else
|
else
|
||||||
while IFS= read -r f; do
|
while IFS= read -r f; do
|
||||||
@@ -67,22 +82,234 @@ select_manifests() {
|
|||||||
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
|
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# --- post-apply verification and rollback -------------------------------------
|
||||||
|
#
|
||||||
|
# A green `kubectl apply` says nothing about the cluster being healthy. These
|
||||||
|
# helpers watch exactly the workloads whose spec changed during this apply, and
|
||||||
|
# on failure roll them back to the revision that was running before, so a bad
|
||||||
|
# push to main cannot leave a service crash-looping.
|
||||||
|
#
|
||||||
|
# Verification lives in its own workflow job, not at the end of the apply stage.
|
||||||
|
# Inside a single process it is worthless exactly when it is needed most: a job
|
||||||
|
# killed by timeout-minutes or cancelled mid-apply never reaches the rollback
|
||||||
|
# code, and leaves a half-applied cluster behind. Split out, the apply job can
|
||||||
|
# die in any way and the verify job still runs.
|
||||||
|
#
|
||||||
|
# That split needs a handoff point on the workstation, because the two stages are
|
||||||
|
# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written
|
||||||
|
# before anything is applied, read by the verify stage afterwards.
|
||||||
|
|
||||||
|
# Creates this run's snapshot directory and publishes it as the handoff point for
|
||||||
|
# the verify stage. Fails hard by design: a deploy that cannot record what it is
|
||||||
|
# about to change must not start, because then nothing can be rolled back for it
|
||||||
|
# automatically. Publishing happens before the first apply, so an apply killed
|
||||||
|
# mid-flight still leaves a usable baseline behind.
|
||||||
|
snapshot_dir() {
|
||||||
|
local stamp dir
|
||||||
|
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
|
||||||
|
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
|
||||||
|
|
||||||
|
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
|
||||||
|
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
|
||||||
|
echo "The verify job needs it to learn which workloads this deploy touches." >&2
|
||||||
|
echo "Refusing to deploy without a way to roll back." >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
|
||||||
|
echo "ERROR: cannot create snapshot dir $dir" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
|
||||||
|
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
printf '%s\n' "$dir"
|
||||||
|
}
|
||||||
|
|
||||||
|
save_snapshot() {
|
||||||
|
local dir="$1"
|
||||||
|
log "Saving pre-apply snapshot to $dir"
|
||||||
|
workload_generations >"$dir/generations.before" 2>/dev/null \
|
||||||
|
|| warn "could not snapshot workload generations"
|
||||||
|
kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \
|
||||||
|
|| warn "could not snapshot workloads"
|
||||||
|
for release in prometheus-stack loki alloy; do
|
||||||
|
if helm status "$release" -n prometheus >/dev/null 2>&1; then
|
||||||
|
{
|
||||||
|
echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)"
|
||||||
|
helm get values "$release" -n prometheus --all 2>/dev/null
|
||||||
|
} >"$dir/helm-$release.txt"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
# The verify stage compares this against the commit it is deploying, to refuse
|
||||||
|
# rolling back against a baseline left by an earlier run. A snapshot we cannot
|
||||||
|
# attribute to a commit is unusable for that, so fail before anything is applied.
|
||||||
|
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
|
||||||
|
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
|
||||||
|
workload_generations() {
|
||||||
|
kubectl get "$WORKLOAD_KINDS" -A \
|
||||||
|
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
|
||||||
|
--no-headers 2>/dev/null \
|
||||||
|
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
|
||||||
|
}
|
||||||
|
|
||||||
|
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
|
||||||
|
# moved since the snapshot, i.e. the ones this apply actually touched.
|
||||||
|
changed_workloads() {
|
||||||
|
local before="$1"
|
||||||
|
local ns name kind gen old
|
||||||
|
while read -r ns name kind gen; do
|
||||||
|
[ -n "${gen:-}" ] || continue
|
||||||
|
old="$(awk -v want_ns="$ns" -v want_name="$name" \
|
||||||
|
'$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)"
|
||||||
|
if [ "$old" != "$gen" ]; then
|
||||||
|
printf '%s %s %s\n' "$kind" "$ns" "$name"
|
||||||
|
fi
|
||||||
|
done < <(workload_generations)
|
||||||
|
}
|
||||||
|
|
||||||
|
# verify_workloads <failed-file> <kind> <ns> <name> ...
|
||||||
|
# Watches every workload in parallel and records the ones that never became
|
||||||
|
# healthy. Returns non-zero if any of them failed.
|
||||||
|
verify_workloads() {
|
||||||
|
local failed_file="$1"
|
||||||
|
shift
|
||||||
|
[ "$#" -gt 0 ] || return 0
|
||||||
|
: >"$failed_file"
|
||||||
|
local running=0 pid kind ns name
|
||||||
|
local -a pids=()
|
||||||
|
for entry in "$@"; do
|
||||||
|
read -r kind ns name <<<"$entry"
|
||||||
|
(
|
||||||
|
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
||||||
|
echo " ok: ${kind}/${ns}/${name}"
|
||||||
|
else
|
||||||
|
echo " FAILED: ${kind}/${ns}/${name}"
|
||||||
|
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
|
||||||
|
fi
|
||||||
|
) &
|
||||||
|
pids+=($!)
|
||||||
|
running=$((running + 1))
|
||||||
|
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
|
||||||
|
wait -n 2>/dev/null || true
|
||||||
|
running=$((running - 1))
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
for pid in ${pids[@]+"${pids[@]}"}; do
|
||||||
|
wait "$pid" || true
|
||||||
|
done
|
||||||
|
# Non-zero when the file holds at least one failure, i.e. a workload never
|
||||||
|
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
|
||||||
|
# every rollback this stage is meant to perform.
|
||||||
|
[ ! -s "$failed_file" ]
|
||||||
|
}
|
||||||
|
|
||||||
|
# rollback_workloads <failed-file>
|
||||||
|
# Restores the previous revision of every failed workload and waits for it to
|
||||||
|
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
|
||||||
|
# so the operator knows manual recovery is required.
|
||||||
|
rollback_workloads() {
|
||||||
|
local failed_file="$1"
|
||||||
|
local kind ns name unrecovered=()
|
||||||
|
local -a recovered=()
|
||||||
|
while read -r kind ns name; do
|
||||||
|
[ -n "${kind:-}" ] || continue
|
||||||
|
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
|
||||||
|
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
||||||
|
echo " rolled back: ${kind}/${ns}/${name}"
|
||||||
|
recovered+=("${kind}/${ns}/${name}")
|
||||||
|
else
|
||||||
|
echo " NOT RECOVERED: ${kind}/${ns}/${name}"
|
||||||
|
unrecovered+=("${kind}/${ns}/${name}")
|
||||||
|
fi
|
||||||
|
done <"$failed_file"
|
||||||
|
echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file"
|
||||||
|
echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file"
|
||||||
|
[ "${#unrecovered[@]}" -eq 0 ]
|
||||||
|
}
|
||||||
|
|
||||||
|
# Helm releases owned by this stage, one line each:
|
||||||
|
#
|
||||||
|
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
|
||||||
|
#
|
||||||
|
# The chart version is the field Renovate keeps current. The helmv3 manager only
|
||||||
|
# understands Chart.yaml and the helm-values manager only values files, so a pin
|
||||||
|
# written straight into a `helm upgrade` command would never be updated: these
|
||||||
|
# have to be declared as custom.regex managers in renovate/renovate.json.
|
||||||
|
HELM_RELEASES=(
|
||||||
|
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
|
||||||
|
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
|
||||||
|
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
|
||||||
|
)
|
||||||
|
|
||||||
|
# "name url" for the Helm repository hosting a chart, empty if unknown.
|
||||||
|
helm_repo_for() {
|
||||||
|
case "$1" in
|
||||||
|
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
|
||||||
|
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
|
||||||
|
upgrade_helm_releases() {
|
||||||
|
local entry release chart namespace version values marker repo
|
||||||
|
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||||
|
IFS='|' read -r release chart namespace version values marker <<<"$entry"
|
||||||
|
if [ ! -f "$REPO/$marker" ]; then
|
||||||
|
echo "skip (no $marker): $release"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
if [ ! -f "$REPO/$values" ]; then
|
||||||
|
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
repo="$(helm_repo_for "$chart")"
|
||||||
|
if [ -z "$repo" ]; then
|
||||||
|
echo "ERROR: no Helm repository configured for chart $chart"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||||
|
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||||
|
log "Upgrading $release ($chart $version)"
|
||||||
|
# --atomic rolls the release back when the upgrade times out or the workloads
|
||||||
|
# it touches never become ready, so a bad chart bump is not left half applied.
|
||||||
|
helm upgrade --install "$release" "$chart" \
|
||||||
|
--namespace "$namespace" \
|
||||||
|
--version "$version" \
|
||||||
|
--values "$REPO/$values" \
|
||||||
|
--atomic --cleanup-on-fail --timeout 10m
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
stage_preflight() {
|
stage_preflight() {
|
||||||
if [ ! -d "$REPO/.git" ]; then
|
if [ ! -d "$REPO/.git" ]; then
|
||||||
echo "Repository not found at $REPO"
|
echo "Repository not found at $REPO"
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
git -C "$REPO" fetch origin main
|
if [ -n "$DEPLOY_SHA" ]; then
|
||||||
|
log "Checking out the commit CI validated ($DEPLOY_SHA)"
|
||||||
|
git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \
|
||||||
|
|| git -C "$REPO" fetch origin main
|
||||||
|
else
|
||||||
|
git -C "$REPO" fetch origin main
|
||||||
|
fi
|
||||||
|
target="${DEPLOY_SHA:-origin/main}"
|
||||||
log "Workstation state"
|
log "Workstation state"
|
||||||
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
|
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
|
||||||
echo " remote: $(git -C "$REPO" rev-parse --short origin/main)"
|
echo " target: $(git -C "$REPO" rev-parse --short "$target")"
|
||||||
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
|
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
|
||||||
echo "ERROR: workstation has local tracked modifications, refusing reset:"
|
echo "ERROR: workstation has local tracked modifications, refusing reset:"
|
||||||
git -C "$REPO" status --porcelain --untracked-files=no
|
git -C "$REPO" status --porcelain --untracked-files=no
|
||||||
git -C "$REPO" diff --stat
|
git -C "$REPO" diff --stat
|
||||||
|
echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy."
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
git -C "$REPO" reset --hard origin/main
|
git -C "$REPO" reset --hard "$target"
|
||||||
}
|
}
|
||||||
|
|
||||||
stage_validate() {
|
stage_validate() {
|
||||||
@@ -90,26 +317,22 @@ stage_validate() {
|
|||||||
select_manifests
|
select_manifests
|
||||||
local m k cf
|
local m k cf
|
||||||
# Compose .env files and secret files are gitignored by design, so the
|
# Compose .env files and secret files are gitignored by design, so the
|
||||||
# workstation never has real values for them. Validate structure only:
|
# workstation never has real values for the inactive stacks. This stage only
|
||||||
# skip interpolation, env-file resolution, and path resolution so that
|
# runs the full check on active stacks; the general structure check for every
|
||||||
# required-variable guards (:?) and missing local files don't fail CI.
|
# committed Compose file (active or not) lives in the ci workflow, which has no
|
||||||
# Normalization and consistency checks stay enabled.
|
# .env at all.
|
||||||
|
#
|
||||||
|
# Active stacks are still validated with interpolation and env-file resolution
|
||||||
|
# off, so required-variable guards (:?) and missing local files do not fail the
|
||||||
|
# deploy. Normalization and consistency checks stay enabled.
|
||||||
|
# shellcheck source=compose-lint.sh
|
||||||
|
source "$REPO/.gitea/workflows/compose-lint.sh"
|
||||||
local compose_validate_flags=()
|
local compose_validate_flags=()
|
||||||
local compose_config_help
|
mapfile -t compose_validate_flags < <(compose_safe_flags)
|
||||||
compose_config_help="$(docker compose config --help 2>/dev/null || true)"
|
|
||||||
if printf '%s' "$compose_config_help" | grep -q -- '--no-interpolate'; then
|
|
||||||
compose_validate_flags+=(--no-interpolate)
|
|
||||||
fi
|
|
||||||
if printf '%s' "$compose_config_help" | grep -q -- '--no-env-resolution'; then
|
|
||||||
compose_validate_flags+=(--no-env-resolution)
|
|
||||||
fi
|
|
||||||
if printf '%s' "$compose_config_help" | grep -q -- '--no-path-resolution'; then
|
|
||||||
compose_validate_flags+=(--no-path-resolution)
|
|
||||||
fi
|
|
||||||
log "Validate compose stacks"
|
log "Validate compose stacks"
|
||||||
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
||||||
echo " config: $cf"
|
echo " config: $cf"
|
||||||
docker compose -f "$cf" config --quiet "${compose_validate_flags[@]}"
|
validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"}
|
||||||
done
|
done
|
||||||
log "Validate k8s manifests (kubectl dry-run=client)"
|
log "Validate k8s manifests (kubectl dry-run=client)"
|
||||||
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
||||||
@@ -169,6 +392,13 @@ stage_apply_k8s() {
|
|||||||
if [ "$APPLY_PRUNE" = "true" ]; then
|
if [ "$APPLY_PRUNE" = "true" ]; then
|
||||||
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
|
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# Record what is about to change, and publish it for the verify job, before
|
||||||
|
# the first apply. Both are fatal on failure: see snapshot_dir.
|
||||||
|
local snapshot
|
||||||
|
snapshot="$(snapshot_dir)" || return 1
|
||||||
|
save_snapshot "$snapshot" || return 1
|
||||||
|
|
||||||
if [ "${#ns_files[@]}" -gt 0 ]; then
|
if [ "${#ns_files[@]}" -gt 0 ]; then
|
||||||
log "Applying namespaces (${#ns_files[@]} files)"
|
log "Applying namespaces (${#ns_files[@]} files)"
|
||||||
for m in "${ns_files[@]}"; do
|
for m in "${ns_files[@]}"; do
|
||||||
@@ -180,28 +410,8 @@ stage_apply_k8s() {
|
|||||||
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
|
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
log "Upgrading kube-prometheus-stack"
|
|
||||||
helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \
|
|
||||||
--namespace prometheus \
|
|
||||||
--version 86.2.3 \
|
|
||||||
--values "$REPO/prometheus-stack/k8s/grafana-values.yaml" \
|
|
||||||
--wait --timeout 10m
|
|
||||||
fi
|
|
||||||
if [ -f "$REPO/loki/k8s/active" ]; then
|
|
||||||
log "Upgrading loki/alloy"
|
|
||||||
helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true
|
|
||||||
helm repo update grafana >/dev/null 2>&1 || true
|
|
||||||
helm upgrade --install loki grafana/loki \
|
|
||||||
--version 7.3.0 \
|
|
||||||
--namespace prometheus \
|
|
||||||
--values "$REPO/loki/k8s/loki-values.yaml" \
|
|
||||||
--wait --timeout 10m
|
|
||||||
helm upgrade --install alloy grafana/alloy \
|
|
||||||
--version 1.12.1 \
|
|
||||||
--namespace prometheus \
|
|
||||||
--values "$REPO/loki/k8s/alloy-values.yaml" \
|
|
||||||
--wait --timeout 10m
|
|
||||||
fi
|
fi
|
||||||
|
upgrade_helm_releases
|
||||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||||
log "Applying resources (${#other_files[@]} files)"
|
log "Applying resources (${#other_files[@]} files)"
|
||||||
for m in "${other_files[@]}"; do
|
for m in "${other_files[@]}"; do
|
||||||
@@ -209,7 +419,7 @@ stage_apply_k8s() {
|
|||||||
done
|
done
|
||||||
fi
|
fi
|
||||||
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
||||||
log "Applying kustomize app: ${k#$REPO/}"
|
log "Applying kustomize app: ${k#"$REPO"/}"
|
||||||
kubectl apply -k "$k"
|
kubectl apply -k "$k"
|
||||||
done
|
done
|
||||||
if [ -f "$REPO/userbot/k8s/active" ]; then
|
if [ -f "$REPO/userbot/k8s/active" ]; then
|
||||||
@@ -225,8 +435,123 @@ stage_apply_k8s() {
|
|||||||
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
|
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
|
||||||
fi
|
fi
|
||||||
kubectl rollout restart deployment/userbot-panel -n userbot
|
kubectl rollout restart deployment/userbot-panel -n userbot
|
||||||
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
|
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# No verification here on purpose. This stage may be killed at any point by
|
||||||
|
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
|
||||||
|
# connection, and any code below that line would simply not run. stage_verify_k8s
|
||||||
|
# picks the work up from the snapshot instead.
|
||||||
|
log "Applied. Verification and rollback are the verify job's job, not this one's."
|
||||||
|
}
|
||||||
|
|
||||||
|
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
|
||||||
|
# including when they failed, timed out or were cancelled. Reads the baseline the
|
||||||
|
# apply stage published and works out what it changed, watches those workloads,
|
||||||
|
# and rolls back the ones that never became healthy.
|
||||||
|
stage_verify_k8s() {
|
||||||
|
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
|
||||||
|
local snapshot want have generations
|
||||||
|
local -a touched=()
|
||||||
|
|
||||||
|
if [ ! -s "$pointer" ]; then
|
||||||
|
echo "ERROR: no snapshot pointer at $pointer."
|
||||||
|
echo "The apply stage died before publishing any state, so there is no baseline to"
|
||||||
|
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
|
||||||
|
echo "inspect the cluster by hand."
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
snapshot="$(head -1 "$pointer")"
|
||||||
|
if [ ! -d "$snapshot" ]; then
|
||||||
|
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Never trust the pointer blindly. If the apply stage was killed before it
|
||||||
|
# published its own snapshot, `current` still points at the previous deploy's
|
||||||
|
# baseline. Verifying against that would watch the wrong workloads and the
|
||||||
|
# rollback would revert the wrong revisions, so refuse instead.
|
||||||
|
want="${DEPLOY_SHA:-}"
|
||||||
|
if [ -z "$want" ]; then
|
||||||
|
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
|
||||||
|
fi
|
||||||
|
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
|
||||||
|
if [ -z "$want" ] || [ "$have" != "$want" ]; then
|
||||||
|
echo "ERROR: refusing to verify or roll back against a stale snapshot."
|
||||||
|
echo " snapshot: $snapshot"
|
||||||
|
echo " snapshot commit: ${have:-<missing>}"
|
||||||
|
echo " deploy commit: ${want:-<unknown>}"
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
echo " snapshot: $snapshot (commit ${have:0:12})"
|
||||||
|
|
||||||
|
generations="$snapshot/generations.before"
|
||||||
|
if [ ! -s "$generations" ]; then
|
||||||
|
# Without a baseline we cannot tell which workloads the apply touched, so
|
||||||
|
# fall back to watching everything rather than silently skipping the check.
|
||||||
|
warn "no pre-apply baseline, verifying every workload in the cluster"
|
||||||
|
: >"$generations"
|
||||||
|
fi
|
||||||
|
|
||||||
|
while read -r kind ns name; do
|
||||||
|
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
|
||||||
|
done < <(changed_workloads "$generations")
|
||||||
|
|
||||||
|
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
|
||||||
|
if [ "${#touched[@]}" -eq 0 ]; then
|
||||||
|
echo " nothing to verify"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
printf ' watching: %s\n' "${touched[@]/#/ }"
|
||||||
|
|
||||||
|
local failed_file="$snapshot/failed-workloads"
|
||||||
|
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
|
||||||
|
echo
|
||||||
|
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
|
||||||
|
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
|
||||||
|
echo
|
||||||
|
log "Rolling back to the previous revision"
|
||||||
|
if rollback_workloads "$failed_file"; then
|
||||||
|
echo
|
||||||
|
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
|
||||||
|
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
|
||||||
|
echo "externally created resources from this commit are still in place. Review the failed"
|
||||||
|
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
|
||||||
|
else
|
||||||
|
echo
|
||||||
|
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
|
||||||
|
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
|
||||||
|
echo "Pre-apply snapshot: $snapshot"
|
||||||
|
fi
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
# verify_compose_stack <compose-file>
|
||||||
|
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
|
||||||
|
# come back broken with a green pipeline. Require every long-running service to
|
||||||
|
# actually be running.
|
||||||
|
verify_compose_stack() {
|
||||||
|
local cf="$1"
|
||||||
|
local expected running missing=()
|
||||||
|
expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)"
|
||||||
|
running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)"
|
||||||
|
[ -n "$expected" ] || return 0
|
||||||
|
while IFS= read -r svc; do
|
||||||
|
[ -n "$svc" ] || continue
|
||||||
|
# restart:"no" services are allowed to have exited.
|
||||||
|
if ! printf '%s\n' "$running" | grep -qx "$svc" \
|
||||||
|
&& ! docker compose -f "$cf" config 2>/dev/null \
|
||||||
|
| grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then
|
||||||
|
missing+=("$svc")
|
||||||
|
fi
|
||||||
|
done <<<"$expected"
|
||||||
|
if [ "${#missing[@]}" -gt 0 ]; then
|
||||||
|
echo " NOT RUNNING: ${missing[*]}"
|
||||||
|
docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
echo " all ${#expected} service(s) running"
|
||||||
|
return 0
|
||||||
}
|
}
|
||||||
|
|
||||||
stage_apply_compose() {
|
stage_apply_compose() {
|
||||||
@@ -242,6 +567,23 @@ stage_apply_compose() {
|
|||||||
fi
|
fi
|
||||||
docker compose -f "$cf" up -d --pull always --remove-orphans
|
docker compose -f "$cf" up -d --pull always --remove-orphans
|
||||||
done
|
done
|
||||||
|
|
||||||
|
local -a broken=()
|
||||||
|
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
||||||
|
echo " verifying: $cf"
|
||||||
|
if ! verify_compose_stack "$cf"; then
|
||||||
|
broken+=("$cf")
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
if [ "${#broken[@]}" -gt 0 ]; then
|
||||||
|
echo
|
||||||
|
echo "ERROR: ${#broken[@]} compose stack(s) did not come up:"
|
||||||
|
printf ' - %s\n' "${broken[@]}"
|
||||||
|
echo "Compose stacks are not rolled back automatically: their images use mutable"
|
||||||
|
echo "':latest' tags, so there is no previous version to return to. Check the logs"
|
||||||
|
echo "above, then re-run the deploy once the cause is fixed."
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
run_stage() {
|
run_stage() {
|
||||||
@@ -249,6 +591,7 @@ run_stage() {
|
|||||||
preflight) stage_preflight ;;
|
preflight) stage_preflight ;;
|
||||||
validate) stage_validate ;;
|
validate) stage_validate ;;
|
||||||
apply-k8s) stage_apply_k8s ;;
|
apply-k8s) stage_apply_k8s ;;
|
||||||
|
verify-k8s) stage_verify_k8s ;;
|
||||||
apply-compose) stage_apply_compose ;;
|
apply-compose) stage_apply_compose ;;
|
||||||
*)
|
*)
|
||||||
echo "ERROR: unknown stage: $1"
|
echo "ERROR: unknown stage: $1"
|
||||||
|
|||||||
@@ -1,14 +1,21 @@
|
|||||||
name: deploy
|
name: deploy
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
# Deploy only what CI already validated. workflow_run is used instead of
|
||||||
branches:
|
# workflow_dispatch so a red lint/validate run can never reach the cluster.
|
||||||
- main
|
workflow_run:
|
||||||
|
workflows: [ci]
|
||||||
|
types: [completed]
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
|
|
||||||
concurrency:
|
concurrency:
|
||||||
group: deploy-main
|
group: deploy-main
|
||||||
cancel-in-progress: true
|
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
|
||||||
|
# takes the verify job down with it, so a superseded deploy would leave the
|
||||||
|
# cluster half-applied and unchecked — the exact failure the verify job exists
|
||||||
|
# to catch. kubectl apply and docker compose up are both idempotent, so letting
|
||||||
|
# the older run finish and then deploying the newer commit costs little.
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
env:
|
env:
|
||||||
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
||||||
@@ -17,10 +24,21 @@ env:
|
|||||||
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
||||||
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
||||||
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
||||||
|
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
|
||||||
|
# finished ci run checked. Pin the exact validated commit instead, so a push
|
||||||
|
# landing mid-deploy cannot make the workstation deploy something else. Also
|
||||||
|
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
|
||||||
|
# which falls back to the current origin/main.
|
||||||
|
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
preflight:
|
preflight:
|
||||||
|
if: >-
|
||||||
|
github.event_name != 'workflow_run' ||
|
||||||
|
(github.event.workflow_run.conclusion == 'success' &&
|
||||||
|
github.event.workflow_run.head_branch == 'main')
|
||||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||||
|
timeout-minutes: 10
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||||
@@ -34,6 +52,7 @@ jobs:
|
|||||||
validate:
|
validate:
|
||||||
needs: [preflight]
|
needs: [preflight]
|
||||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||||
|
timeout-minutes: 20
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||||
@@ -47,6 +66,10 @@ jobs:
|
|||||||
apply-k8s:
|
apply-k8s:
|
||||||
needs: [validate]
|
needs: [validate]
|
||||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||||
|
# Apply only, no verification, so this is just the work itself: snapshot,
|
||||||
|
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
|
||||||
|
# apply loop. Verification has its own job and its own budget.
|
||||||
|
timeout-minutes: 45
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||||
@@ -60,6 +83,7 @@ jobs:
|
|||||||
apply-compose:
|
apply-compose:
|
||||||
needs: [validate]
|
needs: [validate]
|
||||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||||
|
timeout-minutes: 30
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||||
@@ -69,3 +93,27 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
./.gitea/workflows/ssh-run.sh apply-compose
|
./.gitea/workflows/ssh-run.sh apply-compose
|
||||||
|
|
||||||
|
# Watches the workloads this deploy changed and rolls back the ones that never
|
||||||
|
# became healthy. Runs even when the apply jobs failed, timed out or were
|
||||||
|
# cancelled — that is the whole point of splitting it out. `always()` is what
|
||||||
|
# lets it start after a failed dependency; the needs on apply-compose are a
|
||||||
|
# barrier, so verification begins only once both applies are done.
|
||||||
|
verify-k8s:
|
||||||
|
needs: [apply-k8s, apply-compose]
|
||||||
|
if: >-
|
||||||
|
always() &&
|
||||||
|
needs.apply-k8s.result != 'skipped' &&
|
||||||
|
needs.apply-compose.result != 'skipped'
|
||||||
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||||
|
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
|
||||||
|
timeout-minutes: 30
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||||
|
|
||||||
|
- name: Verify workloads and roll back on failure
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
./.gitea/workflows/ssh-run.sh verify-k8s
|
||||||
@@ -11,15 +11,20 @@ deploy_port="${DEPLOY_PORT:-22}"
|
|||||||
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
||||||
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
|
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
|
||||||
|
|
||||||
ssh_key="$RUNNER_TEMP/deploy_key"
|
# The private key is written to a per-run directory that is removed on exit, so a
|
||||||
mkdir -p "$RUNNER_TEMP"
|
# failed or cancelled job cannot leave deploy credentials in the runner's temp
|
||||||
|
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
|
||||||
|
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
|
||||||
|
trap 'rm -rf "$key_dir"' EXIT INT TERM
|
||||||
|
|
||||||
|
ssh_key="$key_dir/deploy_key"
|
||||||
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
||||||
chmod 600 "$ssh_key"
|
chmod 600 "$ssh_key"
|
||||||
|
|
||||||
ssh -i "$ssh_key" -p "$deploy_port" \
|
ssh -i "$ssh_key" -p "$deploy_port" \
|
||||||
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
|
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
|
||||||
"${DEPLOY_USER}@${DEPLOY_HOST}" \
|
"${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||||
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} STAGE=$1 bash -se" <<'EOF'
|
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} STAGE=$1 bash -se" <<'EOF'
|
||||||
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||||
run_stage "$STAGE"
|
run_stage "$STAGE"
|
||||||
EOF
|
EOF
|
||||||
Reference in new issue
Block a user