fix(ci): skip heavy jobs on renovate branches, automerge digest and patch
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / scan-deps (push) Canceled after 0s
ci / test-backend (push) Canceled after 0s
ci / test-frontend (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Canceled after 0s

Renovate branches only carry version/digest bumps, so scan-deps, test-backend, test-frontend and build just burn runner time on the box that also serves prod. Static checks and validate still run. Digest and patch updates automerge (playwright, helm and major rules below still override to no-automerge). Also replaces deprecated helm --atomic with --wait --rollback-on-failure.
This commit is contained in:
forust committed 2026-09-28 22:19:57 +02:00
1 parent 86730ff0c4
commit 76f39da90c
5 files changed
+29 -10

No files matched your search

+9 -1
View File
@@ -221,6 +221,10 @@ jobs:
# Deleting an entry here is how you accept a new advisory, so the diff says
# so out loud.
scan-deps:
# Renovate branches only ever carry version/digest bumps: nothing here can
# change the shipped dependency tree, so the audits would just burn runner
# time on the same box that serves prod. Static checks still run.
if: ${{ !startsWith(github.head_ref, 'renovate/') && !startsWith(github.ref_name, 'renovate/') }}
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
@@ -268,6 +272,8 @@ jobs:
npm audit --omit=dev --audit-level=high
test-backend:
# Same as scan-deps: renovate bumps cannot break panel tests.
if: ${{ !startsWith(github.head_ref, 'renovate/') && !startsWith(github.ref_name, 'renovate/') }}
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
@@ -313,6 +319,8 @@ jobs:
"$venv/bin/python" -m pytest tests/ -q
test-frontend:
# Same as scan-deps: renovate bumps cannot break panel tests.
if: ${{ !startsWith(github.head_ref, 'renovate/') && !startsWith(github.ref_name, 'renovate/') }}
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
@@ -482,7 +490,7 @@ jobs:
test-frontend,
validate,
]
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev') && !startsWith(github.ref_name, 'renovate/')
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 60
outputs:
+8 -7
View File
@@ -489,7 +489,7 @@ rollback_workloads() {
local -a recovered=()
while read -r kind ns name; do
[ -n "${kind:-}" ] || continue
# Helm-owned workloads are already rolled back by the release's --atomic
# Helm-owned workloads are already rolled back by the release's --rollback-on-failure
# upgrade. `rollout undo` here would step back to the revision Helm just
# escaped (the failed one), so leave them for the operator instead.
if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then
@@ -548,7 +548,7 @@ helm_release_status() {
}
# recover_pending_release <release> <namespace>
# Rolls a release out of a pending-* state left by a failed --atomic upgrade
# Rolls a release out of a pending-* state left by a failed upgrade with --rollback-on-failure
# whose own rollback never completed. Without this every future upgrade errors
# out until a human runs `helm rollback`. Passes through releases that are not
# pending (deployed, failed, not-found). Returns non-zero when the release is
@@ -593,22 +593,23 @@ upgrade_helm_releases() {
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
log "Upgrading $release ($chart $version)"
# A previous --atomic run whose own rollback never finished leaves the
# A previous run with --rollback-on-failure whose own rollback never finished leaves the
# release in pending-*, which blocks every future upgrade. Recover first
# so one wedged revision cannot wedge the pipeline forever.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
return 1
fi
# --atomic rolls the release back when the upgrade times out or the workloads
# it touches never become ready, so a bad chart bump is not left half applied.
# --rollback-on-failure (+ --wait) rolls the release back when the upgrade
# times out or the workloads it touches never become ready, so a bad chart
# bump is not left half applied. (--atomic was this combo; deprecated.)
if ! helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \
--version "$version" \
--values "$REPO/$values" \
--atomic --cleanup-on-fail --timeout 10m; then
--wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then
echo "WARN: upgrade of $release failed, checking release state"
# --atomic already attempted its own rollback; finish the job when that
# --rollback-on-failure already attempted its own rollback; finish the job when that
# rollback never completed, otherwise the release stays pending-* and
# blocks every future run.
if ! recover_pending_release "$release" "$namespace"; then
+2 -2
View File
@@ -73,14 +73,14 @@ jobs:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
# Apply only, no verification, so this is just the work itself: snapshot,
# then sequential `helm upgrade --atomic --timeout 10m`, then the apply loop.
# then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop.
# Verification has its own job and its own budget.
#
# 45 is roughly four times the measured cost of the stage, which is
# deliberately not raised on a theory:
#
# helm, healthy 3 no-op upgrades ~3-5 min
# helm, one release bad --atomic spends its 10m, ~10-15 min
# helm, one release bad rollback-on-failure spends its 10m, ~10-15 min
# then rolls that one back
# apply loop ~40 manifests, 4 of which ~1 min
# resolve an image digest
+5
View File
@@ -160,6 +160,11 @@ data:
}
],
"packageRules": [
{
"description": "Automerge digest and patch updates - safe by definition, review adds nothing, keeps the renovate queue and the deploy line short. Specific no-automerge rules below still override this for playwright, helm and majors.",
"matchUpdateTypes": ["digest", "patch"],
"automerge": true
},
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
+5
View File
@@ -149,6 +149,11 @@
}
],
"packageRules": [
{
"description": "Automerge digest and patch updates - safe by definition, review adds nothing, keeps the renovate queue and the deploy line short. Specific no-automerge rules below still override this for playwright, helm and majors.",
"matchUpdateTypes": ["digest", "patch"],
"automerge": true
},
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],