refactor(ci): use native runner and durable incremental deploys
This commit is contained in:
1 parent
5f9354b9a8
commit
9a76529be8
25 files changed
+2090
-1063
No files matched your search
+67
-165
@@ -1,202 +1,104 @@
|
||||
name: deploy
|
||||
|
||||
on:
|
||||
# Deploy only what CI already validated. workflow_run is used instead of
|
||||
# workflow_dispatch so a red lint/validate run can never reach the cluster.
|
||||
workflow_run:
|
||||
workflows: [ci]
|
||||
branches: [main]
|
||||
types: [completed]
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
deploy_ref:
|
||||
description: "Commit already checked by successful main CI (main or SHA)"
|
||||
default: main
|
||||
required: true
|
||||
deploy_mode:
|
||||
description: "First deploy requires full; plan changes no production resources"
|
||||
type: choice
|
||||
options: [changed, full, plan]
|
||||
default: changed
|
||||
refresh_images:
|
||||
description: "Explicitly refresh mutable third-party Compose tags"
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
# The deploy jobs read the tree, then reach the cluster over SSH with the
|
||||
# deploy key. The Actions token itself is not part of that path, so it gets
|
||||
# read-only contents and no more.
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
|
||||
concurrency:
|
||||
group: deploy-main
|
||||
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
|
||||
# takes the verify job down with it, so a superseded deploy would leave the
|
||||
# cluster half-applied and unchecked — the exact failure the verify job exists
|
||||
# to catch. kubectl apply and docker compose up are both idempotent, so letting
|
||||
# the older run finish and then deploying the newer commit costs little.
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
||||
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
|
||||
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
|
||||
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
||||
DEPLOY_HOST: ${{ vars.DEPLOY_HOST || secrets.DEPLOY_HOST }}
|
||||
DEPLOY_PORT: ${{ vars.DEPLOY_PORT || secrets.DEPLOY_PORT }}
|
||||
DEPLOY_USER: ${{ vars.DEPLOY_USER || secrets.DEPLOY_USER }}
|
||||
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
||||
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
||||
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
|
||||
# finished ci run checked. Pin the exact validated commit instead, so a push
|
||||
# landing mid-deploy cannot make the workstation deploy something else. Also
|
||||
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
|
||||
# which falls back to the current origin/main.
|
||||
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
DEPLOY_KNOWN_HOSTS: ${{ vars.DEPLOY_KNOWN_HOSTS }}
|
||||
DEPLOY_RUN_ID: ${{ github.run_id }}-${{ github.run_attempt || 1 }}
|
||||
DEPLOY_MODE: ${{ inputs.deploy_mode || 'changed' }}
|
||||
REFRESH_IMAGES: ${{ inputs.refresh_images && 'true' || 'false' }}
|
||||
|
||||
jobs:
|
||||
preflight:
|
||||
# Autodeploy defaults to OFF: pushes deploy only when the AUTODEPLOY repo
|
||||
# variable is set to 'true' (Settings -> Actions -> Variables). A manual
|
||||
# Run workflow always bypasses the switch: dispatching it is the explicit
|
||||
# intent to deploy.
|
||||
gate:
|
||||
if: >-
|
||||
github.ref == 'refs/heads/main' &&
|
||||
(vars.AUTODEPLOY == 'true' || github.event_name == 'workflow_dispatch') &&
|
||||
(github.event_name != 'workflow_run' ||
|
||||
(github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.head_branch == 'main'))
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
(github.event.workflow_run.conclusion == 'success' && github.event.workflow_run.head_branch == 'main'))
|
||||
runs-on: homelab
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
sha: ${{ steps.release.outputs.sha }}
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Check successful CI and download the exact commit release
|
||||
id: release
|
||||
env:
|
||||
GITEA_TOKEN: ${{ github.token }}
|
||||
DEPLOY_REF: ${{ inputs.deploy_ref || 'main' }}
|
||||
EVENT_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA"
|
||||
- name: Submit durable deploy to workstation
|
||||
run: bash .gitea/workflows/ssh-run.sh start
|
||||
|
||||
- name: Fetch and reset workstation
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh preflight
|
||||
|
||||
validate:
|
||||
needs: [preflight]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 20
|
||||
apply:
|
||||
needs: [gate]
|
||||
runs-on: homelab
|
||||
timeout-minutes: 100
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
- name: Checkout checked commit
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
with:
|
||||
ref: ${{ needs.gate.outputs.sha }}
|
||||
- name: Follow validation and sequential Kubernetes / Compose apply
|
||||
run: bash .gitea/workflows/ssh-run.sh apply
|
||||
|
||||
- name: Dry-run manifests and check Secrets
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh validate
|
||||
|
||||
apply-k8s:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# Apply only, no verification, so this is just the work itself: snapshot,
|
||||
# then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop.
|
||||
# Verification has its own job and its own budget.
|
||||
#
|
||||
# 45 is roughly four times the measured cost of the stage, which is
|
||||
# deliberately not raised on a theory:
|
||||
#
|
||||
# helm, healthy 3 no-op upgrades ~3-5 min
|
||||
# helm, one release bad rollback-on-failure spends its 10m, ~10-15 min
|
||||
# then rolls that one back
|
||||
# apply loop ~40 manifests, 4 of which ~1 min
|
||||
# resolve an image digest
|
||||
# restart_stale_images 7.6s to find 8 workloads, ~0.5 min
|
||||
# 9.8s to resolve their digests
|
||||
#
|
||||
# The helm figure is one release, not three: `set -e` aborts
|
||||
# upgrade_helm_releases on the first failure, so a broken release costs
|
||||
# 10m and the other two are never attempted. Multiplying 10m by three
|
||||
# overstates the worst case by 20 minutes.
|
||||
#
|
||||
# The 45 minutes this was last raised to 45 were still not enough, and the
|
||||
# job logs for those runs no longer exist, so what actually consumed the
|
||||
# budget is not known - the two measurable candidates above account for
|
||||
# ~15 of it. The unbounded `docker manifest inspect` against the registry's
|
||||
# known hang mode is now bounded inside registry_digest (25s timeout, 3
|
||||
# attempts): a dead registry fails each owned image after ~85s instead of
|
||||
# hanging the stage, and a blinking one is retried instead of failing the
|
||||
# whole apply file. Still open: make the stage announce which manifest it
|
||||
# is working on, so a killed run leaves a diagnosable last line.
|
||||
timeout-minutes: 45
|
||||
verify:
|
||||
needs: [gate, apply]
|
||||
if: always() && needs.gate.result == 'success'
|
||||
runs-on: homelab
|
||||
timeout-minutes: 130
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
- name: Checkout checked commit
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
with:
|
||||
ref: ${{ needs.gate.outputs.sha }}
|
||||
- name: Follow workload verification and recovery
|
||||
run: bash .gitea/workflows/ssh-run.sh verify
|
||||
|
||||
- name: Apply Kubernetes manifests
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh apply-k8s
|
||||
|
||||
apply-compose:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Redeploy docker compose stacks
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh apply-compose
|
||||
|
||||
# Watches the workloads this deploy changed and rolls back the ones that never
|
||||
# became healthy. Runs even when the apply jobs failed, timed out or were
|
||||
# cancelled — that is the whole point of splitting it out. `always()` is what
|
||||
# lets it start after a failed dependency; the needs on apply-compose are a
|
||||
# barrier, so verification begins only once both applies are done.
|
||||
verify-k8s:
|
||||
needs: [preflight, apply-k8s, apply-compose]
|
||||
if: >-
|
||||
always() &&
|
||||
needs.preflight.result == 'success' &&
|
||||
needs.apply-k8s.result != 'skipped' &&
|
||||
needs.apply-compose.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# Not raised, because the arithmetic does not close.
|
||||
#
|
||||
# 32 workloads are under management and the wave width is 8, so the verify
|
||||
# itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when
|
||||
# every rollout times out rather than converging. That is already 20 of 30.
|
||||
#
|
||||
# The other 10 would have to absorb rollback, and rollback_workloads is a
|
||||
# serial `while read` loop at 300s per failed workload. 10 minutes buys two.
|
||||
# Any larger number is buying a bigger multiple of an unbounded term rather
|
||||
# than covering a known cost: 60 minutes buys eight, and 60 minutes is
|
||||
# therefore not a bound, it is a guess with two digits.
|
||||
#
|
||||
# The number becomes derivable the moment rollback uses the same wave width
|
||||
# as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160,
|
||||
# and 45 covers verify plus rollback at full width. That change is to the
|
||||
# recovery path and is not folded into a timeout edit.
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Verify workloads and roll back on failure
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh verify-k8s
|
||||
|
||||
# Asks the public route of every active service whether it is actually
|
||||
# serving, which the rollout check above structurally cannot: a pod can
|
||||
# converge and still be crash-looping, or be listening on a port no Service
|
||||
# points at, or answer 500.
|
||||
#
|
||||
# `always()` for the same reason verify-k8s has it, and it runs after that job
|
||||
# specifically because a rollback is when a route most needs re-checking. The
|
||||
# needs is a barrier, not a filter: whether verify-k8s passed, failed or was
|
||||
# cancelled, the probes are what say whether the cluster is serving, and
|
||||
# suppressing them on a rollback would hide the one run where the answer
|
||||
# matters most.
|
||||
smoke:
|
||||
needs: [preflight, verify-k8s]
|
||||
if: >-
|
||||
always() &&
|
||||
needs.preflight.result == 'success' &&
|
||||
needs.verify-k8s.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 10
|
||||
needs: [gate, verify]
|
||||
if: always() && needs.gate.result == 'success'
|
||||
runs-on: homelab
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
- name: Checkout checked commit
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Probe the public route of every active service
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh smoke
|
||||
with:
|
||||
ref: ${{ needs.gate.outputs.sha }}
|
||||
- name: Follow public route checks
|
||||
run: bash .gitea/workflows/ssh-run.sh smoke
|
||||
Reference in new issue
Block a user