Two things, both about not finding out late. No workflow declared `permissions`, so all eighteen jobs across the four workflows ran on a token with the default full repository scope. Every one of them only checks out code, and deploy reaches the cluster over SSH with the deploy key, and Renovate writes through its own bot PAT rather than the Actions token. So `contents: read` is all any of them needed. The panel image ships 15 known advisories and nothing was looking. Add a scan-deps job that fails on anything new, and record the eight current ones by ID in the workflow. It is a list rather than a baseline count so that the diff that accepts an advisory says so in words, and it lives in our workflow instead of the package manifest so a subtree sync from forust/userbot cannot quietly widen the exemption. Both halves were checked to fail on a regression, not just to pass today: removing one --ignore-vuln turns the Python step red, and dropping --audit-level to moderate turns the npm one red on the devalue advisory. npm audits production dependencies only. All seven findings in the full tree are build- or test-time: the esbuild advisory needs a vite dev server exposed to the internet, and nanoid's infinite loop needs a custom generator called with size 0, which postcss does not do. None are in the 91 kB bundle the panel serves, so failing on them would be noise that trains people to ignore the job. The starlette entries are the reason the job is not "fail on everything": fastapi 0.115.12 pins starlette<0.47.0 and the last four fixes need 0.49.1 through 1.3.1, so clearing them is a jump to fastapi 0.141.x and is upstream's call, not a drive-by. Four of the seven are reachable in principle, which the comment on the job sets out. The panel answers only on userbot.workstation.internal with no public route, which is what keeps those four from being an internet-facing DoS. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
126 lines
4.5 KiB
YAML
126 lines
4.5 KiB
YAML
name: deploy
|
|
|
|
on:
|
|
# Deploy only what CI already validated. workflow_run is used instead of
|
|
# workflow_dispatch so a red lint/validate run can never reach the cluster.
|
|
workflow_run:
|
|
workflows: [ci]
|
|
types: [completed]
|
|
workflow_dispatch:
|
|
|
|
# The deploy jobs read the tree, then reach the cluster over SSH with the
|
|
# deploy key. The Actions token itself is not part of that path, so it gets
|
|
# read-only contents and no more.
|
|
permissions:
|
|
contents: read
|
|
|
|
concurrency:
|
|
group: deploy-main
|
|
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
|
|
# takes the verify job down with it, so a superseded deploy would leave the
|
|
# cluster half-applied and unchecked — the exact failure the verify job exists
|
|
# to catch. kubectl apply and docker compose up are both idempotent, so letting
|
|
# the older run finish and then deploying the newer commit costs little.
|
|
cancel-in-progress: false
|
|
|
|
env:
|
|
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
|
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
|
|
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
|
|
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
|
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
|
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
|
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
|
|
# finished ci run checked. Pin the exact validated commit instead, so a push
|
|
# landing mid-deploy cannot make the workstation deploy something else. Also
|
|
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
|
|
# which falls back to the current origin/main.
|
|
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
|
|
|
|
jobs:
|
|
preflight:
|
|
if: >-
|
|
github.event_name != 'workflow_run' ||
|
|
(github.event.workflow_run.conclusion == 'success' &&
|
|
github.event.workflow_run.head_branch == 'main')
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
timeout-minutes: 10
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Fetch and reset workstation
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh preflight
|
|
|
|
validate:
|
|
needs: [preflight]
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
timeout-minutes: 20
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Dry-run manifests and check Secrets
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh validate
|
|
|
|
apply-k8s:
|
|
needs: [validate]
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
# Apply only, no verification, so this is just the work itself: snapshot,
|
|
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
|
|
# apply loop. Verification has its own job and its own budget.
|
|
timeout-minutes: 45
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Apply Kubernetes manifests
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh apply-k8s
|
|
|
|
apply-compose:
|
|
needs: [validate]
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
timeout-minutes: 30
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Redeploy docker compose stacks
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh apply-compose
|
|
|
|
# Watches the workloads this deploy changed and rolls back the ones that never
|
|
# became healthy. Runs even when the apply jobs failed, timed out or were
|
|
# cancelled — that is the whole point of splitting it out. `always()` is what
|
|
# lets it start after a failed dependency; the needs on apply-compose are a
|
|
# barrier, so verification begins only once both applies are done.
|
|
verify-k8s:
|
|
needs: [apply-k8s, apply-compose]
|
|
if: >-
|
|
always() &&
|
|
needs.apply-k8s.result != 'skipped' &&
|
|
needs.apply-compose.result != 'skipped'
|
|
runs-on: [self-hosted, linux, arch, homelab, prod]
|
|
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
|
|
timeout-minutes: 30
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
|
|
|
- name: Verify workloads and roll back on failure
|
|
shell: bash
|
|
run: |
|
|
set -euo pipefail
|
|
./.gitea/workflows/ssh-run.sh verify-k8s
|