Compare commits

..
Author SHA1 Message Date
forust 73d2af73e5 fix(cicd): support Helm 4 release listing
ci / build (pull_request) Skipped
ci / Workflows (pull_request) Successful in 7s
ci / Python and tests (pull_request) Successful in 5s
ci / Compose (pull_request) Successful in 11s
ci / Shell (pull_request) Successful in 17s
ci / Formatting (pull_request) Successful in 17s
ci / YAML (pull_request) Successful in 8s
ci / Dockerfiles (pull_request) Successful in 5s
ci / Kubernetes (pull_request) Successful in 7s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
2026-10-07 15:51:57 +02:00
forust 64253e005e Merge pull request 'fix(cicd): use Gitea artifact v4 backend' (#109) from fix/gitea-v4-artifacts into main
ci / Workflows (push) Successful in 8s
ci / Shell (push) Successful in 22s
ci / YAML (push) Successful in 9s
ci / image-plan (push) Successful in 49s
ci / Compose (push) Successful in 14s
ci / Formatting (push) Successful in 19s
ci / Python and tests (push) Successful in 8s
ci / Dockerfiles (push) Successful in 5s
ci / Kubernetes (push) Successful in 8s
ci / Image (error-pages) (push) Successful in 39s
ci / Image (forust-homepage) (push) Successful in 16s
ci / Image (xdfnx-homepage) (push) Successful in 13s
ci / build (push) Successful in 17s
Reviewed-on: #109
2026-10-07 13:35:36 +00:00
forust 69accd1752 fix(cicd): use Gitea artifact v4 backend
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Kubernetes (pull_request) Successful in 7s
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Workflows (pull_request) Successful in 7s
ci / Shell (pull_request) Successful in 18s
ci / Formatting (pull_request) Successful in 20s
ci / Python and tests (pull_request) Successful in 7s
ci / YAML (pull_request) Successful in 8s
ci / Dockerfiles (pull_request) Successful in 4s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-07 15:29:49 +02:00
forust 0cf4b08a95 Merge pull request 'fix(cicd): isolate pull request runner jobs' (#108) from fix/cicd-pr-runner into main
ci / Workflows (push) Successful in 7s
ci / Shell (push) Successful in 21s
ci / Python and tests (push) Successful in 5s
ci / Compose (push) Successful in 15s
ci / Formatting (push) Successful in 17s
ci / Kubernetes (push) Successful in 6s
ci / YAML (push) Successful in 9s
ci / Dockerfiles (push) Successful in 4s
renovate-ci / validate-renovate (push) Successful in 2m21s
ci / image-plan (push) Successful in 18s
ci / Image (error-pages) (push) Successful in 13s
ci / Image (xdfnx-homepage) (push) Successful in 12s
ci / Image (forust-homepage) (push) Successful in 11s
ci / build (push) Successful in 16s
Reviewed-on: #108
2026-10-07 13:03:01 +00:00
forust 86df5d9048 docs(cicd): document user-scoped PR runner
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (pull_request) Successful in 14s
ci / Shell (pull_request) Successful in 33s
ci / Formatting (pull_request) Successful in 36s
ci / YAML (pull_request) Successful in 28s
ci / Kubernetes (pull_request) Successful in 11s
ci / YAML (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 23s
ci / Python and tests (pull_request) Successful in 16s
ci / Dockerfiles (pull_request) Successful in 11s
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
ci / image-plan (pull_request) Skipped
2026-10-07 14:50:32 +02:00
forust d7441bbbc2 fix(cicd): isolate pull request runner jobs
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Compose (pull_request) Successful in 37s
ci / Shell (pull_request) Successful in 18s
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / Workflows (pull_request) Successful in 8s
ci / Python and tests (pull_request) Successful in 11s
ci / Kubernetes (pull_request) Successful in 8s
ci / Formatting (pull_request) Successful in 18s
ci / YAML (pull_request) Successful in 11s
ci / Dockerfiles (pull_request) Successful in 7s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-07 14:39:57 +02:00
forust 2be089e048 Merge pull request 'Collect Headscale, NetBird, Gitea and Immich metrics' (#100) from feat/service-metrics into main
ci / Compose (push) Successful in 11s
ci / Workflows (push) Successful in 7s
ci / YAML (push) Successful in 9s
ci / Dockerfiles (push) Successful in 5s
ci / Shell (push) Successful in 16s
ci / Formatting (push) Successful in 17s
ci / Python and tests (push) Successful in 7s
ci / Kubernetes (push) Successful in 6s
ci / image-plan (push) Successful in 15s
ci / Image (error-pages) (push) Successful in 12s
ci / Image (forust-homepage) (push) Successful in 12s
ci / Image (xdfnx-homepage) (push) Successful in 11s
ci / build (push) Successful in 13s
Reviewed-on: #100
2026-10-07 11:46:16 +00:00
forust 83b2e68371 feat(metrics): collect Headscale, NetBird, Gitea and Immich metrics 2026-10-07 11:46:16 +00:00
forust 597f64cbb0 Merge pull request 'Show CI checks and deployment results' (#99) from codex/ci-visible-checks into main
ci / Workflows (push) Successful in 7s
ci / Formatting (push) Successful in 17s
ci / Python and tests (push) Successful in 7s
ci / YAML (push) Successful in 9s
ci / Compose (push) Successful in 11s
ci / Shell (push) Successful in 16s
ci / Kubernetes (push) Successful in 7s
ci / Dockerfiles (push) Successful in 5s
ci / Image (forust-homepage) (push) Successful in 12s
ci / Image (xdfnx-homepage) (push) Successful in 11s
renovate-ci / validate-renovate (push) Successful in 11s
ci / image-plan (push) Successful in 16s
ci / Image (error-pages) (push) Successful in 41s
ci / build (push) Successful in 14s
Reviewed-on: #99
2026-10-07 11:46:05 +00:00
forust f767f3ce1a docs: update EDU handoff status
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Shell (pull_request) Successful in 15s
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (pull_request) Successful in 6s
ci / Formatting (pull_request) Successful in 16s
ci / Python and tests (pull_request) Successful in 6s
ci / YAML (pull_request) Successful in 7s
ci / Dockerfiles (pull_request) Successful in 5s
ci / build (pull_request) Skipped
ci / Kubernetes (pull_request) Successful in 6s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 9s
2026-10-07 10:25:17 +02:00
forust 95e4d8f875 ci: add per-image jobs to homelab CI
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Workflows (pull_request) Successful in 7s
ci / Shell (pull_request) Successful in 15s
ci / Formatting (pull_request) Successful in 15s
ci / Python and tests (pull_request) Successful in 5s
ci / YAML (pull_request) Successful in 7s
ci / Dockerfiles (pull_request) Successful in 5s
ci / Kubernetes (pull_request) Successful in 6s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 11s
2026-10-07 10:19:12 +02:00
forust 1d93588e06 refactor: remove EDU ownership from homelab 2026-10-07 10:19:12 +02:00
forust fb400eea6e Use plain text for the deploy request details
ci / Compose (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / YAML (pull_request) Successful in 11s
ci / build (pull_request) Skipped
ci / Workflows (push) Skipped
ci / Python and tests (push) Skipped
ci / Compose (pull_request) Successful in 13s
ci / Workflows (pull_request) Successful in 9s
ci / Shell (pull_request) Successful in 21s
ci / Formatting (pull_request) Successful in 22s
ci / Python and tests (pull_request) Successful in 6s
ci / Dockerfiles (pull_request) Successful in 5s
ci / Kubernetes (pull_request) Successful in 7s
2026-10-06 23:44:24 +02:00
forust 0c76426c17 Show failure details in CI and deploy summaries
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (push) Skipped
ci / Shell (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Workflows (pull_request) Failing after 8s
ci / Formatting (pull_request) Canceled after 0s
ci / Python and tests (pull_request) Canceled after 0s
ci / YAML (pull_request) Canceled after 0s
ci / Dockerfiles (pull_request) Canceled after 0s
ci / Kubernetes (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
ci / Shell (pull_request) Canceled after 7s
2026-10-06 23:44:00 +02:00
forust d74822cd27 Add CI and deploy summaries
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Kubernetes (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Compose (pull_request) Successful in 14s
ci / Workflows (pull_request) Successful in 7s
ci / Shell (pull_request) Successful in 23s
ci / Formatting (pull_request) Successful in 26s
ci / Python and tests (pull_request) Successful in 8s
ci / YAML (pull_request) Successful in 14s
ci / Dockerfiles (pull_request) Successful in 7s
ci / Kubernetes (pull_request) Successful in 10s
ci / build (pull_request) Skipped
2026-10-06 23:28:56 +02:00
forust b677d553b4 refactor(ci): split checks into visible jobs
ci / YAML (pull_request) Successful in 11s
ci / Dockerfiles (pull_request) Successful in 6s
ci / Kubernetes (pull_request) Successful in 6s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (push) Skipped
renovate-ci / validate-renovate (pull_request) Skipped
ci / Compose (pull_request) Successful in 10s
ci / Workflows (pull_request) Successful in 5s
ci / Shell (pull_request) Successful in 22s
ci / Formatting (pull_request) Successful in 21s
ci / Python and tests (pull_request) Successful in 8s
2026-10-06 23:22:15 +02:00
51 changed files with 1325 additions and 2824 deletions

No files matched your search

+42
View File
@@ -0,0 +1,42 @@
# EDU ownership handoff
## Current status
EDU PR #1 merged at 2026-10-07 08:04:30 UTC. Main release `5094952464ce315130839303985fd04d721bc1f2` passed CI run 1585 and deploy run 1586. The workstation checkout `/srv/edu-master` is at that SHA. The release changed the application image digests:
- Session keeper: `sha256:1e59473bd40fe4c22622017d808a8927a68788275fe073dc23d718c44b2fd5dd`
- Webinar checker: `sha256:987d9bf0770272766523ea5b94c7f3f849175d737551d46591ae55e058cf9f12`
The live workloads remain healthy in context `Default`, namespace `edu-master`. Both health and live probes return 200. Redis AUTH passes, session TTL is 1178 seconds, the delivery backlog is zero, all nine EDU alert rules are healthy, and the scrape target is UP. The unauthorized-pod Redis check passed. The Redis PVC UID and Secret UID and values, including the Fernet key, match their pre-release state.
The homelab EDU active marker was present after the EDU deployment. It was moved to the private snapshot as `homelab-k8s-active.marker` while holding `/tmp/homelab-apply.lock`. The homelab checkout at `/srv/homelab` is at `5f9354b` and has the tracked marker deletion. Its deploy preflight blocks a dirty checkout until this removal is reconciled. Preserve private ignored configuration when syncing that checkout.
The remaining homelab change is PR #105, branch `feat/edu-handoff-matrix`, based on `codex/ci-visible-checks`. Its eight protected checks passed. Renovate runs 1587 and 1588 passed. Image publishing was skipped for the PR. The EDU runtime changes are in PR #3 from `fix/handoff-runtime` to `main`; CI run 1589 is in progress. Those runtime changes have not been released.
## Approval gate and next steps
PR #99 must merge before PR #105 can target `main`. A merge attempt for PR #99 returned HTTP 405 because it needs one approval; the protected branch has `required_approvals=1` and whitelist approval is enabled. This approval gate prevents the remaining transfer steps.
After the required approval:
1. Merge PR #99.
2. Retarget PR #105 to `main`. Complete CI and review, then approve and merge it.
3. Under the homelab apply lock, sync `/srv/homelab` to the merged removal. Preserve private ignored configuration and keep the active marker removed. Confirm the deploy preflight is clean.
4. Merge the EDU runtime PR after its CI and review pass. The main-push CI run must complete successfully before its exact SHA can deploy.
5. Verify the new release SHA, image digests, workload health, Redis AUTH and TTL, backlog, PVC and Secret identity, and monitoring. Record the results in the EDU PR.
`AUTODEPLOY=false` is explicitly configured. The EDU repository path and port secrets are confirmed, and `EDU_KUBE_CONTEXT=Default` is configured as a repository variable. Keep deployment and registry credentials outside Git. Never run both homelab and EDU deployment paths at the same time.
## Change summary
The homelab PR removes the EDU subtree, deployment and image selection, rollback and verification cases, route probes, Renovate references, and external-image exceptions. It adds a serial dynamic matrix for the three homelab images. Each job builds an image or reuses a matching immutable digest. The final job checks all image results and publishes full-SHA tags and the existing release artifact only after they pass. PRs do not publish images. The protected check names from PR #99 are preserved. PR #100's service-metrics work is independent of this handoff.
The EDU runtime PR adds the Playwright service manifest, reconciles Redis storage and Secret reload annotations, and adds pre-apply Redis backup and identity checks. It verifies application endpoints, Redis AUTH, session TTL, metrics, and all nine vmalert rules. Rollback checks workload and application health and reports when manual recovery is needed. Its deployment guard rejects an unexpected or dirty checkout and refuses deployment while either legacy homelab EDU marker exists.
## Rollback and limits
The private snapshot is `/home/forust/.local/state/edu-master-deploy/handoff-20261007T080838Z` on the workstation. It contains the pre-handoff Redis RDB and recovery data. RDB checksum verification confirmed twelve keys. Keep the snapshot outside Git. For an EDU release failure, restore the saved Kubernetes resources and inspect application health. The rollback does not automatically restore the Redis RDB; restore old Redis data only when recovery requires it.
For an ownership rollback, stop EDU deployment triggers first, restore the reviewed homelab source and marker, then reapply recorded immutable image digests. Verify both workload and application health. Never delete or recreate the Redis PVC.
The initial EDU release and the homelab marker move are complete. PR #99 approval and merge, PR #105 retarget and merge, homelab checkout reconciliation, EDU runtime PR merge, and release of those runtime changes remain pending. Synthetic Telegram delivery and Alertmanager-to-Telegram notification were not tested.
+1
View File
@@ -7,4 +7,5 @@ self-hosted-runner:
labels:
- arch
- homelab
- homelab-pr
- prod
+42 -5
View File
@@ -1,8 +1,20 @@
# Homelab CI/CD
The native Gitea runner runs on **vps**; production runs on **workstation**.
Jobs run on `homelab:host`, one at a time. No job images or Kubernetes credentials
are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows.
The native Gitea runners run on **vps**; production runs on **workstation**.
Main-branch checks and image builds use `homelab:host`. Pull request and
non-main checks use `homelab-pr:host` under a separate account without Docker
access. The `homelab-pr` runner is registered at User scope for `forust`, so
any repository under that account can schedule jobs that request this label.
Each runner accepts one job at a time; the build waits for every check to pass.
CI and deploy runs also show a summary with
the release SHA, image build or reuse results, deploy mode, selected services,
and image digests. Failed runs keep a summary of completed image builds, stage
results, apply results, and recorded Kubernetes recovery. The final deploy
summary is in the smoke job; earlier jobs show the state observed at that time.
Apply success is separate from health and recovery. Update the installed
workstation controller with `setup-workstation.sh` when no deploy is running.
No job images or Kubernetes credentials are needed on the VPS. Builds use one
pinned BuildKit helper container. CI and deploy are separate workflows.
## Runner installation
@@ -28,6 +40,32 @@ pushes directly to the registry, and caps retained local cache at 1 GiB with a
2 GiB free-space target. This is not a hard limit on peak build disk usage.
Nothing runs `docker system prune`, removes unrelated images, or deletes volumes.
### Pull request runner
Install the unprivileged host runner on the VPS:
```sh
sudo bash .gitea/runner/setup-pr-runner.sh
```
Get a registration token from the user Actions runner settings. Run the
installer in a terminal. It asks for the token without echoing it, registers the
runner as `homelab-pr` with label `homelab-pr:host`, then enables the service.
The work directory is `/var/lib/gitea-pr-runner`. Confirm that Gitea lists the
runner as User scope before merging the workflow change. An unmatched label can
fall back to the default job image.
Renovate PR validation uses `pull_request_target`, which reads the workflow from
the base branch. It checks out the PR head only after runner selection and runs
that code on `homelab-pr`. Keep this workflow read-only and do not add secrets.
The PR runner has a separate home and tool cache. Do not add it to the `docker`
group or give it access to `/var/run/docker.sock`. It runs repository code from
pull requests, so keep its registration and permissions separate from the
trusted `homelab` runner. This separates users and host permissions, but both
runners still share the VPS kernel and network. Use a disposable VM if PRs from
untrusted external authors must be fully isolated.
## Workstation setup
As the existing SSH deploy user on workstation:
@@ -58,8 +96,7 @@ The deploy user's existing Docker registry authentication remains necessary.
CI publishes `release-<full SHA>` as a Gitea artifact with all three owned image
digests and build input fingerprints. Unchanged images are reused only from a
successful main CI artifact, never from `:prod`. EDU images remain pinned to the
digests released by their application repository. Expired artifacts cause CI to
successful main CI artifact, never from `:prod`. Expired artifacts cause CI to
rebuild images; they block deployment until CI is rerun.
Run deploy from main with `deploy_ref=main` or a checked SHA:
+8
View File
@@ -0,0 +1,8 @@
runner:
file: /var/lib/gitea-pr-runner/.runner
capacity: 1
timeout: 5h
labels:
- homelab-pr:host
cache:
enabled: false
+27
View File
@@ -0,0 +1,27 @@
[Unit]
Description=Gitea Actions untrusted pull request runner
After=network-online.target
Wants=network-online.target
[Service]
User=gitea-pr-runner
Group=gitea-pr-runner
WorkingDirectory=/var/lib/gitea-pr-runner
Environment=HOME=/var/lib/gitea-pr-runner
Environment=PATH=/var/lib/gitea-pr-runner/.cache/homelab-ci/bin:/usr/local/bin:/usr/bin:/bin
ExecStart=/usr/local/bin/gitea-runner daemon --config /etc/gitea-pr-runner/config.yaml
Restart=on-failure
RestartSec=5
NoNewPrivileges=yes
PrivateTmp=yes
ProtectSystem=full
ProtectHome=yes
ProtectKernelTunables=yes
ProtectKernelModules=yes
ProtectControlGroups=yes
RestrictSUIDSGID=yes
LockPersonality=yes
UMask=0077
[Install]
WantedBy=multi-user.target
+56
View File
@@ -0,0 +1,56 @@
#!/usr/bin/env bash
# Install a native runner for untrusted PR jobs without Docker access.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
[ "$(id -u)" -eq 0 ] || { echo 'Run with sudo on the runner host' >&2; exit 1; }
for tool in cp cut date getent id install runuser systemctl useradd; do
command -v "$tool" >/dev/null || { echo "Install missing prerequisite: $tool" >&2; exit 1; }
done
command -v /usr/local/bin/gitea-runner >/dev/null || {
echo 'Install gitea-runner 3.0.2 at /usr/local/bin/gitea-runner first' >&2
exit 1
}
id gitea-pr-runner >/dev/null 2>&1 || \
useradd --system --create-home --home-dir /var/lib/gitea-pr-runner --shell /usr/bin/bash gitea-pr-runner
runner_home="$(getent passwd gitea-pr-runner | cut -d: -f6)"
[ "$runner_home" = /var/lib/gitea-pr-runner ] || {
echo 'Unexpected PR runner home; inspect the existing service first' >&2
exit 1
}
case " $(id -nG gitea-pr-runner) " in
*' docker '*)
echo 'The PR runner account must not belong to the docker group' >&2
exit 1
;;
esac
install -d -m 0755 /etc/gitea-pr-runner
stamp="$(date -u +%Y%m%dT%H%M%SZ)"
for existing in /etc/gitea-pr-runner/config.yaml /etc/systemd/system/gitea-pr-runner.service; do
[ ! -f "$existing" ] || cp -p "$existing" "$existing.before-$stamp"
done
install -m 0644 "$here/pr-config.yaml" /etc/gitea-pr-runner/config.yaml
install -m 0644 "$here/pr-runner.service" /etc/systemd/system/gitea-pr-runner.service
if [ ! -f /var/lib/gitea-pr-runner/.runner ]; then
read -r -s -p 'Enter the Gitea repository runner registration token: ' runner_token
printf '\n'
[ -n "$runner_token" ] || { echo 'Runner token is required' >&2; exit 1; }
export GITEA_RUNNER_REGISTRATION_TOKEN="$runner_token"
unset runner_token
runuser --preserve-environment -u gitea-pr-runner -- \
/usr/local/bin/gitea-runner register \
--config /etc/gitea-pr-runner/config.yaml \
--instance https://gitea.forust.xyz \
--name homelab-pr \
--labels homelab-pr:host \
--no-interactive
unset GITEA_RUNNER_REGISTRATION_TOKEN
fi
chmod 0600 /var/lib/gitea-pr-runner/.runner
systemctl daemon-reload
systemctl enable --now gitea-pr-runner.service
systemctl restart gitea-pr-runner.service
echo "PR runner ready. Configuration backups: *.before-$stamp"
+364 -16
View File
@@ -12,18 +12,14 @@ concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
jobs:
checks:
runs-on: homelab
timeout-minutes: 30
compose:
name: Compose
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: source
- name: Validate Compose files
shell: bash
run: |
@@ -52,11 +48,73 @@ jobs:
exit 1
fi
echo "checked ${#files[@]} Compose file(s)"
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Compose
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.source.conclusion == 'failure'
&& 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
workflows:
name: Workflows
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint shellcheck)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Lint Gitea Actions workflows with actionlint
shell: bash
run: |
set -euo pipefail
actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Workflows
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
shell:
name: Shell
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Lint shell scripts with ShellCheck
shell: bash
run: |
@@ -70,6 +128,37 @@ jobs:
fi
shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}"
bash .gitea/tests/deploy-validation.sh
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Shell
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
formatting:
name: Formatting
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Check formatting with Prettier
shell: bash
run: |
@@ -87,6 +176,37 @@ jobs:
fi
prettier --check --ignore-unknown "${prettier_files[@]}"
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Formatting
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
python:
name: Python and tests
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff jq)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Lint and format-check Python with Ruff
shell: bash
run: |
@@ -94,6 +214,37 @@ jobs:
ruff check . .gitea/workflows
ruff format --check . .gitea/workflows
python3 -m unittest discover -s tests -v
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Python and tests
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
yaml:
name: YAML
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Lint YAML syntax
shell: bash
run: |
@@ -111,6 +262,37 @@ jobs:
fi
yamllint -c .yamllint "${yaml_files[@]}"
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: YAML
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
dockerfiles:
name: Dockerfiles
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Lint Dockerfiles
shell: bash
run: |
@@ -126,6 +308,37 @@ jobs:
fi
hadolint -c .hadolint.yaml "${dockerfiles[@]}"
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Dockerfiles
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
kubernetes:
name: Kubernetes
runs-on: ${{ github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
id: source
- name: Prepare pinned tools
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
echo "$tools_dir" >> "$GITHUB_PATH"
id: tools
- name: Validate Kubernetes manifests against JSON schemas
shell: bash
run: |
@@ -146,27 +359,162 @@ jobs:
-ignore-missing-schemas \
-summary \
"${manifests[@]}"
build:
needs:
- checks
id: check
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Kubernetes
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP:
${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure'
&& 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
image-plan:
needs: [compose, workflows, shell, formatting, python, yaml, dockerfiles, kubernetes]
if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main'
runs-on: homelab
timeout-minutes: 60
timeout-minutes: 10
outputs:
matrix: ${{ steps.plan.outputs.matrix }}
steps:
- name: Checkout repository
id: source
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
with:
fetch-depth: 0
- name: Build changed images and write release
- name: Detect build inputs against successful CI
id: plan
env:
GITEA_TOKEN: ${{ github.token }}
run: python3 .gitea/workflows/release.py prepare --output build-plan.json
- name: Store the image plan
id: artifact
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: build-plan
path: build-plan.json
if-no-files-found: error
retention-days: 30
- name: Write the plan result
if: always()
env:
SUMMARY_CHECK: Image plan
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP: >-
${{ steps.plan.conclusion == 'failure' && 'Build input detection' ||
steps.artifact.conclusion == 'failure' && 'Plan upload' ||
steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## Image plan\n\nResult: %s\n' "$SUMMARY_RESULT" >>"$GITHUB_STEP_SUMMARY" || true
fi
images:
name: Image (${{ matrix.name }})
needs: [image-plan]
if: needs.image-plan.result == 'success'
runs-on: homelab
timeout-minutes: 60
strategy:
max-parallel: 1
fail-fast: false
matrix: ${{ fromJSON(needs.image-plan.outputs.matrix || '{"include":[{"name":"inactive"}]}') }}
steps:
- name: Checkout repository
id: source
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
- name: Download the checked image plan
id: inputs
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: build-plan
- name: Build or reuse this image
id: check
env:
IMAGE_NAME: ${{ matrix.name }}
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
run: python3 .gitea/workflows/release.py build
run: python3 .gitea/workflows/release.py image --image "$IMAGE_NAME" --output image.json
- name: Store the image result
id: artifact
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: image-${{ matrix.name }}
path: image.json
if-no-files-found: error
retention-days: 30
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Image (${{ matrix.name }})
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP: >-
${{ steps.check.conclusion == 'failure' && 'Build or tag images' ||
steps.artifact.conclusion == 'failure' && 'Artifact upload' ||
steps.inputs.conclusion == 'failure' && 'Artifact download' ||
steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
# Retain the build job name required by the immutable release deployment gate.
build:
needs: [image-plan, images]
runs-on: homelab
timeout-minutes: 15
steps:
- name: Checkout repository
id: source
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
- name: Download all image results
id: inputs
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
path: artifacts
- name: Pin SHA tags and write the complete release
id: check
env:
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
run: >-
python3 .gitea/workflows/release.py finalize
--plan artifacts/build-plan/build-plan.json
- name: Store commit release
uses: actions/upload-artifact@c6a366c94c3e0affe28c06c8df20a878f24da3cf
id: artifact
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: release-${{ github.sha }}
path: release.json
if-no-files-found: error
retention-days: 30
- name: Write the job result
if: always()
env:
SUMMARY_CHECK: Image release and SHA tags
SUMMARY_RESULT: ${{ job.status }}
SUMMARY_FAILED_STEP: >-
${{ steps.check.conclusion == 'failure' && 'Build or tag images' ||
steps.artifact.conclusion == 'failure' && 'Artifact upload' ||
steps.inputs.conclusion == 'failure' && 'Artifact download' ||
steps.source.conclusion == 'failure' && 'Source checkout' || '' }}
shell: bash
run: |
if [ -f .gitea/workflows/release.py ]; then
python3 .gitea/workflows/release.py check-summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true
fi
+102 -2
View File
@@ -141,7 +141,8 @@ def make_plan(directory):
planner = load_module('deploy_plan', source / '.gitea/workflows/deploy-plan.py')
request = json.loads((directory / 'request.json').read_text())
previous = json.loads((STATE / 'last-success.json').read_text()) if (STATE / 'last-success.json').exists() else None
helm = json.loads(command('helm', 'list', '--all', '-A', '-o', 'json'))
# Helm 4 lists every release status by default and removed the --all flag.
helm = json.loads(command('helm', 'list', '-A', '-o', 'json'))
plan = planner.make_plan(source, CONFIG_REPO, request['release'], previous, request['mode'], helm)
if request['refresh_images']:
plan['selected']['compose'] = plan['active']['compose']
@@ -212,14 +213,17 @@ def execute(run_id):
raise ValueError(f'Interrupted deploy {other.name}; run recover first')
status['state'] = 'running'
atomic_json(directory / 'status.json', status)
phase = 'plan'
try:
plan = make_plan(directory)
print(
json.dumps({'selected': plan['selected'], 'helm': plan['helm'], 'manual_removals': plan['removed']}),
flush=True,
)
phase = 'doctor'
if not stage(directory, 'doctor', 600):
raise RuntimeError('Preflight failed')
phase = 'validate'
if not stage(directory, 'validate', 1200):
raise RuntimeError('Validation failed')
if json.loads((directory / 'request.json').read_text())['mode'] == 'plan':
@@ -228,6 +232,7 @@ def execute(run_id):
atomic_json(directory / 'status.json', status)
return
# Budget includes both rollout checks and rollback waves, plus API overhead.
phase = 'Recovery budget'
count = int(
command(
'bash',
@@ -239,14 +244,24 @@ def execute(run_id):
verify_budget = max(600, 2 * math.ceil(count / 4) * 300 + 120)
if verify_budget > 7200:
raise ValueError('More than two hours of recovery required; split this deploy')
phase = 'apply-k8s'
k8s_ok = stage(directory, 'apply-k8s', 2700)
phase = 'apply-compose'
compose_ok = stage(directory, 'apply-compose', 1800) if k8s_ok else False
phase = 'verify-k8s'
verify_ok = stage(directory, 'verify-k8s', verify_budget)
phase = 'smoke'
smoke_ok = stage(directory, 'smoke', 600)
if not all((k8s_ok, compose_ok, verify_ok, smoke_ok)):
raise RuntimeError('Deploy failed; inspect stage logs and recovery report')
phase = 'Save the successful baseline'
finish_success(directory, plan)
except Exception as error:
status = json.loads((directory / 'status.json').read_text())
status['failure_stage'] = next(
(name for name, result in status['stages'].items() if result.get('result') == 'failure'), phase
)
atomic_json(directory / 'status.json', status)
with (directory / 'controller.log').open('a') as stream:
stream.write(f'{error}\n')
recover(directory)
@@ -294,10 +309,93 @@ def follow(run_id, phase):
time.sleep(3)
def summary(run_id):
directory = run_directory(run_id)
request = json.loads((directory / 'request.json').read_text())
release = request['release']
plan_file = directory / 'plan.json'
lines = [
f'## Deploy `{release["sha"]}`',
'',
f'- Mode: `{request["mode"]}`',
f'- Refresh third-party images: `{request["refresh_images"]}`',
]
status = json.loads((directory / 'status.json').read_text())
if status.get('failure_stage'):
lines.append(f'- Failed stage: **{status["failure_stage"]}**')
lines.extend(
[
'',
f'- Observed run state: **{status["state"]}**',
'',
'### Stage results',
'| Stage | Result | Exit code |',
'| --- | --- | --- |',
]
)
for name in ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke'):
stage_result = status['stages'].get(name, {})
lines.append(f'| {name} | {stage_result.get("result", "not started")} | {stage_result.get("exit_code", "—")} |')
lines.extend(['', '### Apply and Helm recovery results'])
events_file = directory / 'apply-events.jsonl'
events = []
if events_file.exists():
for line in events_file.read_text().splitlines():
try:
events.append(json.loads(line))
except json.JSONDecodeError:
lines.append('- An operation record is incomplete. Check the stage log.')
latest = {(event['action'], event['target']): event['result'] for event in events}
lines.extend(f'- `{action}` `{target}`: **{result}**' for (action, target), result in latest.items())
if not latest:
lines.append('- No apply results were recorded.')
lines.append('- A completed apply does not confirm health. See verification and smoke results.')
lines.extend(['', '### Kubernetes recovery'])
pointer = directory / 'snapshot/current'
failed = Path(pointer.read_text().strip()) / 'failed-workloads' if pointer.exists() else None
if failed and failed.exists():
contents = failed.read_text()
counts = dict(re.findall(r'^(ROLLED_BACK|UNRECOVERED)=([0-9]+)$', contents, re.MULTILINE))
if not contents.strip():
lines.append('- No failed workloads were recorded. See the verification result above.')
elif counts:
lines.append(f'- Workloads restored: **{counts.get("ROLLED_BACK", "unknown")}**')
lines.append(f'- Workloads that need manual recovery: **{counts.get("UNRECOVERED", "unknown")}**')
else:
lines.append('- Rollback has no recorded result yet. Check the verification log.')
else:
lines.append('- No workload rollback was recorded. This does not confirm health.')
lines.append('- Compose requires manual recovery. Use the saved command in the apply log.')
if not plan_file.exists():
lines.extend(['', 'Plan was not created. Check the controller log.'])
print('\n'.join(lines))
return
plan = json.loads(plan_file.read_text())
lines.extend(['', '### Selected services'])
count = 0
for kind, services in plan['selected'].items():
for service in services:
lines.append(f'- `{kind}`: `{service}`')
count += 1
if not count:
lines.append('- None')
lines.extend(['', '### Selected Helm releases'])
lines.extend(f'- `{release}`' for release in plan.get('helm', []))
if not plan.get('helm'):
lines.append('- None')
lines.extend(['', '### Images pinned in the checked release'])
lines.extend(f'- `{image}@{digest}`' for image, digest in sorted(release['images'].items()))
lines.extend(['', '### Removed resources requiring manual review'])
lines.extend(f'- `{item}`' for item in plan.get('removed', []))
if not plan.get('removed'):
lines.append('- None')
print('\n'.join(lines))
def main():
os.umask(0o077)
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow'))
parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow', 'summary'))
parser.add_argument('run_id')
parser.add_argument('phase', nargs='?', choices=('apply', 'verify', 'smoke'))
parser.add_argument('--retry', action='store_true', help='Retry failed recovery checks; never repeat apply')
@@ -315,6 +413,8 @@ def main():
if (directory / 'plan.json').exists():
plan = json.loads((directory / 'plan.json').read_text())
print(json.dumps({k: plan[k] for k in ('sha', 'selected', 'helm', 'removed')}, indent=2))
elif args.action == 'summary':
summary(args.run_id)
elif not follow(args.run_id, args.phase):
sys.exit(1)
+38 -4
View File
@@ -27,6 +27,15 @@ log() {
echo "== $* =="
}
# Store operation results without command output or local configuration values.
record_apply() {
[ -n "${RUN_DIR:-}" ] || return 0
jq -cn --arg action "$1" --arg target "$2" --arg result "$3" \
'{action: $action, target: $target, result: $result}' >>"$RUN_DIR/apply-events.jsonl" \
|| echo 'WARNING: cannot record an apply result' >&2
return 0
}
warn() {
echo "WARNING: $*" >&2
}
@@ -171,7 +180,8 @@ save_snapshot() {
| select(any(.metadata.ownerReferences[]?; .uid == $w.metadata.uid))
| select($w.kind != "StatefulSet" or .metadata.name == $w.status.currentRevision) | .revision] | max // 0) end)
}]' "$dir/workloads.json" >"$dir/revisions.json" || return 1
releases="$(helm list --all -A -o json)" || return 1
# Helm 4 lists every release status by default and removed the --all flag.
releases="$(helm list -A -o json)" || return 1
for entry in "${HELM_RELEASES[@]}"; do
IFS='|' read -r release _ namespace _ _ _ <<<"$entry"
if ! jq -e --arg r "$release" --arg n "$namespace" \
@@ -323,7 +333,7 @@ HELM_RELEASES=(
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
"victoria-operator|victoriametrics/victoria-metrics-operator|prometheus|0.68.1|prometheus-stack/k8s/victoria-operator-values.yaml|prometheus-stack/k8s/active"
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
"alloy|grafana/alloy|prometheus|1.13.0|loki/k8s/alloy-values.yaml|loki/k8s/active"
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
"reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active"
)
@@ -376,15 +386,19 @@ recover_pending_release() {
echo "ERROR: no captured Helm revision for $release; manual recovery required"
return 1
fi
record_apply helm-rollback "$namespace/$release" started
if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then
record_apply helm-rollback "$namespace/$release" failure
echo "WARN: helm rollback of $release did not complete"
return 1
fi
status="$(helm_release_status "$release" "$namespace")" || return 1
if [ "$status" != "deployed" ]; then
record_apply helm-rollback "$namespace/$release" failure
echo "WARN: $release is $status after rollback"
return 1
fi
record_apply helm-rollback "$namespace/$release" success
;;
esac
return 0
@@ -448,11 +462,13 @@ upgrade_helm_releases() {
# --rollback-on-failure (+ --wait) rolls the release back when the upgrade
# times out or the workloads it touches never become ready, so a bad chart
# bump is not left half applied. (--atomic was this combo; deprecated.)
record_apply helm-upgrade "$namespace/$release" started
if ! helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \
--version "$version" \
--values "$values" \
--wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then
record_apply helm-upgrade "$namespace/$release" failure
echo "WARN: upgrade of $release failed, checking release state"
# --rollback-on-failure already attempted its own rollback; finish the job when that
# rollback never completed, otherwise the release stays pending-* and
@@ -462,8 +478,10 @@ upgrade_helm_releases() {
else
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
fi
record_apply helm-recovery-state "$namespace/$release" "$(helm_release_status "$release" "$namespace" || echo unknown)"
return 1
fi
record_apply helm-upgrade "$namespace/$release" success
done
}
@@ -631,7 +649,12 @@ stage_apply_k8s() {
if [ "${#ns_files[@]}" -gt 0 ]; then
log "Applying namespaces (${#ns_files[@]} files)"
for m in "${ns_files[@]}"; do
kubectl apply -f "$m"
record_apply kubectl "${m#"$REPO"/}" started
if ! kubectl apply -f "$m"; then
record_apply kubectl "${m#"$REPO"/}" failure
return 1
fi
record_apply kubectl "${m#"$REPO"/}" success
done
fi
if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then
@@ -646,18 +669,24 @@ stage_apply_k8s() {
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
for m in "${other_files[@]}"; do
log "Applying ${m#"$REPO"/}"
record_apply kubectl "${m#"$REPO"/}" started
if ! render_pinned <"$m" | kubectl apply -f -; then
record_apply kubectl "${m#"$REPO"/}" failure
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
exit 1
fi
record_apply kubectl "${m#"$REPO"/}" success
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
record_apply kustomize "${k#"$REPO"/}" started
if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then
record_apply kustomize "${k#"$REPO"/}" failure
echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2
exit 1
fi
record_apply kustomize "${k#"$REPO"/}" success
done
# No verification here on purpose. This stage may be killed at any point by
@@ -959,7 +988,12 @@ stage_apply_compose() {
local cf
for cf in "${COMPOSE_STACKS[@]}"; do
log "Applying Compose ${cf#"$REPO"/}"
compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans
record_apply compose "${cf#"$REPO"/}" started
if ! compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans; then
record_apply compose "${cf#"$REPO"/}" failure
return 1
fi
record_apply compose "${cf#"$REPO"/}" success
verify_compose_stack "$cf"
done
echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)"
+36
View File
@@ -64,6 +64,18 @@ jobs:
run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA"
- name: Submit durable deploy to workstation
run: bash .gitea/workflows/ssh-run.sh start
- name: Write the request result
if: always()
env:
REQUEST_RESULT: ${{ job.status }}
CHECKED_SHA: ${{ steps.release.outputs.sha }}
run: |
if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
printf '## Deploy request\n\n- Result: **%s**\n- Checked commit: %s\n- Mode: %s\n' "$REQUEST_RESULT" "${CHECKED_SHA:-not checked}" "$DEPLOY_MODE" >>"$GITHUB_STEP_SUMMARY"
if [ "$REQUEST_RESULT" != success ]; then
echo 'Open the failed step log. If SSH submission failed, check the remote controller state.' >>"$GITHUB_STEP_SUMMARY"
fi
fi
apply:
needs: [gate]
@@ -76,6 +88,14 @@ jobs:
ref: ${{ needs.gate.outputs.sha }}
- name: Follow validation and sequential Kubernetes / Compose apply
run: bash .gitea/workflows/ssh-run.sh apply
- name: Write the deploy result
if: always()
run: |
if [ -f .gitea/workflows/ssh-run.sh ]; then
bash .gitea/workflows/ssh-run.sh summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY"
fi
verify:
needs: [gate, apply]
@@ -89,6 +109,14 @@ jobs:
ref: ${{ needs.gate.outputs.sha }}
- name: Follow workload verification and recovery
run: bash .gitea/workflows/ssh-run.sh verify
- name: Write the deploy result
if: always()
run: |
if [ -f .gitea/workflows/ssh-run.sh ]; then
bash .gitea/workflows/ssh-run.sh summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY"
fi
smoke:
needs: [gate, verify]
@@ -102,3 +130,11 @@ jobs:
ref: ${{ needs.gate.outputs.sha }}
- name: Follow public route checks
run: bash .gitea/workflows/ssh-run.sh smoke
- name: Write the deploy result
if: always()
run: |
if [ -f .gitea/workflows/ssh-run.sh ]; then
bash .gitea/workflows/ssh-run.sh summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY"
fi
+229 -63
View File
@@ -26,9 +26,6 @@ IMAGES = {
'xdfnx-homepage': ('homepages', 'homepages/Dockerfile.xdfnx'),
}
# These images are released by the EDU application repository.
EXTERNAL_IMAGES = {'gcr.forust.xyz/forust/session-keeper', 'gcr.forust.xyz/forust/webinar-checker'}
def command(*args, **kwargs):
"""Arguments are passed directly to the executable, never to a shell."""
@@ -163,7 +160,7 @@ def gate(output, requested_ref, event_sha):
print(f'CI gate accepted {sha}')
def build(output):
def prepare_images(output):
sha = command('git', 'rev-parse', 'HEAD')
if sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha):
raise ValueError('Build checkout does not match GITHUB_SHA')
@@ -178,11 +175,61 @@ def build(output):
except ValueError:
# Expired artifacts only cost a rebuild; mutable tags are never a fallback.
continue
targets = []
for name, (context, dockerfile) in IMAGES.items():
image = f'gcr.forust.xyz/forust/{name}'
inputs = fingerprint(context, dockerfile)
old_digest = (previous or {}).get('images', {}).get(image)
targets.append(
{
'name': name,
'image': image,
'context': context,
'dockerfile': dockerfile,
'inputs': inputs,
'reuse_digest': old_digest if (previous or {}).get('inputs', {}).get(image) == inputs else None,
}
)
output.write_text(json.dumps({'sha': sha, 'targets': targets}, indent=2) + '\n')
if os.environ.get('GITHUB_OUTPUT'):
with Path(os.environ['GITHUB_OUTPUT']).open('a') as stream:
stream.write('matrix=' + json.dumps({'include': targets}, separators=(',', ':')) + '\n')
print(f'Prepared {len(targets)} image jobs; {sum(t["reuse_digest"] is None for t in targets)} require builds')
def checked_plan(path):
data = json.loads(path.read_text())
sha = command('git', 'rev-parse', 'HEAD')
if data.get('sha') != sha or sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha):
raise ValueError('Image plan does not match the checked source commit')
targets = data.get('targets', [])
if sorted(t['name'] for t in targets) != sorted(IMAGES):
raise ValueError('Image plan must contain each owned image once')
for target in targets:
name = target['name']
context, dockerfile = IMAGES[name]
if (target['context'], target['dockerfile'], target['image']) != (
context,
dockerfile,
f'gcr.forust.xyz/forust/{name}',
) or target['inputs'] != fingerprint(context, dockerfile):
raise ValueError('Image plan has invalid build inputs')
if target['reuse_digest'] is not None and not DIGEST.fullmatch(target['reuse_digest']):
raise ValueError('Image plan has an invalid reuse digest')
return data
def build_images(output, report, name, plan):
data = checked_plan(plan)
sha = data['sha']
target = next(t for t in data['targets'] if t['name'] == name)
context, dockerfile = IMAGES[name]
docker_config = tempfile.mkdtemp(prefix='homelab-registry-')
builder_config = Path.home() / '.cache/homelab-ci/buildx'
builder_config.mkdir(parents=True, exist_ok=True)
env = {**os.environ, 'DOCKER_CONFIG': docker_config, 'BUILDX_CONFIG': str(builder_config)}
try:
report['phase'] = 'Registry login'
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
@@ -197,6 +244,7 @@ def build(output):
check=True,
env=env,
)
report['phase'] = 'Prepare the builder'
builder = 'homelab-ci'
versions = dict(
re.findall(r'^([A-Z_]+)="([^"\n]+)"$', Path('.gitea/workflows/tool-versions.env').read_text(), re.MULTILINE)
@@ -231,61 +279,67 @@ def build(output):
)
signature.write_text(image + '\n')
release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}}
for name, (context, dockerfile) in IMAGES.items():
image = f'gcr.forust.xyz/forust/{name}'
inputs = fingerprint(context, dockerfile)
old_digest = (previous or {}).get('images', {}).get(image)
exists = False
if old_digest and previous['inputs'].get(image) == inputs:
exists = (
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
'buildx',
'imagetools',
'inspect',
f'{image}@{old_digest}',
],
capture_output=True,
env=env,
timeout=60,
).returncode
== 0
)
if exists:
print(f'Reuse {name}: inputs unchanged')
digest = old_digest
else:
print(f'Build {name}', flush=True)
metadata = Path(docker_config) / 'metadata.json'
command(
'docker',
'buildx',
'build',
'--builder',
builder,
'--push',
'--platform',
'linux/amd64',
'--provenance=false',
'--cache-from',
f'type=registry,ref={image}:buildcache',
'--cache-to',
f'type=registry,ref={image}:buildcache,mode=max',
'--tag',
f'{image}:sha-{sha}',
'--metadata-file',
str(metadata),
'--file',
dockerfile,
context,
report['images'] = release['images']
report['phase'] = f'Build or reuse {name}'
report['current'] = name
image = f'gcr.forust.xyz/forust/{name}'
inputs = target['inputs']
old_digest = target['reuse_digest']
exists = False
if old_digest:
exists = (
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
'buildx',
'imagetools',
'inspect',
f'{image}@{old_digest}',
],
capture_output=True,
env=env,
)
digest = json.loads(metadata.read_text())['containerimage.digest']
release['images'][image] = digest
release['inputs'][image] = inputs
validate_release(release, sha)
timeout=60,
).returncode
== 0
)
if exists:
print(f'Reuse {name}: inputs unchanged')
digest = old_digest
report['reused'].append(name)
else:
print(f'Build {name}', flush=True)
metadata = Path(docker_config) / 'metadata.json'
command(
'docker',
'buildx',
'build',
'--builder',
builder,
'--platform',
'linux/amd64',
'--provenance=false',
'--cache-from',
f'type=registry,ref={image}:buildcache',
'--cache-to',
f'type=registry,ref={image}:buildcache,mode=max',
'--output',
f'type=image,name={image},push-by-digest=true,name-canonical=true,push=true',
'--metadata-file',
str(metadata),
'--file',
dockerfile,
context,
env=env,
)
digest = json.loads(metadata.read_text())['containerimage.digest']
report['built'].append(name)
release['images'][image] = digest
release['inputs'][image] = inputs
if not DIGEST.fullmatch(digest):
raise ValueError('Image job returned an invalid digest')
output.write_text(json.dumps(release, indent=2) + '\n')
report['current'] = None
report['phase'] = 'Release file saved'
finally:
# Cleanup errors must neither leak credentials nor mask the original build error.
try:
@@ -309,6 +363,59 @@ def build(output):
shutil.rmtree(docker_config)
def write_summary(lines):
path = os.environ.get('GITHUB_STEP_SUMMARY')
if path:
try:
with Path(path).open('a') as stream:
stream.write('\n'.join(lines) + '\n\n')
except OSError:
print('WARNING: cannot write the job summary')
def check_summary():
lines = [
f'## {os.environ["SUMMARY_CHECK"]}',
'',
f'- Commit: `{os.environ.get("GITHUB_SHA", "unknown")}`',
f'- Result: **{os.environ["SUMMARY_RESULT"]}**',
]
if os.environ.get('SUMMARY_FAILED_STEP'):
lines.append(f'- Failed step: {os.environ["SUMMARY_FAILED_STEP"]}')
if os.environ['SUMMARY_RESULT'] != 'success':
lines.append('- Open the failed step log for the error details.')
write_summary(lines)
def build(output, name, plan):
report = {'phase': 'Check the source commit', 'current': None, 'built': [], 'reused': [], 'images': {}}
result = 'failure'
try:
build_images(output, report, name, plan)
result = 'success'
finally:
lines = [
f'## Image release `{os.environ.get("GITHUB_SHA", "unknown")}`',
'',
f'- Result: **{result}**',
f'- Last stage: {report["phase"]}',
]
if result == 'failure':
lines.append('- No release from this build can be deployed. Open the failed step log.')
if report['current']:
lines.append(f'- Image at the failure: `{report["current"]}`')
for title, key in (('Built', 'built'), ('Reused from successful CI', 'reused')):
lines.extend(['', f'### {title}'])
lines.extend(f'- `{name}`' for name in report[key])
if not report[key]:
lines.append('- None')
lines.extend(['', '### Completed image digests'])
lines.extend(f'- `{image}@{digest}`' for image, digest in report['images'].items())
if not report['images']:
lines.append('- None')
write_summary(lines)
def render(stream, destination):
release = validate_release(json.loads(Path(os.environ['RELEASE_FILE']).read_text()), os.environ['DEPLOY_SHA'])
image_line = re.compile(
@@ -319,9 +426,6 @@ def render(stream, destination):
match = image_line.fullmatch(line.rstrip('\n'))
if match:
prefix, quote, image, tail = match.groups()
if image in EXTERNAL_IMAGES and f'{image}@sha256:' in line:
rendered.append(line)
continue
if image not in release['images']:
raise ValueError(f'Owned image missing from checked release: {image}')
line = f'{prefix}{quote}{image}@{release["images"][image]}{quote}{tail}\n'
@@ -331,19 +435,81 @@ def render(stream, destination):
destination.writelines(rendered)
def finalize_images(output, fragments, plan):
data = checked_plan(plan)
sha = data['sha']
release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}}
for name in IMAGES:
fragment = json.loads((fragments / f'image-{name}' / 'image.json').read_text())
image = f'gcr.forust.xyz/forust/{name}'
if fragment.get('sha') != sha or fragment.get('version') != 1 or set(fragment.get('images', {})) != {image}:
raise ValueError('Image job artifact is missing or belongs to another commit')
target = next(t for t in data['targets'] if t['name'] == name)
if fragment.get('inputs') != {image: target['inputs']}:
raise ValueError('Image artifact does not match the build plan')
release['images'].update(fragment['images'])
release['inputs'].update(fragment['inputs'])
validate_release(release, sha)
# Only a complete set of successful image jobs can publish the release tags.
docker_config = tempfile.mkdtemp(prefix='homelab-registry-')
env = {**os.environ, 'DOCKER_CONFIG': docker_config}
try:
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
'login',
'gcr.forust.xyz',
'-u',
os.environ['REGISTRY_USERNAME'],
'--password-stdin',
],
input=os.environ['REGISTRY_PASSWORD'],
text=True,
check=True,
env=env,
)
for image, digest in release['images'].items():
command(
'docker',
'buildx',
'imagetools',
'create',
'--prefer-index=false',
'--tag',
f'{image}:sha-{sha}',
f'{image}@{digest}',
env=env,
timeout=90,
)
output.write_text(json.dumps(release, indent=2) + '\n')
finally:
shutil.rmtree(docker_config)
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('action', choices=('build', 'gate', 'render'))
parser.add_argument('action', choices=('prepare', 'image', 'finalize', 'gate', 'render', 'check-summary'))
parser.add_argument('--output', type=Path, default=Path('release.json'))
parser.add_argument('--ref', default='main')
parser.add_argument('--event-sha', default='')
parser.add_argument('--image', choices=IMAGES)
parser.add_argument('--plan', type=Path, default=Path('build-plan.json'))
parser.add_argument('--fragments', type=Path, default=Path('artifacts'))
args = parser.parse_args()
if args.action == 'render':
if args.action == 'check-summary':
check_summary()
elif args.action == 'render':
render(sys.stdin, sys.stdout)
elif args.action == 'gate':
gate(args.output, args.ref, args.event_sha)
elif args.action == 'prepare':
prepare_images(args.output)
elif args.action == 'image':
if not args.image:
parser.error('--image is required')
build(args.output, args.image, args.plan)
else:
build(args.output)
finalize_images(args.output, args.fragments, args.plan)
if __name__ == '__main__':
+27 -17
View File
@@ -1,7 +1,9 @@
name: renovate-ci
on:
pull_request:
# Read the workflow from the trusted base branch. PR code runs only on the
# unprivileged runner selected below.
pull_request_target:
paths:
- "renovate/**"
- ".gitea/workflows/renovate-ci.yaml"
@@ -26,37 +28,47 @@ permissions:
jobs:
validate-renovate:
runs-on: homelab
runs-on: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
ref: ${{ github.event_name == 'pull_request_target' && github.event.pull_request.head.sha || github.sha }}
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag,
# so the same version that runs in the cluster is the one validated here.
- name: Resolve the deployed Renovate image
# renovate/k8s/cronjob.yaml is the single source of truth for the version.
- name: Resolve the deployed Renovate version
id: image
shell: bash
run: |
set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
if [[ ! "$image" =~ ^renovate/renovate:([0-9]+\.[0-9]+\.[0-9]+)$ ]]; then
echo "::error::expected a pinned renovate/renovate semantic version in renovate/k8s/cronjob.yaml"
exit 1
fi
echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT"
version="${BASH_REMATCH[1]}"
echo "using Renovate $version"
printf 'version=%s\n' "$version" >> "$GITHUB_OUTPUT"
- name: Validate Renovate repository config
- name: Prepare pinned validation tools
shell: bash
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate:/opt/renovate:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \
renovate-config-validator /opt/renovate/renovate.json
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform node)"
echo "$tools_dir" >> "$GITHUB_PATH"
- name: Validate Renovate repository config
shell: bash
env:
RENOVATE_VERSION: ${{ steps.image.outputs.version }}
run: |
set -euo pipefail
npm_cache="$(mktemp -d "${RUNNER_TEMP:-/tmp}/renovate-npm-cache.XXXXXXXX")"
trap 'rm -rf "$npm_cache"' EXIT
NPM_CONFIG_CACHE="$npm_cache" RENOVATE_CONFIG_FILE="$PWD/renovate/renovate.json" \
npm exec --yes --package="renovate@${RENOVATE_VERSION}" -- renovate-config-validator
# The CronJob cannot read the repository, so renovate/k8s/configmap.yaml
# carries an inlined copy of the config. Fail if it no longer matches.
@@ -70,8 +82,6 @@ jobs:
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
export PATH="$tools_dir:$PATH"
kubeconform \
-strict \
-ignore-missing-schemas \
+11 -5
View File
@@ -32,11 +32,14 @@ concurrency:
jobs:
run-renovate:
if: github.ref == 'refs/heads/main'
runs-on: homelab
timeout-minutes: 60
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
ref: refs/heads/main
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag.
# Reading it here means this workflow validates and runs the exact version
@@ -48,21 +51,23 @@ jobs:
set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
if [[ ! "$image" =~ ^renovate/renovate:[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
echo "::error::expected a pinned renovate/renovate semantic version in renovate/k8s/cronjob.yaml"
exit 1
fi
echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT"
printf 'image=%s\n' "$image" >> "$GITHUB_OUTPUT"
- name: Validate Renovate config
shell: bash
env:
RENOVATE_IMAGE: ${{ steps.image.outputs.image }}
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \
"$RENOVATE_IMAGE" \
renovate-config-validator
- name: Run Renovate
@@ -73,6 +78,7 @@ jobs:
RENOVATE_REPOSITORIES: ${{ inputs.repositories }}
RENOVATE_DRY_RUN: ${{ inputs.dry_run && 'full' || '' }}
LOG_LEVEL: ${{ inputs.log_level }}
RENOVATE_IMAGE: ${{ steps.image.outputs.image }}
run: |
set -euo pipefail
@@ -89,4 +95,4 @@ jobs:
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
-e RENOVATE_BASE_DIR=/tmp/renovate \
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
"${{ steps.image.outputs.image }}"
"$RENOVATE_IMAGE"
+18 -4
View File
@@ -20,7 +20,7 @@ ssh_opts=(-i "$key_dir/key" -p "${DEPLOY_PORT:-22}" -o BatchMode=yes -o StrictHo
-o "UserKnownHostsFile=$key_dir/known_hosts" -o ConnectTimeout=15
-o ServerAliveInterval=15 -o ServerAliveCountMax=4)
controller=.local/lib/homelab-deploy/controller.py
case "${1:?start, apply, verify or smoke required}" in
case "${1:?start, apply, verify, smoke or summary required}" in
start)
python3 - <<'PY' >"$key_dir/request.json"
import json
@@ -41,16 +41,30 @@ PY
exit "$rc"
;;
apply|verify|smoke)
result=0
for attempt in 1 2 3; do
rc=0
# shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables.
ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$?
[ "$rc" -eq 0 ] && exit 0
[ "$rc" -eq 255 ] || exit "$rc"
[ "$rc" -eq 0 ] && break
[ "$rc" -eq 255 ] || { result="$rc"; break; }
echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)"
if [ "$attempt" -eq 3 ]; then result=255; break; fi
sleep 5
done
exit "$rc"
exit "$result"
;;
summary)
if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
rc=0
# shellcheck disable=SC2029 # The run ID is validated above.
ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" summary "$DEPLOY_RUN_ID" >"$key_dir/deploy-summary.md" || rc=$?
if [ "$rc" -eq 0 ]; then
cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" || echo "WARNING: cannot write the deploy summary"
else
echo 'Deploy summary is unavailable. The SSH connection failed or the controller did not respond. Check the job log.' >>"$GITHUB_STEP_SUMMARY" || true
fi
fi
;;
*) echo "Unknown SSH operation: $1" >&2; exit 1 ;;
esac
-1
View File
@@ -94,7 +94,6 @@ replacements.txt
.idea
# Temp files
edu_master/temp/
temp/*
# Local-only tooling scratch space (pinned CI tools, verification scripts)
tmp/
-14
View File
@@ -1,14 +0,0 @@
EDU_LOGIN=your_edu_login_here
EDU_PASSWORD=your_edu_password_here
EDU_URL_LOGIN=https://edu.edu.vn.ua/user/login
EDU_URL_VERIFY=https://edu.edu.vn.ua/course/userlist
PHPSESSID_INTERVAL=10
USER_AGENT="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36"
WEBINAR_URL=https://edu.edu.vn.ua/webinar/useractive
WEBINAR_CHECK_INTERVAL=60
REDIS_HOST=redis
REDIS_PORT=6379
PLAYWRIGHT_WS=ws://playwright-service:3000/ws
TZ=Europe/Kyiv
WEBINAR_TELEGRAM_TOKEN=your_telegram_bot_token_here
WEBINAR_ADMIN_ID=123456789
-1
View File
@@ -1 +0,0 @@
1.56.0
-14
View File
@@ -1,14 +0,0 @@
# EDU deployment ownership
Application source and release builds: `forust/edu-master`.
The homelab pipeline deploys `edu_master/k8s` and preserves explicit image digests.
The application copies in this directory are legacy and are not build inputs.
Do not publish EDU `prod` images from homelab or resolve releases from moving tags.
For an EDU release, validate both images, select their digests in the keeper and
checker manifests, and run the existing homelab validation/apply/verification
helpers against this service. Keep the existing Secret and Redis PVC.
Coordinate Redis authentication changes with both clients and all init/probes;
keep a pre-rollout Redis backup and both previous compatible image references.
The current HTTP checker does not depend on Playwright; check other consumers
before removing the separate browser service.
-49
View File
@@ -1,49 +0,0 @@
services:
redis:
image: redis:8.10.2-alpine
restart: unless-stopped
volumes:
- redis-data:/data
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 5s
timeout: 3s
retries: 5
playwright-service:
image: mcr.microsoft.com/playwright:v1.56.0-jammy
restart: unless-stopped
command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws
session-keeper:
build: ./phpsessid-bot
image: gcr.forust.xyz/forust/session-keeper:prod
pull_policy: build
env_file: .env
restart: unless-stopped
depends_on:
redis:
condition: service_healthy
healthcheck:
test: ["CMD-SHELL", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
interval: 30s
timeout: 5s
retries: 10
start_period: 60s
webinar-checker:
build: ./webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:prod
pull_policy: build
env_file: .env
restart: unless-stopped
depends_on:
redis:
condition: service_healthy
session-keeper:
condition: service_healthy
playwright-service:
condition: service_started
volumes:
redis-data:
View File
Whitespace-only changes.
-115
View File
@@ -1,115 +0,0 @@
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: edu-master-webinar
namespace: edu-master
labels:
release: prometheus-stack
spec:
groups:
- name: edu_master.webinar
rules:
# No successful webinar check for 5m (~2-3 missed 2-min checks).
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
# The last_success > 0 guard is mandatory: checker.py initialises
# last_success to 0, so without it `time() - 0` equals the current epoch
# and humanizeDuration renders ~20722d on every pod restart. Keep the
# duration expression on the left so $value stays the real gap.
- alert: WebinarCheckerNoSuccessfulCheck
expr: |
((time() - webinar_check_last_success_timestamp_seconds) > 300)
and (webinar_check_last_success_timestamp_seconds > 0)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 2m
labels:
severity: critical
annotations:
summary: "Webinar checker has no successful check for 5m"
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
# Checks are running but none has ever succeeded since pod start.
# Split out from the rule above so a zeroed gauge never feeds
# humanizeDuration.
- alert: WebinarCheckerNeverSucceeded
expr: |
(webinar_check_last_success_timestamp_seconds == 0)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker has never completed a successful check"
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
- alert: WebinarCheckerConsecutiveFailures
expr: |
webinar_check_consecutive_failures >= 3
for: 5m
labels:
severity: critical
annotations:
summary: "Webinar checker failing consecutively"
description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / http error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
- alert: WebinarCheckerNeverStarted
expr: |
(time() - edu_process_start > 120)
and (webinar_check_last_run_timestamp_seconds == 0)
for: 2m
labels:
severity: critical
annotations:
summary: "Webinar checker job has not started"
description: "The process exposes metrics but its webinar job has never started."
- alert: WebinarDeliveryPending
expr: edu_delivery_pending > 0
for: 5m
labels:
severity: warning
annotations:
summary: "Webinar notifications await delivery"
description: "Telegram delivery has pending recipients. Check delivery failures and retry status."
- alert: EduRedisUnavailable
expr: edu_redis_connected == 0
for: 2m
labels:
severity: critical
annotations:
summary: "EDU checker cannot reach Redis"
description: "Redis health checks are failing; checker commands and delivery may be unavailable."
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
- alert: WebinarCheckerScrapeDown
expr: |
absent(webinar_check_last_run_timestamp_seconds) == 1
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker metrics missing"
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
- alert: EduPhpsessidMissing
expr: |
edu_phpsessid_present == 0
for: 10m
labels:
severity: critical
annotations:
summary: "EDU_PHPSESSID missing"
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
# Hard deps: checker deployment unavailable.
- alert: WebinarCheckerDeploymentDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker deployment unavailable"
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
-4
View File
@@ -1,4 +0,0 @@
apiVersion: v1
kind: Namespace
metadata:
name: edu-master
-69
View File
@@ -1,69 +0,0 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: playwright-service
namespace: edu-master
labels:
app: edu-master-playwright
spec:
replicas: 1
selector:
matchLabels:
app: edu-master-playwright
strategy:
type: Recreate
template:
metadata:
labels:
app: edu-master-playwright
spec:
containers:
- name: playwright
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
image: mcr.microsoft.com/playwright:v1.56.0-jammy
imagePullPolicy: IfNotPresent
# p95 412M, max 478M over 7 days, no limit before. Request is set at p95
# so the pod is not an eviction candidate; the limit stays above 2x the
# request because browser page lifetimes are unpredictable.
resources:
requests:
cpu: "200m"
memory: "416Mi"
limits:
memory: "1Gi"
command:
- npx
- -y
- playwright@1.56.0
- run-server
- --port
- "3000"
- --path
- /ws
ports:
- containerPort: 3000
readinessProbe:
tcpSocket:
port: 3000
initialDelaySeconds: 5
periodSeconds: 10
timeoutSeconds: 3
livenessProbe:
tcpSocket:
port: 3000
initialDelaySeconds: 15
periodSeconds: 20
timeoutSeconds: 3
---
apiVersion: v1
kind: Service
metadata:
name: playwright-service
namespace: edu-master
spec:
selector:
app: edu-master-playwright
ports:
- name: ws
port: 3000
targetPort: 3000
-22
View File
@@ -1,22 +0,0 @@
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: redis-clients-only
namespace: edu-master
spec:
podSelector:
matchLabels:
app: edu-master-redis
policyTypes:
- Ingress
ingress:
- from:
- podSelector:
matchLabels:
app: edu-master-session-keeper
- podSelector:
matchLabels:
app: edu-master-webinar-checker
ports:
- protocol: TCP
port: 6379
-96
View File
@@ -1,96 +0,0 @@
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: redis
namespace: edu-master
labels:
app: edu-master-redis
spec:
serviceName: redis
replicas: 1
selector:
matchLabels:
app: edu-master-redis
template:
metadata:
labels:
app: edu-master-redis
spec:
containers:
- name: redis
image: redis:8.10.2-alpine
imagePullPolicy: IfNotPresent
env:
- name: REDIS_PASSWORD
valueFrom:
secretKeyRef:
name: edu-master-secrets
key: REDIS_PASSWORD
- name: REDISCLI_AUTH
valueFrom:
secretKeyRef:
name: edu-master-secrets
key: REDIS_PASSWORD
command:
- /bin/sh
- -ec
- |
case "$REDIS_PASSWORD" in *[!0-9a-fA-F]*|'') echo 'REDIS_PASSWORD must be 64 hex characters' >&2; exit 1;; esac
[ "${#REDIS_PASSWORD}" -eq 64 ] || { echo 'REDIS_PASSWORD must be 64 hex characters' >&2; exit 1; }
umask 077
printf 'requirepass "%s"\n' "$REDIS_PASSWORD" > /tmp/redis-auth.conf
chown redis:redis /tmp/redis-auth.conf
exec docker-entrypoint.sh redis-server /tmp/redis-auth.conf
ports:
- containerPort: 6379
volumeMounts:
- name: redis-data
mountPath: /data
resources:
requests:
cpu: 25m
memory: 32Mi
limits:
cpu: 250m
memory: 128Mi
readinessProbe:
exec:
command: ["redis-cli", "ping"]
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 3
livenessProbe:
exec:
command: ["redis-cli", "ping"]
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 3
volumes:
- name: redis-data
persistentVolumeClaim:
claimName: redis-data-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: redis-data-pvc
namespace: edu-master
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 1Gi
---
apiVersion: v1
kind: Service
metadata:
name: redis
namespace: edu-master
spec:
selector:
app: edu-master-redis
ports:
- name: redis
port: 6379
targetPort: 6379
@@ -1,50 +0,0 @@
# One-time Job to migrate redis state from docker compose to k8s (maintenance window).
# The .example file is not applied by the deploy pipeline (mask *.example.yaml).
#
# Runbook:
# 1. docker compose -f <repo>/edu_master/compose.yaml stop # SIGTERM -> redis will flush dump.rdb
# 2. docker run --rm -v edu_master_redis-data:/data \
# -v /tmp/edu-master-backup:/backup \
# redis:alpine sh -c "cp /data/dump.rdb /backup/ && ls -la /backup"
# 3. kubectl apply -f edu_master/k8s/namespace.yaml
# 4. kubectl apply -f <only the PVC from redis.yaml> # seed must come BEFORE redis pod starts
# 5. kubectl apply -f edu_master/k8s/restore-seed-job.yaml.example
# kubectl wait --for=condition=complete job/redis-restore-seed -n edu-master --timeout=120s
# 6. kubectl delete job redis-restore-seed -n edu-master
# 7. kubectl apply -f edu_master/k8s/ -R # apply remaining manifests
apiVersion: batch/v1
kind: Job
metadata:
name: redis-restore-seed
namespace: edu-master
spec:
backoffLimit: 2
ttlSecondsAfterFinished: 3600
template:
spec:
restartPolicy: Never
containers:
- name: seed
image: redis:alpine
command:
- /bin/sh
- -ec
- |
ls -la /backup
cp /backup/dump.rdb /data/dump.rdb
chmod 644 /data/dump.rdb
ls -la /data
volumeMounts:
- name: redis-data
mountPath: /data
- name: backup
mountPath: /backup
readOnly: true
volumes:
- name: redis-data
persistentVolumeClaim:
claimName: redis-data-pvc
- name: backup
hostPath:
path: /tmp/edu-master-backup
type: DirectoryOrCreate
-29
View File
@@ -1,29 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: edu-master-secrets
namespace: edu-master
type: Opaque
stringData:
# Session keeper credentials
KEEPER_LOGIN: ""
KEEPER_PASSWORD: ""
KEEPER_INTERVAL: "10"
# EDU links
EDU_URL_BASE: "https://edu.edu.vn.ua"
EDU_URL_LOGIN: "/user/login"
EDU_URL_COURSES: "/course/userlist"
EDU_URL_WEBINAR: "/webinar/useractive"
# Playwright
USER_AGENT: ""
PLAYWRIGHT_WS: "ws://playwright-service:3000/ws"
# Webinar-checker
WEBINAR_TELEGRAM_TOKEN: ""
WEBINAR_ADMIN_ID: ""
WEBINAR_CHECK_INTERVAL: "60"
# Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml)
METRICS_PORT: "8000"
# Database
REDIS_HOST: "redis"
REDIS_PORT: "6379"
TZ: "Europe/Kyiv"
-15
View File
@@ -1,15 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: webinar-checker
namespace: edu-master
labels:
app: edu-master-webinar-checker
spec:
selector:
app: edu-master-webinar-checker
ports:
- name: metrics
port: 8000
targetPort: metrics
protocol: TCP
-69
View File
@@ -1,69 +0,0 @@
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: session-keeper
namespace: edu-master
labels:
app: edu-master-session-keeper
spec:
replicas: 1
selector:
matchLabels:
app: edu-master-session-keeper
strategy:
type: Recreate
template:
metadata:
annotations:
edu.forust.xyz/source-commit: "90829d6c8080b9928f9da23587678e640939e10a"
labels:
app: edu-master-session-keeper
spec:
initContainers:
- name: wait-redis
image: redis:8.10.2-alpine
env:
- name: REDISCLI_AUTH
valueFrom:
secretKeyRef:
name: edu-master-secrets
key: REDIS_PASSWORD
command:
- /bin/sh
- -ec
- |
i=0
until redis-cli -h redis ping | grep -q PONG; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: redis not ready" && exit 1
sleep 2
done
echo "redis is ready"
containers:
- name: session-keeper
image: gcr.forust.xyz/forust/session-keeper@sha256:49285e87cc5bc4cf4ffe190813d87927916c2df8a206daac0aeb7d227c636450
envFrom:
- secretRef:
name: edu-master-secrets
env:
- name: REDISCLI_AUTH
valueFrom:
secretKeyRef:
name: edu-master-secrets
key: REDIS_PASSWORD
resources:
requests:
cpu: 25m
memory: 32Mi
limits:
cpu: 250m
memory: 128Mi
readinessProbe:
exec:
command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
initialDelaySeconds: 15
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 10
-87
View File
@@ -1,87 +0,0 @@
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: webinar-checker
namespace: edu-master
labels:
app: edu-master-webinar-checker
spec:
replicas: 1
selector:
matchLabels:
app: edu-master-webinar-checker
strategy:
type: Recreate
template:
metadata:
annotations:
edu.forust.xyz/source-commit: "90829d6c8080b9928f9da23587678e640939e10a"
labels:
app: edu-master-webinar-checker
spec:
# Enforces dependency order like compose depends_on:
# redis healthy -> session-keeper healthy (EXISTS EDU_PHPSESSID)
initContainers:
- name: wait-deps
image: redis:8.10.2-alpine
env:
- name: REDISCLI_AUTH
valueFrom:
secretKeyRef:
name: edu-master-secrets
key: REDIS_PASSWORD
command:
- /bin/sh
- -ec
- |
i=0
until redis-cli -h redis ping | grep -q PONG; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: redis not ready" && exit 1
sleep 2
done
echo "redis ok"
until [ "$(redis-cli -h redis EXISTS EDU_PHPSESSID)" = "1" ]; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: no PHPSESSID (session-keeper down?)" && exit 1
sleep 2
done
echo "PHPSESSID ok"
containers:
- name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker@sha256:66c146f7b43cb9f0dc31ba9aa36d217e01df42ddafba5971b79c12ec215b2c01
ports:
- name: metrics
containerPort: 8000
protocol: TCP
readinessProbe:
httpGet:
path: /health
port: metrics
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 12
initialDelaySeconds: 10
livenessProbe:
httpGet:
path: /live
port: metrics
initialDelaySeconds: 60
periodSeconds: 15
timeoutSeconds: 3
failureThreshold: 4
envFrom:
- secretRef:
name: edu-master-secrets
env:
- name: TZ
value: "Europe/Kyiv"
resources:
requests:
cpu: "50m"
memory: "192Mi"
limits:
cpu: "600m"
memory: "384Mi"
-15
View File
@@ -1,15 +0,0 @@
FROM python:3.14-slim
WORKDIR /app
# Install system dependencies
RUN apt-get update && apt-get install -y --no-install-recommends redis-tools && rm -rf /var/lib/apt/lists/*
# Install dependencies
RUN pip install --no-cache-dir requests==2.32.3 redis==5.2.1
# Copy application code
COPY . .
# Run the bot
CMD ["python", "bot.py"]
-132
View File
@@ -1,132 +0,0 @@
import logging
import os
import time
from datetime import datetime
import redis
import requests
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# Load configuration (adapted to .env keys)
def _env(key, default=None):
v = os.getenv(key, default)
if isinstance(v, str) and len(v) >= 2 and ((v[0] == '"' and v[-1] == '"') or (v[0] == "'" and v[-1] == "'")):
return v[1:-1]
return v
LOGIN = _env('KEEPER_LOGIN')
PASSWORD = _env('KEEPER_PASSWORD')
EDU_BASE = _env('EDU_URL_BASE', 'https://edu.edu.vn.ua')
EDU_LOGIN_PATH = _env('EDU_URL_LOGIN', '/user/login')
EDU_COURSES_PATH = _env('EDU_URL_COURSES', '/course/userlist')
URL_LOGIN = f'{EDU_BASE.rstrip("/")}/{EDU_LOGIN_PATH.lstrip("/")}'
URL_VERIFY = f'{EDU_BASE.rstrip("/")}/{EDU_COURSES_PATH.lstrip("/")}'
INTERVAL = int(_env('KEEPER_INTERVAL', 10))
USER_AGENT = _env(
'USER_AGENT',
'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36',
)
REDIS_HOST = _env('REDIS_HOST', 'redis')
REDIS_PORT = int(_env('REDIS_PORT', 6379))
SUCCESS_FILE = '/tmp/last_success' # noqa: S108
def touch_success_file():
"""Updates the timestamp of the success file for healthchecks."""
try:
with open(SUCCESS_FILE, 'w') as f:
f.write(str(datetime.now().timestamp()))
except Exception as e:
logger.error(f'Failed to touch success file: {e}')
def main():
logger.info('Starting Session Keeper Bot')
# Connect to Redis
try:
redis_client = redis.Redis(host=REDIS_HOST, port=REDIS_PORT, decode_responses=True)
redis_client.ping()
logger.info(f'Connected to Redis at {REDIS_HOST}:{REDIS_PORT}')
except Exception as e:
logger.error(f'Failed to connect to Redis: {e}')
return
session = requests.Session()
# Set headers
headers = {
'User-Agent': USER_AGENT,
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
'Accept-Language': 'en-US,en;q=0.9',
'Cache-Control': 'max-age=0',
'Upgrade-Insecure-Requests': '1',
'Sec-Fetch-Site': 'same-origin',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-User': '?1',
'Sec-Fetch-Dest': 'document',
'Sec-Ch-Ua': '"Not_A Brand";v="99", "Chromium";v="142"',
'Sec-Ch-Ua-Mobile': '?0',
'Sec-Ch-Ua-Platform': '"Linux"',
'Accept-Encoding': 'gzip, deflate, br',
'Priority': 'u=0, i',
}
session.headers.update(headers)
while True:
try:
logger.info('Attempting login...')
# Login payload
payload = {'login': LOGIN, 'password': PASSWORD}
# Perform Login
# Note: The user request shows a POST to /user/login with form data
# We need to make sure we handle the PHPSESSID correctly.
# If we already have a PHPSESSID, requests will send it.
login_response = session.post(URL_LOGIN, data=payload, allow_redirects=True)
logger.info(f'Login Response Status: {login_response.status_code}')
logger.info(f'Cookies after login: {session.cookies.get_dict()}')
# Verify Session
logger.info('Verifying session...')
verify_response = session.get(URL_VERIFY, allow_redirects=False)
logger.info(f'Verify Response Status: {verify_response.status_code}')
if verify_response.status_code == 200:
logger.info('Session verification SUCCESS (200 OK).')
touch_success_file()
# Save PHPSESSID to Redis
phpsessid = session.cookies.get('PHPSESSID')
if phpsessid:
try:
redis_client.set('EDU_PHPSESSID', phpsessid)
logger.info(f'Saved PHPSESSID to Redis: {phpsessid}')
except Exception as e:
logger.error(f'Failed to save PHPSESSID to Redis: {e}')
elif verify_response.status_code == 302:
logger.warning('Session verification FAILED (302 Redirect). Session might be invalid.')
else:
logger.warning(f'Session verification returned unexpected status: {verify_response.status_code}')
except Exception as e:
logger.error(f'An error occurred: {e}')
logger.info(f'Sleeping for {INTERVAL} minutes...')
time.sleep(INTERVAL * 60)
if __name__ == '__main__':
main()
-13
View File
@@ -1,13 +0,0 @@
FROM python:3.14-slim
WORKDIR /app
# renovate: datasource=pypi depName=playwright versioning=pep440
ARG PLAYWRIGHT_VERSION=1.56.0
# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
COPY checker.py .
CMD ["python", "checker.py"]
File diff suppressed because it is too large. Load diff
+2
View File
@@ -20,6 +20,8 @@ data:
GITEA__mailer__ENABLED: "false"
GITEA__metrics__ENABLED: "true"
# No code/issue search needed: bleve reindexes the whole issue index on
# every pod restart (cron.rebuild_issue_indexer RUN_AT_START) and hammers
# the rotational disk for an hour. "db" serves issue search from postgres.
+2
View File
@@ -3,6 +3,8 @@ kind: Service
metadata:
name: gitea-service
namespace: gitea
labels:
app: gitea
spec:
selector:
app: gitea
+2 -1
View File
@@ -7,7 +7,8 @@ spec:
entryPoints:
- websecure
routes:
- match: Host(`gitea.forust.xyz`) || Host(`git.forust.xyz`)
# Metrics are scraped directly through the cluster Service.
- match: (Host(`gitea.forust.xyz`) || Host(`git.forust.xyz`)) && !PathPrefix(`/metrics`)
kind: Rule
services:
- name: gitea-service
+16
View File
@@ -0,0 +1,16 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: gitea
namespace: gitea
labels:
release: prometheus-stack
spec:
selector:
matchLabels:
app: gitea
endpoints:
- port: http
path: /metrics
interval: 30s
scrapeTimeout: 10s
+1 -1
View File
@@ -7,7 +7,7 @@
# # Dev server_url
# server_url: https://hs.dev_internal_domain.internal
listen_addr: 0.0.0.0:8080
metrics_listen_addr: 127.0.0.1:9090
metrics_listen_addr: 0.0.0.0:9090
grpc_listen_addr: 127.0.0.1:50443
grpc_allow_insecure: false
noise:
@@ -3,6 +3,8 @@ kind: Service
metadata:
name: headscale-server-external
namespace: headscale
labels:
app: headscale
spec:
ports:
- port: 8080
+18
View File
@@ -0,0 +1,18 @@
apiVersion: operator.victoriametrics.com/v1beta1
kind: VMServiceScrape
metadata:
name: headscale
namespace: headscale
labels:
release: prometheus-stack
spec:
# The external Service has a manually managed EndpointSlice, not Endpoints.
discoveryRole: endpointslice
selector:
matchLabels:
app: headscale
endpoints:
- port: metrics
path: /metrics
interval: 30s
scrapeTimeout: 10s
+4
View File
@@ -6,6 +6,10 @@ metadata:
data:
TZ: "Europe/Bratislava"
IMMICH_TELEMETRY_INCLUDE: "all"
IMMICH_API_METRICS_PORT: "8081"
IMMICH_MICROSERVICES_METRICS_PORT: "8082"
# The database in this namespace, not the shared one in the database
# namespace: v3 needs VectorChord, and only the dedicated image carries it.
DB_HOSTNAME: "immich-postgres"
+12
View File
@@ -3,6 +3,8 @@ kind: Service
metadata:
name: immich-service
namespace: immich
labels:
app: immich
spec:
selector:
app: immich
@@ -10,6 +12,12 @@ spec:
- name: http
port: 2283
targetPort: 2283
- name: api-metrics
port: 8081
targetPort: api-metrics
- name: worker-metrics
port: 8082
targetPort: worker-metrics
---
apiVersion: apps/v1
kind: Deployment
@@ -41,6 +49,10 @@ spec:
ports:
- name: http
containerPort: 2283
- name: api-metrics
containerPort: 8081
- name: worker-metrics
containerPort: 8082
volumeMounts:
- name: immich-data
mountPath: /data
+20
View File
@@ -0,0 +1,20 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: immich
namespace: immich
labels:
release: prometheus-stack
spec:
selector:
matchLabels:
app: immich
endpoints:
- port: api-metrics
path: /metrics
interval: 30s
scrapeTimeout: 10s
- port: worker-metrics
path: /metrics
interval: 30s
scrapeTimeout: 10s
+9
View File
@@ -3,6 +3,8 @@ kind: Service
metadata:
name: netbird-server-service
namespace: netbird
labels:
app: netbird-server
spec:
selector:
app: netbird-server
@@ -11,6 +13,10 @@ spec:
name: http
targetPort: 80
protocol: TCP
- port: 9090
name: metrics
targetPort: metrics
protocol: TCP
- port: 3478
name: stun
targetPort: 3478
@@ -59,6 +65,9 @@ spec:
- containerPort: 80
name: http
protocol: TCP
- containerPort: 9090
name: metrics
protocol: TCP
- containerPort: 3478
name: stun
protocol: UDP
@@ -1,14 +1,14 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: webinar-checker
namespace: edu-master
name: netbird-server
namespace: netbird
labels:
release: prometheus-stack
spec:
selector:
matchLabels:
app: edu-master-webinar-checker
app: netbird-server
endpoints:
- port: metrics
path: /metrics
+26
View File
@@ -15,3 +15,29 @@ The VictoriaMetrics Operator chart and its CRDs are installed before the
Kubernetes manifests by the normal deploy workflow. On a cluster where the
operator CRDs are not installed yet, CI skips the server-side dry-run of the
`VMAgent` resource; the deploy installs the chart before applying that resource.
## Application metrics
The application ServiceMonitors use a 30s interval and a 10s timeout:
- Headscale: the external Service points to the Compose host on port 19090.
A VMServiceScrape uses EndpointSlice discovery for this manually managed target.
The Compose configuration must bind metrics to `0.0.0.0:9090`.
- NetBird: the combined server exports `/metrics` on port 9090. The existing
`server.metricsPort` setting enables the listener.
- Gitea: `GITEA__metrics__ENABLED` enables `/metrics` on the HTTP port. The public
ingress excludes this path. The monitor uses the internal Service directly.
- Immich: `IMMICH_TELEMETRY_INCLUDE=all` enables API and worker metrics on ports
8081 and 8082. The monitor scrapes both ports on each server replica.
Deploy through the existing CI and deploy workflow. Gitea and Immich reload their
ConfigMap changes through Reloader. Check the VMAgent targets after deployment
and query `up{scraper="victoria",namespace=~"netbird|gitea|immich|headscale"}` in
VictoriaMetrics. All targets should report 1.
For rollback, revert the application metrics changes, run CI, and deploy the
revert. Remove the three application ServiceMonitors and the Headscale VMServiceScrape explicitly: the deployment
workflow applies manifests and does not prune removed resources.
For Headscale rollback, remove its VMServiceScrape and Service label, restore the
previous Compose metrics bind address, and restart only the Headscale service.
-36
View File
@@ -29,31 +29,6 @@ data:
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": ["edu_master/k8s/playwright.yaml", "edu_master/compose.yaml"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["edu_master/PLAYWRIGHT_VERSION"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)(?:\\r?\\n)?$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: playwright Python client version pinned in Dockerfile ARG",
"managerFilePatterns": ["edu_master/webinar-checker/Dockerfile"],
"matchStrings": ["(?:^|\\n)ARG PLAYWRIGHT_VERSION=(?<currentValue>\\d+\\.\\d+\\.\\d+)(?:\\r?\\n|$)"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright",
"versioningTemplate": "pep440"
},
{
"customType": "regex",
"description": "kube-prometheus-stack chart version pinned in the deploy workflow",
@@ -212,17 +187,6 @@ data:
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
},
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks the WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{
"description": "Renovate updates itself in lockstep across the CronJob and the Compose file",
"matchPackageNames": ["renovate/renovate"],
-36
View File
@@ -18,31 +18,6 @@
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": ["edu_master/k8s/playwright.yaml", "edu_master/compose.yaml"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["edu_master/PLAYWRIGHT_VERSION"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)(?:\\r?\\n)?$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: playwright Python client version pinned in Dockerfile ARG",
"managerFilePatterns": ["edu_master/webinar-checker/Dockerfile"],
"matchStrings": ["(?:^|\\n)ARG PLAYWRIGHT_VERSION=(?<currentValue>\\d+\\.\\d+\\.\\d+)(?:\\r?\\n|$)"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright",
"versioningTemplate": "pep440"
},
{
"customType": "regex",
"description": "kube-prometheus-stack chart version pinned in the deploy workflow",
@@ -201,17 +176,6 @@
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
},
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks the WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{
"description": "Renovate updates itself in lockstep across the CronJob and the Compose file",
"matchPackageNames": ["renovate/renovate"],
+63 -15
View File
@@ -40,19 +40,6 @@ class ArtifactTests(unittest.TestCase):
release_module.render(io.StringIO(f' image: "{image}:prod" # note\n'), result)
self.assertEqual(result.getvalue(), f' image: "{image}@sha256:{"b" * 64}" # note\n')
def test_edu_release_digest_is_preserved(self):
with tempfile.TemporaryDirectory() as scratch:
path = Path(scratch) / 'release.json'
path.write_text(json.dumps(release()))
image = 'gcr.forust.xyz/forust/session-keeper'
line = f'image: {image}@sha256:{"e" * 64}\n'
result = io.StringIO()
with patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}):
release_module.render(io.StringIO(line), result)
self.assertEqual(result.getvalue(), line)
with self.assertRaises(ValueError):
release_module.render(io.StringIO(f'image: {image}:prod\n'), io.StringIO())
def test_unknown_image_cannot_emit_partial_manifest(self):
with tempfile.TemporaryDirectory() as scratch:
path = Path(scratch) / 'release.json'
@@ -107,10 +94,13 @@ class ArtifactTests(unittest.TestCase):
patch.object(release_module, 'command', side_effect=fake_command),
patch.object(subprocess, 'run', return_value=subprocess.CompletedProcess([], 0)),
):
release_module.build(root / 'release.json')
plan = root / 'plan.json'
release_module.prepare_images(plan)
for name in release_module.IMAGES:
release_module.build(root / f'{name}.json', name, plan)
self.assertEqual(built, ['errorpages/Dockerfile'])
self.assertTrue(all(not directory.exists() for directory in auth_directories))
self.assertEqual(json.loads((root / 'release.json').read_text())['sha'], 'e' * 40)
self.assertEqual(json.loads((root / 'error-pages.json').read_text())['sha'], 'e' * 40)
class DurableRunTests(unittest.TestCase):
@@ -213,6 +203,64 @@ class DurableRunTests(unittest.TestCase):
self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure')
class FailureSummaryTests(unittest.TestCase):
def test_build_failure_keeps_progress_and_does_not_expose_exception_text(self):
with tempfile.TemporaryDirectory() as scratch:
summary = Path(scratch) / 'summary.md'
def failed_build(_output, report, _name, _plan):
report.update(phase='Build or reuse xdfnx-homepage', current='xdfnx-homepage', built=['error-pages'])
report['images']['gcr.forust.xyz/forust/error-pages'] = 'sha256:' + 'b' * 64
raise RuntimeError('private value must not appear in the summary')
with (
patch.dict(os.environ, {'GITHUB_STEP_SUMMARY': str(summary), 'GITHUB_SHA': 'a' * 40}),
patch.object(release_module, 'build_images', side_effect=failed_build),
self.assertRaises(RuntimeError),
):
release_module.build(Path(scratch) / 'release.json', 'xdfnx-homepage', Path('plan.json'))
content = summary.read_text()
self.assertIn('**failure**', content)
self.assertIn('error-pages', content)
self.assertIn('xdfnx-homepage', content)
self.assertNotIn('private value', content)
def test_deploy_failure_reports_completed_apply_and_rollback_result(self):
with tempfile.TemporaryDirectory() as scratch:
state = Path(scratch)
directory = state / 'runs/123-1'
snapshot = directory / 'snapshot/before'
snapshot.mkdir(parents=True)
(directory / 'snapshot/current').write_text(str(snapshot))
(snapshot / 'failed-workloads').write_text('deployment app api\nROLLED_BACK=1\nUNRECOVERED=0\n')
controller.atomic_json(
directory / 'request.json', {'release': release(), 'mode': 'changed', 'refresh_images': False}
)
controller.atomic_json(
directory / 'status.json',
{
'state': 'failure',
'stages': {
'apply-k8s': {'result': 'success', 'exit_code': 0},
'verify-k8s': {'result': 'failure', 'exit_code': 1},
},
},
)
controller.atomic_json(directory / 'plan.json', {'selected': {'k8s': ['app'], 'compose': []}})
(directory / 'apply-events.jsonl').write_text(
json.dumps({'action': 'kubectl', 'target': 'app/k8s/api.yaml', 'result': 'success'}) + '\n'
)
output = io.StringIO()
with patch.object(controller, 'STATE', state), patch('sys.stdout', output):
controller.summary('123-1')
content = output.getvalue()
self.assertIn('verify-k8s | failure | 1', content)
self.assertIn('app/k8s/api.yaml', content)
self.assertIn('Workloads restored: **1**', content)
self.assertIn('manual recovery: **0**', content)
self.assertIn('Compose requires manual recovery', content)
class InstallerTests(unittest.TestCase):
def test_version_comparison_is_exact_without_network_or_host_packages(self):
with tempfile.TemporaryDirectory() as scratch:
+144
View File
@@ -0,0 +1,144 @@
"""Matrix release contracts and failure gates without a registry."""
import json
import os
import tempfile
import unittest
from pathlib import Path
from unittest.mock import Mock, call, patch
from test_cicd import release, release_module
def plan_data(changed):
targets = []
for name, (context, dockerfile) in release_module.IMAGES.items():
targets.append(
{
'name': name,
'image': f'gcr.forust.xyz/forust/{name}',
'context': context,
'dockerfile': dockerfile,
'inputs': ('d' if name in changed else 'c') * 64,
'reuse_digest': None if name in changed else 'sha256:' + 'b' * 64,
}
)
return {'sha': 'a' * 40, 'targets': targets}
class MatrixTests(unittest.TestCase):
def test_no_change_one_image_all_images_and_missing_baseline(self):
for changed in (set(), {'error-pages'}, set(release_module.IMAGES)):
with self.subTest(changed=changed), tempfile.TemporaryDirectory() as scratch:
api = Mock()
api.successful_runs.return_value = iter([{'id': 1}])
api.release.return_value = release()
expected = plan_data(changed)
fingerprints = {t['dockerfile']: t['inputs'] for t in expected['targets']}
output = Path(scratch) / 'plan.json'
with (
patch.dict(os.environ, {'GITHUB_SHA': 'a' * 40, 'GITHUB_RUN_ID': '2'}),
patch.object(release_module, 'command', return_value='a' * 40),
patch.object(release_module, 'Gitea', return_value=api),
patch.object(
release_module, 'fingerprint', side_effect=lambda _c, f, mapping=fingerprints: mapping[f]
),
):
release_module.prepare_images(output)
self.assertEqual(json.loads(output.read_text()), expected)
api.successful_runs.return_value = iter([])
with (
tempfile.TemporaryDirectory() as scratch,
patch.dict(os.environ, {'GITHUB_SHA': 'a' * 40}),
patch.object(release_module, 'command', return_value='a' * 40),
patch.object(release_module, 'Gitea', return_value=api),
patch.object(release_module, 'fingerprint', return_value='c' * 64),
):
output = Path(scratch) / 'plan.json'
release_module.prepare_images(output)
self.assertTrue(all(t['reuse_digest'] is None for t in json.loads(output.read_text())['targets']))
def test_incomplete_or_wrong_sha_fragments_cannot_publish_tags(self):
for wrong_sha in (False, True):
with self.subTest(wrong_sha=wrong_sha), tempfile.TemporaryDirectory() as scratch:
root = Path(scratch)
plan = root / 'plan.json'
plan.write_text(json.dumps(plan_data(set())))
for name in release_module.IMAGES:
if name == 'xdfnx-homepage' and not wrong_sha:
continue
folder = root / f'image-{name}'
folder.mkdir()
image = f'gcr.forust.xyz/forust/{name}'
folder.joinpath('image.json').write_text(
json.dumps(
{
'version': 1,
'sha': ('e' if wrong_sha else 'a') * 40,
'images': {image: 'sha256:' + 'b' * 64},
'inputs': {image: 'c' * 64},
}
)
)
with (
patch.object(release_module, 'checked_plan', return_value=plan_data(set())),
patch.object(release_module.subprocess, 'run') as execute,
self.assertRaises((ValueError, FileNotFoundError)),
):
release_module.finalize_images(root / 'release.json', root, plan)
execute.assert_not_called()
self.assertFalse((root / 'release.json').exists())
def test_manifest_only_release_pins_each_successful_digest(self):
with tempfile.TemporaryDirectory() as scratch:
root = Path(scratch)
data = plan_data(set())
for target in data['targets']:
folder = root / f'image-{target["name"]}'
folder.mkdir()
image = target['image']
folder.joinpath('image.json').write_text(
json.dumps(
{
'version': 1,
'sha': data['sha'],
'images': {image: target['reuse_digest']},
'inputs': {image: target['inputs']},
}
)
)
with (
patch.object(release_module, 'checked_plan', return_value=data),
patch.dict(os.environ, {'REGISTRY_USERNAME': 'test', 'REGISTRY_PASSWORD': 'placeholder'}),
patch.object(release_module.subprocess, 'run'),
patch.object(release_module, 'command') as execute,
):
release_module.finalize_images(root / 'release.json', root, root / 'plan.json')
self.assertEqual(
[entry.args for entry in execute.call_args_list],
[
call(
'docker',
'buildx',
'imagetools',
'create',
'--prefer-index=false',
'--tag',
f'{t["image"]}:sha-{data["sha"]}',
f'{t["image"]}@{t["reuse_digest"]}',
).args
for t in data['targets']
],
)
self.assertEqual(json.loads((root / 'release.json').read_text()), release())
def test_checkout_mismatch_cannot_build(self):
with tempfile.TemporaryDirectory() as scratch:
plan = Path(scratch) / 'plan.json'
plan.write_text(json.dumps(plan_data(set())))
with (
patch.dict(os.environ, {'GITHUB_SHA': 'e' * 40}),
patch.object(release_module, 'command', return_value='a' * 40),
self.assertRaisesRegex(ValueError, 'source commit'),
):
release_module.checked_plan(plan)