Compare commits

..
Author SHA1 Message Date
forust 033a69bc1e feat(paperless): add alternative Compose deployment
ci / Workflows (pull_request) Successful in 9s
ci / Compose (pull_request) Successful in 15s
ci / Shell (pull_request) Successful in 19s
ci / Formatting (pull_request) Successful in 24s
ci / Python and tests (pull_request) Successful in 11s
ci / YAML (pull_request) Successful in 10s
ci / Dockerfiles (pull_request) Successful in 6s
ci / Kubernetes (pull_request) Successful in 7s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-08 08:11:57 +02:00
forust 8a5d23b560 docs(postgres): format Paperless compatibility table
ci / Compose (pull_request) Successful in 26s
ci / Workflows (pull_request) Successful in 18s
ci / Shell (pull_request) Successful in 52s
ci / Formatting (pull_request) Successful in 27s
ci / Python and tests (pull_request) Successful in 8s
ci / YAML (pull_request) Successful in 17s
ci / Dockerfiles (pull_request) Successful in 13s
ci / Kubernetes (pull_request) Successful in 12s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-08 00:34:43 +02:00
forust fb46fb48b5 feat(paperless): deploy Paperless-ngx on shared PostgreSQL
ci / Compose (pull_request) Canceled after 0s
ci / Workflows (pull_request) Canceled after 0s
ci / Shell (pull_request) Canceled after 0s
ci / Formatting (pull_request) Canceled after 0s
ci / Python and tests (pull_request) Canceled after 0s
ci / YAML (pull_request) Canceled after 0s
ci / Dockerfiles (pull_request) Canceled after 0s
ci / Kubernetes (pull_request) Canceled after 0s
ci / image-plan (pull_request) Canceled after 0s
ci / Image (${{ matrix.name }}) (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
2026-10-08 00:32:54 +02:00
forust 4c7c53e0f2 fix(cicd): align apply timeout with stage budgets
ci / Compose (push) Successful in 11s
ci / Workflows (push) Successful in 6s
ci / Shell (push) Successful in 20s
ci / Formatting (push) Successful in 17s
ci / Python and tests (push) Successful in 7s
ci / YAML (push) Successful in 10s
ci / Dockerfiles (push) Successful in 4s
ci / image-plan (push) Successful in 13s
ci / Image (error-pages) (push) Successful in 15s
ci / Image (forust-homepage) (push) Successful in 15s
ci / Kubernetes (push) Successful in 7s
ci / Image (xdfnx-homepage) (push) Successful in 15s
ci / build (push) Successful in 19s
2026-10-07 23:28:17 +02:00
forust ba934265ac Merge pull request 'docs: record completed EDU ownership handoff' (#115) from docs/record-edu-handoff-complete into main
ci / Compose (push) Successful in 14s
ci / Workflows (push) Successful in 8s
ci / Python and tests (push) Successful in 7s
ci / YAML (push) Successful in 8s
ci / Kubernetes (push) Successful in 8s
ci / Image (forust-homepage) (push) Successful in 13s
ci / Shell (push) Successful in 18s
ci / Formatting (push) Successful in 24s
ci / Dockerfiles (push) Successful in 5s
ci / image-plan (push) Successful in 14s
ci / Image (error-pages) (push) Successful in 18s
ci / Image (xdfnx-homepage) (push) Successful in 17s
ci / build (push) Successful in 20s
Reviewed-on: #115
2026-10-07 21:10:52 +00:00
forust c7155808d9 docs: record completed EDU ownership handoff
ci / Workflows (pull_request) Successful in 13s
ci / Shell (pull_request) Successful in 31s
ci / Compose (pull_request) Successful in 26s
ci / Dockerfiles (pull_request) Successful in 11s
ci / Formatting (pull_request) Successful in 45s
ci / Python and tests (pull_request) Successful in 13s
ci / YAML (pull_request) Successful in 21s
ci / Kubernetes (pull_request) Successful in 13s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-07 20:43:12 +02:00
forust fc4d64bdb2 chore(streaming): disable Kubernetes routing
ci / Workflows (push) Successful in 17s
ci / Compose (push) Successful in 29s
ci / Shell (push) Successful in 57s
ci / Formatting (push) Successful in 29s
ci / Python and tests (push) Successful in 14s
ci / YAML (push) Successful in 14s
ci / Dockerfiles (push) Successful in 8s
ci / Kubernetes (push) Successful in 7s
ci / image-plan (push) Successful in 15s
ci / Image (error-pages) (push) Successful in 28s
ci / Image (forust-homepage) (push) Successful in 33s
ci / Image (xdfnx-homepage) (push) Successful in 34s
ci / build (push) Successful in 37s
2026-10-07 18:23:16 +02:00
forust a6f7fc6030 chore(streaming): disable streaming stack
ci / Compose (pull_request) Successful in 14s
ci / Formatting (pull_request) Successful in 21s
ci / Python and tests (pull_request) Successful in 7s
ci / YAML (pull_request) Successful in 8s
ci / Kubernetes (pull_request) Successful in 7s
ci / Workflows (pull_request) Successful in 8s
ci / Shell (pull_request) Successful in 19s
ci / Dockerfiles (pull_request) Successful in 5s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
ci / Compose (push) Successful in 14s
ci / Formatting (push) Successful in 20s
ci / Kubernetes (push) Successful in 7s
ci / image-plan (push) Successful in 11s
ci / Workflows (push) Successful in 8s
ci / Shell (push) Successful in 14s
ci / Python and tests (push) Successful in 8s
ci / YAML (push) Successful in 10s
ci / Dockerfiles (push) Successful in 5s
ci / Image (error-pages) (push) Successful in 13s
ci / Image (forust-homepage) (push) Successful in 15s
ci / Image (xdfnx-homepage) (push) Successful in 15s
ci / build (push) Successful in 16s
2026-10-07 17:48:54 +02:00
forust 11de1d1468 Merge pull request 'fix(cicd): preserve AIO tag in rollback snapshot' (#112) from fix/nextcloud-aio-rollback-tag into main
ci / Workflows (push) Successful in 7s
ci / Image (forust-homepage) (push) Successful in 13s
ci / Image (xdfnx-homepage) (push) Successful in 18s
ci / Compose (push) Successful in 14s
ci / Shell (push) Successful in 21s
ci / Formatting (push) Successful in 24s
ci / Python and tests (push) Successful in 9s
ci / YAML (push) Successful in 8s
ci / Dockerfiles (push) Successful in 5s
ci / Kubernetes (push) Successful in 9s
ci / image-plan (push) Successful in 15s
ci / Image (error-pages) (push) Successful in 17s
ci / build (push) Successful in 18s
Reviewed-on: #112
2026-10-07 15:29:11 +00:00
forust 64962d1a63 fix(cicd): preserve AIO tag in recovery snapshot
ci / Workflows (pull_request) Successful in 6s
ci / Kubernetes (pull_request) Successful in 6s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
ci / Compose (pull_request) Successful in 15s
ci / Shell (pull_request) Successful in 19s
ci / Formatting (pull_request) Successful in 30s
ci / Python and tests (pull_request) Successful in 8s
ci / YAML (pull_request) Successful in 10s
ci / Dockerfiles (pull_request) Successful in 4s
2026-10-07 16:53:14 +02:00
forust b08a0a927d Merge pull request 'fix(cicd): preserve Nextcloud AIO tag' (#111) from fix/preserve-nextcloud-aio-tag into main
ci / Compose (push) Successful in 27s
ci / Workflows (push) Successful in 14s
ci / Shell (push) Successful in 50s
ci / Kubernetes (push) Successful in 5s
ci / image-plan (push) Successful in 12s
ci / Image (forust-homepage) (push) Successful in 14s
ci / Image (xdfnx-homepage) (push) Successful in 15s
ci / Formatting (push) Successful in 24s
ci / Python and tests (push) Successful in 9s
ci / YAML (push) Successful in 11s
ci / Dockerfiles (push) Successful in 5s
ci / Image (error-pages) (push) Successful in 16s
ci / build (push) Successful in 18s
Reviewed-on: #111
2026-10-07 14:46:02 +00:00
forust 8203ba1b0b fix(cicd): preserve Nextcloud AIO image tag
ci / Workflows (pull_request) Successful in 9s
ci / Compose (pull_request) Successful in 14s
ci / Shell (pull_request) Successful in 19s
ci / Formatting (pull_request) Successful in 33s
ci / Python and tests (pull_request) Successful in 16s
ci / Dockerfiles (pull_request) Successful in 14s
ci / Kubernetes (pull_request) Successful in 17s
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
ci / YAML (pull_request) Successful in 19s
ci / image-plan (pull_request) Skipped
2026-10-07 14:45:10 +00:00
forust 48033b5495 Merge pull request 'fix(cicd): support Helm 4 release listing' (#110) from fix/helm4-list into main
ci / Workflows (push) Successful in 6s
ci / Shell (push) Successful in 17s
ci / Image (error-pages) (push) Successful in 13s
ci / Image (forust-homepage) (push) Successful in 13s
ci / Compose (push) Successful in 11s
ci / Formatting (push) Successful in 20s
ci / Python and tests (push) Successful in 6s
ci / YAML (push) Successful in 9s
ci / Dockerfiles (push) Successful in 4s
ci / Kubernetes (push) Successful in 7s
ci / image-plan (push) Successful in 11s
ci / Image (xdfnx-homepage) (push) Successful in 13s
ci / build (push) Successful in 16s
Reviewed-on: #110
2026-10-07 13:55:15 +00:00
forust 73d2af73e5 fix(cicd): support Helm 4 release listing
ci / build (pull_request) Skipped
ci / Workflows (pull_request) Successful in 7s
ci / Python and tests (pull_request) Successful in 5s
ci / Compose (pull_request) Successful in 11s
ci / Shell (pull_request) Successful in 17s
ci / Formatting (pull_request) Successful in 17s
ci / YAML (pull_request) Successful in 8s
ci / Dockerfiles (pull_request) Successful in 5s
ci / Kubernetes (pull_request) Successful in 7s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
2026-10-07 15:51:57 +02:00
forust 64253e005e Merge pull request 'fix(cicd): use Gitea artifact v4 backend' (#109) from fix/gitea-v4-artifacts into main
ci / Workflows (push) Successful in 8s
ci / Shell (push) Successful in 22s
ci / YAML (push) Successful in 9s
ci / image-plan (push) Successful in 49s
ci / Compose (push) Successful in 14s
ci / Formatting (push) Successful in 19s
ci / Python and tests (push) Successful in 8s
ci / Dockerfiles (push) Successful in 5s
ci / Kubernetes (push) Successful in 8s
ci / Image (error-pages) (push) Successful in 39s
ci / Image (forust-homepage) (push) Successful in 16s
ci / Image (xdfnx-homepage) (push) Successful in 13s
ci / build (push) Successful in 17s
Reviewed-on: #109
2026-10-07 13:35:36 +00:00
forust 69accd1752 fix(cicd): use Gitea artifact v4 backend
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Kubernetes (pull_request) Successful in 7s
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Workflows (pull_request) Successful in 7s
ci / Shell (pull_request) Successful in 18s
ci / Formatting (pull_request) Successful in 20s
ci / Python and tests (pull_request) Successful in 7s
ci / YAML (pull_request) Successful in 8s
ci / Dockerfiles (pull_request) Successful in 4s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-07 15:29:49 +02:00
forust 0cf4b08a95 Merge pull request 'fix(cicd): isolate pull request runner jobs' (#108) from fix/cicd-pr-runner into main
ci / Workflows (push) Successful in 7s
ci / Shell (push) Successful in 21s
ci / Python and tests (push) Successful in 5s
ci / Compose (push) Successful in 15s
ci / Formatting (push) Successful in 17s
ci / Kubernetes (push) Successful in 6s
ci / YAML (push) Successful in 9s
ci / Dockerfiles (push) Successful in 4s
renovate-ci / validate-renovate (push) Successful in 2m21s
ci / image-plan (push) Successful in 18s
ci / Image (error-pages) (push) Successful in 13s
ci / Image (xdfnx-homepage) (push) Successful in 12s
ci / Image (forust-homepage) (push) Successful in 11s
ci / build (push) Successful in 16s
Reviewed-on: #108
2026-10-07 13:03:01 +00:00
forust 86df5d9048 docs(cicd): document user-scoped PR runner
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (pull_request) Successful in 14s
ci / Shell (pull_request) Successful in 33s
ci / Formatting (pull_request) Successful in 36s
ci / YAML (pull_request) Successful in 28s
ci / Kubernetes (pull_request) Successful in 11s
ci / YAML (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 23s
ci / Python and tests (pull_request) Successful in 16s
ci / Dockerfiles (pull_request) Successful in 11s
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
ci / image-plan (pull_request) Skipped
2026-10-07 14:50:32 +02:00
forust d7441bbbc2 fix(cicd): isolate pull request runner jobs
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Compose (pull_request) Successful in 37s
ci / Shell (pull_request) Successful in 18s
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / Workflows (pull_request) Successful in 8s
ci / Python and tests (pull_request) Successful in 11s
ci / Kubernetes (pull_request) Successful in 8s
ci / Formatting (pull_request) Successful in 18s
ci / YAML (pull_request) Successful in 11s
ci / Dockerfiles (pull_request) Successful in 7s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
2026-10-07 14:39:57 +02:00
forust 2be089e048 Merge pull request 'Collect Headscale, NetBird, Gitea and Immich metrics' (#100) from feat/service-metrics into main
ci / Compose (push) Successful in 11s
ci / Workflows (push) Successful in 7s
ci / YAML (push) Successful in 9s
ci / Dockerfiles (push) Successful in 5s
ci / Shell (push) Successful in 16s
ci / Formatting (push) Successful in 17s
ci / Python and tests (push) Successful in 7s
ci / Kubernetes (push) Successful in 6s
ci / image-plan (push) Successful in 15s
ci / Image (error-pages) (push) Successful in 12s
ci / Image (forust-homepage) (push) Successful in 12s
ci / Image (xdfnx-homepage) (push) Successful in 11s
ci / build (push) Successful in 13s
Reviewed-on: #100
2026-10-07 11:46:16 +00:00
forust 83b2e68371 feat(metrics): collect Headscale, NetBird, Gitea and Immich metrics 2026-10-07 11:46:16 +00:00
forust 597f64cbb0 Merge pull request 'Show CI checks and deployment results' (#99) from codex/ci-visible-checks into main
ci / Workflows (push) Successful in 7s
ci / Formatting (push) Successful in 17s
ci / Python and tests (push) Successful in 7s
ci / YAML (push) Successful in 9s
ci / Compose (push) Successful in 11s
ci / Shell (push) Successful in 16s
ci / Kubernetes (push) Successful in 7s
ci / Dockerfiles (push) Successful in 5s
ci / Image (forust-homepage) (push) Successful in 12s
ci / Image (xdfnx-homepage) (push) Successful in 11s
renovate-ci / validate-renovate (push) Successful in 11s
ci / image-plan (push) Successful in 16s
ci / Image (error-pages) (push) Successful in 41s
ci / build (push) Successful in 14s
Reviewed-on: #99
2026-10-07 11:46:05 +00:00
forust f767f3ce1a docs: update EDU handoff status
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Shell (pull_request) Successful in 15s
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (pull_request) Successful in 6s
ci / Formatting (pull_request) Successful in 16s
ci / Python and tests (pull_request) Successful in 6s
ci / YAML (pull_request) Successful in 7s
ci / Dockerfiles (pull_request) Successful in 5s
ci / build (pull_request) Skipped
ci / Kubernetes (pull_request) Successful in 6s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 9s
2026-10-07 10:25:17 +02:00
forust 95e4d8f875 ci: add per-image jobs to homelab CI
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Workflows (pull_request) Successful in 7s
ci / Shell (pull_request) Successful in 15s
ci / Formatting (pull_request) Successful in 15s
ci / Python and tests (pull_request) Successful in 5s
ci / YAML (pull_request) Successful in 7s
ci / Dockerfiles (pull_request) Successful in 5s
ci / Kubernetes (pull_request) Successful in 6s
ci / image-plan (pull_request) Skipped
ci / Image (${{ matrix.name }}) (pull_request) Skipped
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 11s
2026-10-07 10:19:12 +02:00
forust 1d93588e06 refactor: remove EDU ownership from homelab 2026-10-07 10:19:12 +02:00
forust fb400eea6e Use plain text for the deploy request details
ci / Compose (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Kubernetes (push) Skipped
ci / YAML (pull_request) Successful in 11s
ci / build (pull_request) Skipped
ci / Workflows (push) Skipped
ci / Python and tests (push) Skipped
ci / Compose (pull_request) Successful in 13s
ci / Workflows (pull_request) Successful in 9s
ci / Shell (pull_request) Successful in 21s
ci / Formatting (pull_request) Successful in 22s
ci / Python and tests (pull_request) Successful in 6s
ci / Dockerfiles (pull_request) Successful in 5s
ci / Kubernetes (pull_request) Successful in 7s
2026-10-06 23:44:24 +02:00
forust 0c76426c17 Show failure details in CI and deploy summaries
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / Kubernetes (push) Skipped
ci / Compose (push) Skipped
ci / Shell (push) Skipped
ci / YAML (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Workflows (push) Skipped
ci / Compose (pull_request) Successful in 11s
ci / Workflows (pull_request) Failing after 8s
ci / Formatting (pull_request) Canceled after 0s
ci / Python and tests (pull_request) Canceled after 0s
ci / YAML (pull_request) Canceled after 0s
ci / Dockerfiles (pull_request) Canceled after 0s
ci / Kubernetes (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
ci / Shell (pull_request) Canceled after 7s
2026-10-06 23:44:00 +02:00
forust d74822cd27 Add CI and deploy summaries
ci / Compose (push) Skipped
ci / Workflows (push) Skipped
ci / Shell (push) Skipped
ci / Formatting (push) Skipped
ci / Python and tests (push) Skipped
ci / YAML (push) Skipped
ci / Kubernetes (push) Skipped
ci / Dockerfiles (push) Skipped
ci / Compose (pull_request) Successful in 14s
ci / Workflows (pull_request) Successful in 7s
ci / Shell (pull_request) Successful in 23s
ci / Formatting (pull_request) Successful in 26s
ci / Python and tests (pull_request) Successful in 8s
ci / YAML (pull_request) Successful in 14s
ci / Dockerfiles (pull_request) Successful in 7s
ci / Kubernetes (pull_request) Successful in 10s
ci / build (pull_request) Skipped
2026-10-06 23:28:56 +02:00
forust b677d553b4 refactor(ci): split checks into visible jobs
ci / YAML (pull_request) Successful in 11s
ci / Dockerfiles (pull_request) Successful in 6s
ci / Kubernetes (pull_request) Successful in 6s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (push) Skipped
renovate-ci / validate-renovate (pull_request) Skipped
ci / Compose (pull_request) Successful in 10s
ci / Workflows (pull_request) Successful in 5s
ci / Shell (pull_request) Successful in 22s
ci / Formatting (pull_request) Successful in 21s
ci / Python and tests (pull_request) Successful in 8s
2026-10-06 23:22:15 +02:00
forust 898463b759 Merge pull request 'refactor(ci): native VPS runner and durable incremental deploys' (#98) from codex/cicd-runner-deploy into main
ci / checks (push) Successful in 53s
renovate-ci / validate-renovate (push) Successful in 12s
ci / build (push) Successful in 1m44s
Reviewed-on: #98
2026-10-06 21:19:55 +00:00
forust 9a76529be8 refactor(ci): use native runner and durable incremental deploys
renovate-ci / validate-renovate (push) Skipped
ci / checks (pull_request) Successful in 51s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 21s
2026-10-06 23:08:47 +02:00
forust 5f9354b9a8 fix(edu): deploy reviewed application release with Redis authentication
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 14s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 13s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 10s
ci / build (push) Successful in 30s
2026-10-06 22:56:02 +02:00
forust fcd16f6128 Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.140.0' (#96) from renovate/renovate-self-update into main
renovate-ci / validate-renovate (push) Successful in 1m25s
ci / lint-compose (push) Successful in 12s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 16s
ci / lint-prettier (push) Successful in 23s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 7s
ci / validate (push) Successful in 6s
ci / build (push) Successful in 25s
Reviewed-on: #96
2026-10-06 17:51:40 +00:00
renovate-bot Bot 2ac2a94bb4 chore(deps): update renovate/renovate docker tag to v44.140.0 2026-10-06 17:51:40 +00:00
forust 3c7e358dd5 Merge pull request 'chore(deps): update lscr.io/linuxserver/qbittorrent docker tag to v20' (#97) from renovate/lscr.io-linuxserver-qbittorrent-20.x into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 16s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 15s
ci / lint-prettier (push) Successful in 20s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 14s
ci / lint-dockerfiles (push) Successful in 7s
ci / validate (push) Successful in 8s
ci / build (push) Successful in 24s
Reviewed-on: #97
2026-10-06 17:51:25 +00:00
renovate-bot Bot 9be8fb6c69 chore(deps): update lscr.io/linuxserver/qbittorrent docker tag to v20 2026-10-06 17:51:25 +00:00
forust 7fdeacffb8 feat(monitoring): stand down Prometheus server during VM trial
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 13s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 10s
ci / lint-prettier (push) Successful in 14s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 7s
ci / build (push) Successful in 19s
vmagent scrapes and remote-writes to VictoriaMetrics, so the Prometheus server scales to 0. Encoded as prometheusSpec.replicas in values instead of a kubectl patch, so helm keeps owning spec.replicas and Helm 4 server-side apply stops conflicting with the kubectl-patch field manager.
2026-10-06 19:29:30 +02:00
forust cef499de73 fix(deploy): skip VMAgent in secrets check before CRD install
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 12s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 15s
ci / lint-ruff (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 6s
ci / lint-prettier (push) Successful in 21s
ci / lint-yaml (push) Successful in 10s
ci / validate (push) Successful in 8s
ci / build (push) Successful in 21s
check_referenced_secrets ran kubectl create on vmagent.yaml even when the VMAgent CRD is not installed yet, failing validate with 'no matches for kind VMAgent'. Apply the same skip_uninstalled_vmagent_crd guard used by both dry-run loops.
2026-10-06 19:19:47 +02:00
forust 1b70a55300 fix(ci): handle malformed push before SHA
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 15s
ci / lint-prettier (push) Failing after 22s
ci / lint-ruff (push) Failing after 2s
ci / lint-yaml (push) Failing after 2s
ci / lint-dockerfiles (push) Failing after 3s
ci / validate (push) Failing after 2s
ci / build (push) Skipped
2026-10-06 18:21:19 +02:00
forust 3f2b4e9acf fix(deploy): skip VMAgent preflight before CRD install
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 10s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 7s
ci / validate (push) Successful in 10s
ci / build (push) Successful in 25s
2026-10-06 18:18:25 +02:00
forust 6057734a4f fix(renovate): sync generated configmap
renovate-ci / validate-renovate (push) Successful in 9s
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 10s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 7s
ci / build (push) Successful in 18s
2026-10-06 18:08:17 +02:00
forust 8eedf8b74b Merge pull request 'feat(monitoring): add VictoriaMetrics trial stack' (#95) from feat/victoria-metrics-migration into main
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 2m36s
ci / lint-shellcheck (push) Successful in 15s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 7s
renovate-ci / validate-renovate (push) Failing after 9s
ci / build (push) Successful in 19s
Reviewed-on: #95
2026-10-06 16:05:46 +00:00
forust 8c0e36a5c0 feat(monitoring): replace scrape dump with vmagent
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
ci / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (pull_request) Successful in 11s
ci / lint-actionlint (pull_request) Successful in 6s
ci / lint-shellcheck (pull_request) Successful in 16s
ci / lint-dockerfiles (pull_request) Successful in 6s
ci / lint-prettier (pull_request) Successful in 16s
ci / lint-ruff (pull_request) Successful in 7s
ci / lint-yaml (pull_request) Successful in 10s
ci / validate (pull_request) Successful in 7s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Failing after 1m40s
2026-10-06 18:03:59 +02:00
forust 78cd15f12c chore(monitoring): remove vmctl backfill job
renovate-ci / validate-renovate (pull_request) Skipped
ci / lint-compose (pull_request) Successful in 11s
ci / lint-yaml (pull_request) Successful in 11s
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
ci / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-actionlint (pull_request) Successful in 5s
ci / lint-shellcheck (pull_request) Successful in 16s
ci / lint-prettier (pull_request) Successful in 15s
ci / lint-ruff (pull_request) Successful in 6s
ci / lint-dockerfiles (pull_request) Successful in 7s
ci / validate (pull_request) Successful in 7s
ci / build (pull_request) Skipped
One-shot Prometheus history backfill is complete; drop the Job and clean up related comments.
2026-10-06 17:53:24 +02:00
forust 18c633c242 feat(monitoring): add VictoriaMetrics trial stack
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
ci / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
renovate-ci / validate-renovate (pull_request) Skipped
ci / lint-compose (pull_request) Successful in 12s
ci / lint-actionlint (pull_request) Successful in 6s
ci / lint-shellcheck (pull_request) Successful in 13s
ci / lint-prettier (pull_request) Successful in 21s
ci / lint-ruff (pull_request) Successful in 7s
ci / lint-yaml (pull_request) Successful in 12s
ci / lint-dockerfiles (pull_request) Successful in 7s
ci / validate (pull_request) Successful in 7s
ci / build (pull_request) Skipped
2026-10-06 17:45:44 +02:00
forust 4510394531 Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.139.0' (#81) from renovate/renovate-self-update into main
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 15s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 9s
renovate-ci / validate-renovate (push) Successful in 13s
ci / build (push) Successful in 20s
Reviewed-on: #81
2026-10-06 15:33:45 +00:00
renovate-bot Bot 2545312db1 chore(deps): update renovate/renovate docker tag to v44.139.0 2026-10-06 15:33:45 +00:00
forust 8ce0b809c0 Merge pull request 'chore(deps): update all minor updates' (#77) from renovate/all-minor into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
Reviewed-on: #77
2026-10-06 15:33:28 +00:00
renovate-bot Bot 506c04e15c chore(deps): update all minor updates 2026-10-06 15:33:28 +00:00
forust bdcbb3d5af Merge pull request 'chore(deps): update all patch updates' (#82) from renovate/all-patch into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
Reviewed-on: #82
2026-10-06 15:33:09 +00:00
renovate-bot Bot faac6febb8 chore(deps): update all patch updates 2026-10-06 15:33:09 +00:00
forust 695da1308c Merge pull request 'fix(renovate): restore custom extraction and update groups' (#93) from fix/renovate-extraction-groups into main
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 17s
ci / lint-prettier (push) Successful in 18s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 8s
renovate-ci / validate-renovate (push) Successful in 11s
ci / build (push) Successful in 37s
Reviewed-on: #93
2026-10-06 15:31:25 +00:00
forust 552b22cb66 fix(renovate): include self-update Compose filename 2026-10-06 15:31:25 +00:00
forust 3756e60c95 fix(renovate): restore custom extraction and update groups 2026-10-06 15:31:25 +00:00
forust d0d217ddf4 Merge pull request 'fix(deploy): skip verify and smoke when deployment is disabled' (#94) from fix/deploy-skip-when-disabled into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
Reviewed-on: #94
2026-10-06 15:24:17 +00:00
forust 24f84bab2f fix(deploy): skip verification when deployment is disabled
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
ci / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
renovate-ci / validate-renovate (pull_request) Skipped
ci / lint-compose (pull_request) Canceled after 0s
ci / lint-actionlint (pull_request) Canceled after 0s
ci / lint-shellcheck (pull_request) Canceled after 0s
ci / lint-prettier (pull_request) Canceled after 0s
ci / lint-ruff (pull_request) Canceled after 0s
ci / lint-yaml (pull_request) Canceled after 0s
ci / lint-dockerfiles (pull_request) Canceled after 0s
ci / validate (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
2026-10-06 15:24:06 +00:00
forust 0b8a3c16a7 Merge pull request 'fix(glance): mount CSS from the correct ConfigMap' (#87) from fix/glance-assets into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
Reviewed-on: #87
2026-10-06 15:17:47 +00:00
forust e9a68aae77 fix(glance): mount CSS from the assets ConfigMap 2026-10-06 15:17:47 +00:00
forust 5359df5ed6 Merge pull request 'fix(postgres): include the required NetBox database password' (#88) from fix/postgres-env-example into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 14s
ci / lint-actionlint (push) Successful in 8s
ci / lint-shellcheck (push) Successful in 17s
ci / lint-prettier (push) Successful in 24s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 11s
ci / build (push) Canceled after 0s
Reviewed-on: #88
2026-10-06 15:17:19 +00:00
forust 8aea0f0d13 fix(postgres): include required NetBox password in Compose env example 2026-10-06 15:17:19 +00:00
forust 65d4f2d482 Merge pull request 'fix(traefik): correct AdGuard and SearXNG Compose rules' (#90) from fix/compose-router-rules into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 12s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 13s
ci / lint-prettier (push) Successful in 16s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 10s
ci / build (push) Successful in 23s
Reviewed-on: #90
2026-10-06 15:10:49 +00:00
forust 2fe9b7632f fix(traefik): correct AdGuard and SearXNG Compose router expressions 2026-10-06 15:10:49 +00:00
forust e436d89eef Merge pull request 'fix(netbird): restore Compose setup and runtime renderer' (#86) from fix/netbird-compose-runtime into main
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 15s
ci / lint-prettier (push) Successful in 21s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 7s
ci / build (push) Successful in 20s
Reviewed-on: #86
2026-10-06 15:10:12 +00:00
forust 8729cb5062 fix(netbird): restore Compose setup and server entrypoint
ci / validate (push) Skipped
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
renovate-ci / validate-renovate (push) Skipped
renovate-ci / validate-renovate (pull_request) Skipped
ci / lint-compose (pull_request) Successful in 12s
ci / lint-actionlint (pull_request) Successful in 8s
ci / lint-shellcheck (pull_request) Successful in 21s
ci / lint-prettier (pull_request) Successful in 17s
ci / lint-ruff (pull_request) Successful in 6s
ci / lint-yaml (pull_request) Successful in 9s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 6s
ci / build (pull_request) Skipped
2026-10-06 15:08:46 +00:00
forust f1e4a1088d Merge pull request 'fix(ci): deduplicate PR checks and filter deploy triggers' (#92) from fix/ci-trigger-dedup into main
renovate-ci / validate-renovate (push) Successful in 9s
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 12s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 7s
ci / validate (push) Successful in 7s
ci / build (push) Successful in 19s
Reviewed-on: #92
2026-10-06 15:06:47 +00:00
forust b357ef95d8 fix(deploy): only create automatic runs for main CI
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
ci / validate (push) Skipped
ci / lint-prettier (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (pull_request) Successful in 13s
ci / lint-actionlint (pull_request) Successful in 6s
ci / lint-shellcheck (pull_request) Successful in 11s
ci / lint-prettier (pull_request) Successful in 18s
ci / lint-ruff (pull_request) Successful in 8s
ci / lint-yaml (pull_request) Successful in 13s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 9s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 12s
2026-10-06 17:00:05 +02:00
forust 67422663b7 fix(ci): avoid duplicate branch and PR runs
ci / validate (push) Skipped
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (pull_request) Canceled after 0s
ci / lint-actionlint (pull_request) Canceled after 0s
ci / lint-shellcheck (pull_request) Canceled after 0s
ci / lint-prettier (pull_request) Canceled after 0s
ci / lint-ruff (pull_request) Canceled after 0s
ci / lint-yaml (pull_request) Canceled after 0s
ci / lint-dockerfiles (pull_request) Canceled after 0s
ci / validate (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
renovate-ci / validate-renovate (pull_request) Successful in 10s
2026-10-06 16:59:26 +02:00
forust 88705fec88 Merge pull request 'fix(deploy): reject unsafe per-file pruning' (#85) from fix/deploy-prune-guard into main
renovate-ci / validate-renovate (push) Successful in 9s
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 11s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 8s
ci / build (push) Successful in 23s
Reviewed-on: #85
2026-10-06 14:57:22 +00:00
forust c098807aa4 fix(deploy): reject destructive per-file pruning before apply
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 14s
ci / lint-actionlint (push) Successful in 8s
ci / lint-shellcheck (push) Successful in 13s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 10s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 12s
ci / lint-actionlint (pull_request) Successful in 5s
ci / lint-shellcheck (pull_request) Successful in 11s
ci / lint-prettier (pull_request) Successful in 16s
ci / lint-ruff (pull_request) Successful in 8s
ci / lint-yaml (pull_request) Successful in 11s
ci / lint-dockerfiles (pull_request) Successful in 7s
ci / validate (pull_request) Successful in 7s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 10s
2026-10-06 14:57:11 +00:00
forust e1eee2d3c7 Merge pull request 'feat(reloader): enable deploy and integrate configuration reloads' (#91) from fix/reloader-integration into main
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 9s
Reviewed-on: #91
2026-10-06 14:55:16 +00:00
forust ff83daed1e feat(reloader): enable deployment and reload runtime-config consumers
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 13s
ci / lint-actionlint (push) Successful in 9s
ci / lint-shellcheck (push) Successful in 14s
ci / lint-prettier (push) Successful in 16s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 8s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 7s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 13s
ci / lint-actionlint (pull_request) Successful in 4s
ci / lint-shellcheck (pull_request) Successful in 12s
ci / lint-prettier (pull_request) Successful in 18s
ci / lint-ruff (pull_request) Successful in 8s
ci / lint-yaml (pull_request) Successful in 13s
ci / lint-dockerfiles (pull_request) Successful in 8s
ci / validate (pull_request) Successful in 7s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 9s
2026-10-06 14:54:45 +00:00
forust 080ae343e6 Merge pull request 'fix(deploy): validate resolved Compose config and namespaced Secrets' (#84) from fix/deploy-validation into main
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 11s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 9s
ci / validate (push) Successful in 11s
renovate-ci / validate-renovate (push) Successful in 10s
ci / build (push) Successful in 29s
Reviewed-on: #84
2026-10-06 14:54:27 +00:00
forust 8d3185f8ab fix(ci): install jq for deploy validation regressions
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 14s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 7s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 11s
ci / lint-actionlint (pull_request) Successful in 7s
ci / lint-shellcheck (pull_request) Successful in 12s
ci / lint-prettier (pull_request) Successful in 14s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Successful in 11s
ci / lint-dockerfiles (pull_request) Successful in 6s
ci / validate (pull_request) Successful in 6s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 8s
2026-10-06 16:31:48 +02:00
forust ff40a71145 fix(deploy): resolve Compose config and check namespaced pod Secrets
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Failing after 10s
ci / lint-prettier (push) Successful in 13s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 7s
ci / validate (push) Successful in 8s
ci / build (push) Skipped
ci / lint-compose (pull_request) Canceled after 0s
ci / lint-actionlint (pull_request) Canceled after 0s
ci / lint-shellcheck (pull_request) Canceled after 0s
ci / lint-prettier (pull_request) Canceled after 0s
ci / lint-ruff (pull_request) Canceled after 0s
ci / lint-yaml (pull_request) Canceled after 0s
ci / lint-dockerfiles (pull_request) Canceled after 0s
ci / validate (pull_request) Canceled after 0s
ci / build (pull_request) Canceled after 0s
renovate-ci / validate-renovate (pull_request) Successful in 8s
2026-10-06 15:58:44 +02:00
forust cc9c3dea88 feat(traefik): expose insecure API on 8080 for Homarr integration
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 8s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-prettier (push) Successful in 16s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 7s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 38s
ClusterIP access to api@internal from Homarr pod. LAN-reachable on 192.168.80.2:8080, accepted.
2026-10-06 11:27:38 +02:00
forust 815cd85b9a feat(homarr): deploy dashboard to k8s on home subdomain
Local-only IngressRoute (home.workstation.internal, home.gigaforust.internal), prod commented out. Compose stack for test stand.
2026-10-06 11:27:35 +02:00
forust 9021eddbc3 chore(deps): pin streaming images, isolate python and floating tags
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-prettier (push) Successful in 13s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 7s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 19s
Python Y-bumps arrived as minor 3.11->3.14 and floating tags (:latest/:beta) were automerged blindly. Pin the linuxserver stack to digest-verified tags and keep python plus rolling images in manual review groups.
2026-10-05 23:53:38 +02:00
forust 2adf17c307 Merge pull request 'chore(deps): update all patch updates' (#76) from renovate/all-patch into main
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-prettier (push) Successful in 12s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 17s
Reviewed-on: #76
2026-10-05 21:41:00 +00:00
renovate-bot Bot 1add5b5cd7 chore(deps): update all patch updates 2026-10-05 21:41:00 +00:00
forust 34f10211ab Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.136.0' (#79) from renovate/renovate-self-update into main
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 13s
ci / lint-yaml (push) Successful in 10s
ci / validate (push) Successful in 7s
renovate-ci / validate-renovate (push) Successful in 9s
ci / lint-ruff (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 5s
ci / build (push) Successful in 18s
Reviewed-on: #79
2026-10-05 21:39:21 +00:00
renovate-bot Bot 02447f2946 chore(deps): update renovate/renovate docker tag to v44.136.0
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-ruff (push) Successful in 5s
ci / validate (push) Successful in 9s
ci / lint-prettier (push) Successful in 13s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 5s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 8s
ci / lint-actionlint (pull_request) Successful in 6s
ci / lint-shellcheck (pull_request) Successful in 12s
ci / lint-prettier (pull_request) Successful in 14s
ci / lint-ruff (pull_request) Successful in 6s
ci / lint-yaml (pull_request) Successful in 9s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 6s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 1m18s
2026-10-05 16:19:11 +00:00
forust 8799962b1c Revert "feat(adguard): add netbird sidecar"
ci / lint-dockerfiles (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 9s
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 10s
ci / validate (push) Successful in 6s
ci / build (push) Successful in 20s
This reverts commit 7a81b4ea8b.
2026-10-04 17:40:19 +02:00
forust fb80024fa2 feat(streaming): add bazarr for subtitles
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 7s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
ci / lint-yaml (push) Successful in 11s
ci / lint-dockerfiles (push) Successful in 6s
2026-10-04 17:25:00 +02:00
forust 0f1a788874 Merge pull request 'feat(streaming): compose *arr streaming stack (k8s routing)' (#78) from feat/streaming-compose-test into main
ci / lint-compose (push) Successful in 8s
ci / lint-prettier (push) Successful in 14s
ci / lint-ruff (push) Successful in 6s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 10s
ci / build (push) Successful in 54s
Reviewed-on: #78
2026-10-04 14:04:28 +00:00
forust 39df442e60 fix(streaming): trailing newline for yamllint
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 6s
ci / lint-ruff (push) Successful in 6s
ci / lint-prettier (push) Successful in 15s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 10s
ci / lint-actionlint (pull_request) Successful in 5s
ci / lint-shellcheck (pull_request) Successful in 8s
ci / lint-prettier (pull_request) Successful in 14s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Successful in 8s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 6s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 8s
2026-10-04 16:03:40 +02:00
forust 62a451773e feat(streaming): compose *arr streaming stack (k8s routing)
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 10s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Failing after 15s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Failing after 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 8s
ci / lint-actionlint (pull_request) Successful in 4s
ci / lint-prettier (pull_request) Failing after 14s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Failing after 11s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 9s
ci / lint-shellcheck (pull_request) Successful in 7s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 7s
2026-10-04 15:57:08 +02:00
forust f196099491 feat(adguard): add DNS readiness probe
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 33s
2026-10-03 22:39:25 +02:00
forust 66502b8279 fix(traefik): protect dashboard with security-chain 2026-10-03 22:39:25 +02:00
forust 9018c091fa feat(uptime-kuma): add ServiceMonitor, alerts and secrets example 2026-10-03 22:39:24 +02:00
forust 114608af2f fix(uptime-kuma): label service and name http port 2026-10-03 22:39:23 +02:00
forust 51a73fb213 chore(gitea): switch domain to git.forust.xyz (28.0.0 update) 2026-10-03 22:37:48 +02:00
forust 939fad4a23 chore(renovate,crowdsec): group minor/patch updates and whitelist VPS
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 6s
ci / lint-prettier (push) Successful in 11s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 10s
ci / build (push) Successful in 17s
Renovate: group all minor updates and all patch updates into reviewable PRs. Crowdsec: whitelist static VPS IP. Remove stale cloudflared/k8s/active marker.
2026-10-03 11:57:46 +02:00
forust 12d5cc6202 Merge pull request 'chore(deps): update docker.gitea.com/gitea docker tag to v28' (#74) from renovate/docker.gitea.com-gitea-28.x into main
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / lint-prettier (push) Successful in 13s
ci / lint-ruff (push) Successful in 5s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 11s
ci / build (push) Successful in 16s
Reviewed-on: #74
2026-10-03 09:44:13 +00:00
renovate-bot Bot a91862d288 chore(deps): update docker.gitea.com/gitea docker tag to v28 2026-10-03 09:44:13 +00:00
forust a6b84d86b4 Merge pull request 'chore(deps): update gitea/gitea docker tag to v28' (#75) from renovate/gitea-gitea-28.x into main
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 8s
Reviewed-on: #75
2026-10-03 09:44:06 +00:00
renovate-bot Bot f2dd9b3fbc chore(deps): update gitea/gitea docker tag to v28
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (pull_request) Successful in 7s
ci / lint-actionlint (pull_request) Successful in 5s
ci / lint-shellcheck (pull_request) Successful in 7s
ci / lint-yaml (pull_request) Successful in 8s
ci / lint-dockerfiles (pull_request) Successful in 4s
ci / validate (pull_request) Successful in 6s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 9s
ci / lint-prettier (pull_request) Successful in 13s
ci / lint-ruff (pull_request) Successful in 5s
ci / build (pull_request) Skipped
ci / lint-compose (push) Successful in 7s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 6s
ci / lint-prettier (push) Successful in 13s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
2026-10-03 09:42:06 +00:00
forust f3a5cedc80 Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.132.2' (#69) from renovate/renovate-self-update into main
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 14s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 6s
ci / build (push) Successful in 19s
renovate-ci / validate-renovate (push) Successful in 1m10s
Reviewed-on: #69
2026-10-03 09:31:07 +00:00
renovate-bot Bot b0282e24f3 chore(deps): update renovate/renovate docker tag to v44.132.2
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 6s
ci / lint-prettier (push) Successful in 12s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 8s
ci / lint-actionlint (pull_request) Successful in 4s
ci / lint-shellcheck (pull_request) Successful in 6s
ci / lint-prettier (pull_request) Successful in 12s
ci / lint-ruff (pull_request) Successful in 6s
ci / lint-yaml (pull_request) Successful in 8s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 6s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Failing after 42s
2026-10-03 09:20:52 +00:00
forust e208ccae62 Merge pull request 'chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.29' (#70) from renovate/container-patch-updates into main
renovate-ci / validate-renovate (push) Successful in 8s
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 6s
ci / lint-prettier (push) Successful in 14s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 5s
ci / build (push) Successful in 18s
Reviewed-on: #70
2026-10-03 09:20:36 +00:00
renovate-bot Bot c65fee7b29 chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.29 2026-10-03 09:20:36 +00:00
forust d3892d2ed5 Merge pull request 'chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v3.4.0' (#71) from renovate/ghcr.io-henriquesebastiao-downtify-3.x into main
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 8s
Reviewed-on: #71
Reviewed-by: forust <vzlomdsisma@gmail.com>
2026-10-03 09:20:06 +00:00
renovate-bot Bot fb025d221b chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v3.4.0 2026-10-03 09:20:06 +00:00
forust 7164b48275 Merge pull request 'chore(deps): update ghcr.io/lukegus/termix docker tag to v2.9.0' (#72) from renovate/ghcr.io-lukegus-termix-2.x into main
renovate-ci / validate-renovate (push) Successful in 8s
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 14s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 8s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 6s
ci / build (push) Successful in 17s
Reviewed-on: #72
2026-10-03 09:19:37 +00:00
renovate-bot Bot 3f484a37c6 chore(deps): update ghcr.io/lukegus/termix docker tag to v2.9.0 2026-10-03 09:19:37 +00:00
forust 956596e6aa Merge pull request 'chore(deps): update docker.n8n.io/n8nio/n8n docker tag to v2.42.2' (#68) from renovate/docker.n8n.io-n8nio-n8n-2.x into main
renovate-ci / validate-renovate (push) Successful in 8s
ci / lint-compose (push) Successful in 7s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 13s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 5s
ci / build (push) Successful in 17s
Reviewed-on: #68
2026-10-03 09:17:01 +00:00
renovate-bot Bot 3320018232 chore(deps): update docker.n8n.io/n8nio/n8n docker tag to v2.42.2 2026-10-03 09:17:01 +00:00
forust 65709e005b Merge pull request 'chore(deps): update netbirdio/netbird-server docker tag to v0.80.0' (#73) from renovate/netbirdio-netbird-server-0.x into main
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 12s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 7s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 18s
Reviewed-on: #73
2026-10-03 09:16:28 +00:00
renovate-bot Bot c24bf90629 chore(deps): update netbirdio/netbird-server docker tag to v0.80.0 2026-10-03 09:16:28 +00:00
forust 78e6363fe2 chore(gitea): add git.forust.xyz route
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 5s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 12s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 17s
2026-10-03 11:07:48 +02:00
forust 22c2f1e108 extract dtek_notif subtree to its own repo
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 10s
2026-10-03 10:48:01 +02:00
forust 64ba4ce9f1 Revert "chore(prometheus): drop dead grafana routes and cert"
This reverts commit 314ec0cda7.
2026-10-03 10:40:39 +02:00
forust 7a81b4ea8b feat(adguard): add netbird sidecar 2026-10-03 10:36:15 +02:00
forust 0859479c0f feat(ingress): replace traefik crowdsec plugin with firewall bouncer
ci / lint-compose (push) Successful in 9s
ci / lint-actionlint (push) Successful in 4s
ci / lint-shellcheck (push) Successful in 7s
ci / lint-prettier (push) Successful in 12s
ci / lint-ruff (push) Successful in 6s
ci / lint-yaml (push) Successful in 9s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Failing after 14m22s
Move L3 enforcement to the host firewall-bouncer (systemd, nftables): drop the Traefik plugin, its secrets volume and the crowdsec Middleware, remove bouncer refs from all IngressRoutes. Disable the http-generic-bf scenario (403-burst bans hurt legit automation under L3 enforcement). Add a Gateway API PoC for homepages prod and CrowdSec PrometheusRule alerts.
2026-09-30 20:14:25 +02:00
forust 872f64b887 fix(ci): silence intentional SC2029 in ssh-run.sh
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 9s
ci / lint-prettier (push) Successful in 16s
ci / lint-ruff (push) Successful in 7s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 8s
renovate-ci / validate-renovate (push) Successful in 22s
ci / build (push) Successful in 38s
2026-09-29 14:54:17 +02:00
forust a90fb19ed4 fix(traefik): size probes for HDD stalls
ci / lint-compose (push) Successful in 12s
ci / lint-actionlint (push) Successful in 9s
ci / lint-shellcheck (push) Failing after 25s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 9s
ci / build (push) Skipped
renovate-ci / validate-renovate (push) Successful in 12s
Single replica is the whole ingress; liveness kills at 2s timeouts took every public service down in a loop. Same stall-sized budgets as postgres/metallb. Applied live via helm (pinned 41.5.0, values from git).
2026-09-29 14:47:53 +02:00
forust 6f4cd03f4b feat(deploy): serialize apply stages and wait for calm node
apply-k8s and apply-compose share a workstation flock so host docker churn never overlaps cluster churn. Each helm upgrade and the apply loop wait up to 10m for load <28 first, so a deploy never piles onto an already-hot node (the load-40/netbird-death/pending-helm spiral).
2026-09-29 14:42:56 +02:00
forust e56662194b fix(ci): log in to registry for manifest-only main pushes
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 9s
ci / lint-prettier (push) Successful in 15s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 13s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 9s
renovate-ci / validate-renovate (push) Successful in 10s
ci / build (push) Successful in 27s
The pin step writes manifest PUTs on every main push, but login was gated on services != ''. Manifest-only pushes skipped login and pushed anonymously (401); this only worked before via a stale persistent login on the old runner.
2026-09-29 14:21:57 +02:00
forust 4856348a6e chore(searxng,glance): disable services
renovate-ci / validate-renovate (push) Successful in 15s
ci / lint-compose (push) Successful in 13s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 11s
ci / lint-prettier (push) Successful in 18s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 13s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 8s
ci / build (push) Failing after 19s
Workloads, services, routes and certs removed from the cluster. Configs, manifests and PVCs kept for easy re-enable via k8s/active.
2026-09-29 13:59:59 +02:00
forust 49d0da1dd0 chore(termix): disable service
renovate-ci / validate-renovate (push) Successful in 11s
ci / lint-compose (push) Successful in 17s
ci / lint-actionlint (push) Successful in 8s
ci / lint-shellcheck (push) Successful in 10s
ci / lint-prettier (push) Successful in 18s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 13s
ci / lint-dockerfiles (push) Successful in 8s
ci / validate (push) Successful in 12s
ci / build (push) Failing after 17s
Deployment was stuck in ImagePullBackOff and only generated log churn. Workload, service, routes and certs removed from the cluster; PVC, config and manifests kept so it can be re-enabled by restoring k8s/active.
2026-09-29 13:58:27 +02:00
forust 6bc938436f Merge pull request 'chore(deps): update grafana/grafana docker tag to v13.2.3' (#67) from renovate/grafana-monorepo into main
ci / lint-compose (push) Successful in 11s
ci / lint-actionlint (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 10s
ci / lint-prettier (push) Successful in 18s
ci / lint-ruff (push) Successful in 8s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 7s
renovate-ci / validate-renovate (push) Successful in 12s
ci / validate (push) Successful in 8s
ci / build (push) Failing after 2m13s
Reviewed-on: #67
2026-09-29 11:57:17 +00:00
renovate-bot Bot 04b33d0736 chore(deps): update grafana/grafana docker tag to v13.2.3 2026-09-29 11:57:17 +00:00
forust 5dcad7eb38 fix(ci): resolve uv from BIN_DIR in install-ci-tools
ci / lint-compose (push) Successful in 14s
ci / lint-actionlint (push) Successful in 10s
ci / lint-shellcheck (push) Successful in 11s
ci / lint-prettier (push) Successful in 17s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 12s
ci / lint-dockerfiles (push) Successful in 9s
ci / validate (push) Successful in 9s
renovate-ci / validate-renovate (push) Successful in 1m34s
ci / build (push) Failing after 3m3s
Callers prepend BIN_DIR to PATH only after the script exits, so the bare uv invocation in install_uv_tool died with 127 on clean runners. Export BIN_DIR to PATH inside the script and invoke the just-installed binary by absolute path.
2026-09-29 13:31:28 +02:00
forust d0872bc918 feat(deploy): autodeploy off unless AUTODEPLOY=true
renovate-ci / validate-renovate (push) Successful in 2m1s
ci / lint-shellcheck (push) Successful in 24s
ci / lint-prettier (push) Successful in 6s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 3s
ci / lint-compose (push) Successful in 16s
ci / lint-actionlint (push) Successful in 8s
ci / build (push) Failing after 18s
Pushes no longer reach the cluster by default; set the AUTODEPLOY repo variable to 'true' to re-enable, or dispatch manually. Skips propagate through the existing needs chain.
2026-09-29 12:54:17 +02:00
forust 91dd749a29 feat(deploy): AUTODEPLOY kill-switch via repo variable
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 3s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 15s
ci / build (push) Canceled after 0s
Set AUTODEPLOY=false under Settings -> Actions -> Variables and pushes stop deploying with no commit; unset means on. Manual Run workflow always bypasses the switch. The existing needs/skipped chaining propagates the skip through verify and smoke untouched.
2026-09-29 12:52:35 +02:00
forust b6e1dc0362 fix(immich): size postgres probes for HDD stalls
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 17s
ci / build (push) Successful in 13s
Postmaster was SIGKILLed in a loop: 70s fsync stalls on the loaded rotational disk outlasted the 5-minute startup budget and the 60s liveness tolerance, and every kill bought another full WAL replay. Startup budget 15min, liveness 5x60s. Already applied live with kubectl; this keeps git in sync.
2026-09-29 12:36:04 +02:00
forust a2af854867 fix(alerts): measure traefik latency per route via exported_service
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 11s
ci / build (push) Successful in 6s
The service label the rule grouped by is the scrape target name, not the backend: with honorLabels=false the per-route value traefik emits is renamed to exported_service, so the old rule measured one global aggregate and both exclusions matched nothing. Group the latency and 5xx alerts by exported_service with exclusions in traefik's real namespace-routename-hash format. Already applied live with kubectl; this commit keeps git in sync.
2026-09-29 09:55:02 +02:00
forust 314ec0cda7 chore(prometheus): drop dead grafana routes and cert
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 19s
ci / build (push) Successful in 6s
No grafana is deployed from this stack, so grafana-prod/grafana-local only ever served 503s. Live objects already deleted directly; this keeps git from resurrecting them on the next deploy.
2026-09-29 09:33:38 +02:00
forust e45ae10204 feat(immich): migrate postgres 14 to 16 with dump/restore
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 5s
ci / lint-ruff (push) Successful in 3s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 3s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 1m7s
ci / build (push) Successful in 11s
Reverts the v16 revert: data migrated via pg_dumpall from PG14 into a fresh PG16 data directory (same extension versions vectorchord 0.4.3/pgvectors 0.2.0). Verified 1944 assets, 2 users, 10 albums on 16.10. Note: the restore OOMed once at the 2Gi limit during index builds and recovered via WAL replay; worth watching under PG16 load.
2026-09-29 09:16:58 +02:00
forust 357ee26111 Revert "Merge pull request 'chore(deps): update ghcr.io/immich-app/postgres docker tag to v16' (#65) from renovate/ghcr.io-immich-app-postgres-16.x into main"
ci / lint-compose (push) Successful in 5s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 34s
ci / build (push) Successful in 6s
This reverts commit 0c9743e241, reversing
changes made to 2cb06debc5.
2026-09-29 08:25:43 +02:00
forust 436fd1ecae chore: extract userbot subtree to its own repo
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 17s
ci / build (push) Successful in 8s
userbot/ (bot plus panel) now lives at /home/forust/userbot as a clone of forust/userbot instead of a subtree in homelab. Cleans up the pipeline references that only existed for it: scan-deps, test-backend and test-frontend jobs, the userbot build matrix entries, the deploy panel hook, and the userbot-only pyrightconfig. Running cluster workloads are untouched; homelab just stops building and testing upstream's code.
2026-09-29 08:24:14 +02:00
forust a564dd8b67 fix(alerts): exclude netbird streams from traefik latency alert
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 12s
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 17s
ci / test-backend (push) Successful in 8s
ci / build (push) Successful in 10s
SignalExchange ConnectStream holds 60s gRPC streams by design; at night they exceed 5% of samples and pin P95 to the 5.0s bucket ceiling, flapping the alert. Also fixes the stale xui exclusion pattern, which matched no real service label.
2026-09-29 08:12:07 +02:00
forust c2c90c490d Merge pull request 'chore(deps): update ghcr.io/autobrr/netronome docker tag to v0.15.0' (#62) from renovate/ghcr.io-autobrr-netronome-0.x into main
ci / lint-compose (push) Successful in 2s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 15s
ci / test-backend (push) Successful in 7s
ci / test-frontend (push) Successful in 11s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 13s
ci / build (push) Successful in 16s
Reviewed-on: #62
2026-09-28 21:05:42 +00:00
renovate-bot Bot da0f9e84c3 chore(deps): update ghcr.io/autobrr/netronome docker tag to v0.15.0 2026-09-28 21:05:42 +00:00
forust 3ecc12300a Merge pull request 'chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.28' (#61) from renovate/container-patch-updates into main
renovate-ci / validate-renovate (push) Successful in 13s
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / scan-deps (push) Canceled after 0s
ci / test-backend (push) Canceled after 0s
ci / test-frontend (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
Reviewed-on: #61
2026-09-28 21:05:33 +00:00
renovate-bot Bot 19029fa012 chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.28 2026-09-28 21:05:33 +00:00
forust 2a5d690e34 Merge pull request 'chore(deps): update netbirdio/dashboard docker tag to v2.94.0' (#63) from renovate/netbirdio-dashboard-2.x into main
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / scan-deps (push) Canceled after 0s
ci / test-backend (push) Canceled after 0s
ci / test-frontend (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 19s
Reviewed-on: #63
2026-09-28 21:05:25 +00:00
renovate-bot Bot b53ce36d89 chore(deps): update netbirdio/dashboard docker tag to v2.94.0 2026-09-28 21:05:25 +00:00
forust a8c4e9bcbe Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.117.0' (#64) from renovate/renovate-self-update into main
renovate-ci / validate-renovate (push) Successful in 16s
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 15s
ci / test-backend (push) Successful in 7s
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 2s
ci / build (push) Successful in 9s
Reviewed-on: #64
2026-09-28 21:02:09 +00:00
renovate-bot Bot 761f97ef8e chore(deps): update renovate/renovate docker tag to v44.117.0 2026-09-28 21:02:09 +00:00
forust 0c9743e241 Merge pull request 'chore(deps): update ghcr.io/immich-app/postgres docker tag to v16' (#65) from renovate/ghcr.io-immich-app-postgres-16.x into main
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 3s
ci / scan-deps (push) Successful in 14s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 27s
ci / build (push) Successful in 8s
Reviewed-on: #65
2026-09-28 21:01:58 +00:00
renovate-bot Bot 2b3c28a46b chore(deps): update ghcr.io/immich-app/postgres docker tag to v16 2026-09-28 21:01:58 +00:00
forust 2cb06debc5 fix(deploy): retry registry lookups with a timeout
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / scan-deps (push) Successful in 19s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 14s
ci / build (push) Successful in 7s
A single blink of the registry failed render_pinned for the whole file and redded the apply stage. registry_digest now retries 3 times under a 25s timeout with a warning per attempt; empty still means unresolvable and callers report it by name as before.
2026-09-28 22:52:54 +02:00
forust a6af69dca0 fix(k8s): Recreate singletons and trim requests for scheduler headroom
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-compose (push) Successful in 3s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 11s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 13s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 18s
ci / build (push) Successful in 1m30s
RollingUpdate with default maxSurge needs a spare pod the single node does not have (99% CPU requested), so multi-workload restarts end Pending and verify times out. Recreate on all replicas:1 Deployments (immich-server and bentopdf keep RollingUpdate at replicas 2). Also trims CPU/memory requests toward measured use (adguard, authentik, gitea, netbox, uptime-kuma, netbird-server) and gives traefik requests/limits so it is no longer BestEffort.
2026-09-28 22:36:46 +02:00
forust 76f39da90c fix(ci): skip heavy jobs on renovate branches, automerge digest and patch
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / scan-deps (push) Canceled after 0s
ci / test-backend (push) Canceled after 0s
ci / test-frontend (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Canceled after 0s
Renovate branches only carry version/digest bumps, so scan-deps, test-backend, test-frontend and build just burn runner time on the box that also serves prod. Static checks and validate still run. Digest and patch updates automerge (playwright, helm and major rules below still override to no-automerge). Also replaces deprecated helm --atomic with --wait --rollback-on-failure.
2026-09-28 22:19:57 +02:00
forust 86730ff0c4 Merge pull request 'chore(deps): update ghcr.io/c4illin/convertx docker tag to v0.19.0' (#55) from renovate/ghcr.io-c4illin-convertx-0.x into main
renovate-ci / validate-renovate (push) Successful in 6s
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 14s
ci / test-backend (push) Successful in 7s
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 3s
ci / build (push) Successful in 10s
Reviewed-on: #55
2026-09-28 19:57:17 +00:00
renovate-bot Bot af66d3e7fd chore(deps): update ghcr.io/c4illin/convertx docker tag to v0.19.0 2026-09-28 19:57:17 +00:00
forust deeaefa695 Merge pull request 'chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v3.2.0' (#60) from renovate/ghcr.io-henriquesebastiao-downtify-3.x into main
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / scan-deps (push) Canceled after 0s
ci / test-backend (push) Canceled after 0s
ci / test-frontend (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 18s
Reviewed-on: #60
2026-09-28 19:57:05 +00:00
renovate-bot Bot 6999dd2728 chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v3.2.0 2026-09-28 19:57:05 +00:00
forust 742d78944b Merge pull request 'chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.27' (#54) from renovate/container-patch-updates into main
ci / lint-compose (push) Canceled after 0s
ci / lint-actionlint (push) Canceled after 0s
ci / lint-shellcheck (push) Canceled after 0s
ci / lint-prettier (push) Canceled after 0s
ci / lint-ruff (push) Canceled after 0s
ci / lint-yaml (push) Canceled after 0s
ci / lint-dockerfiles (push) Canceled after 0s
ci / scan-deps (push) Canceled after 0s
ci / test-backend (push) Canceled after 0s
ci / test-frontend (push) Canceled after 0s
ci / validate (push) Canceled after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 23s
Reviewed-on: #54
2026-09-28 19:56:55 +00:00
renovate-bot Bot e451c97dfc chore(deps): update container patch updates
renovate-ci / validate-renovate (push) Skipped
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / scan-deps (push) Successful in 15s
ci / lint-actionlint (pull_request) Successful in 1s
ci / lint-compose (push) Successful in 3s
ci / test-backend (push) Successful in 7s
ci / test-frontend (push) Successful in 11s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 3s
ci / lint-shellcheck (pull_request) Successful in 2s
ci / lint-prettier (pull_request) Successful in 2s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-yaml (pull_request) Successful in 3s
ci / lint-dockerfiles (pull_request) Successful in 2s
ci / scan-deps (pull_request) Successful in 14s
ci / test-backend (pull_request) Successful in 7s
ci / test-frontend (pull_request) Successful in 11s
ci / validate (pull_request) Successful in 2s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 20s
2026-09-28 19:56:45 +00:00
forust 33c54ac830 Merge pull request 'chore(deps): update docker.io/valkey/valkey:9 docker digest to 418652c' (#59) from renovate/docker.io-valkey-valkey-9 into main
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-yaml (push) Successful in 2s
ci / lint-ruff (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 1s
ci / scan-deps (push) Successful in 19s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 10s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 14s
ci / build (push) Successful in 11s
Reviewed-on: #59
2026-09-28 19:56:27 +00:00
renovate-bot Bot b09d718310 chore(deps): update docker.io/valkey/valkey:9 docker digest to 418652c
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (pull_request) Successful in 6s
ci / lint-actionlint (pull_request) Successful in 2s
ci / lint-shellcheck (pull_request) Successful in 2s
ci / lint-prettier (pull_request) Successful in 3s
ci / lint-ruff (pull_request) Successful in 2s
ci / lint-yaml (pull_request) Successful in 2s
ci / lint-dockerfiles (pull_request) Successful in 3s
ci / scan-deps (pull_request) Successful in 55s
ci / test-backend (pull_request) Successful in 9s
ci / test-frontend (pull_request) Successful in 11s
ci / validate (pull_request) Successful in 3s
ci / build (pull_request) Skipped
ci / lint-compose (push) Successful in 3s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 14s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 11s
ci / validate (push) Successful in 1s
ci / build (push) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 53s
2026-09-28 16:20:21 +00:00
forust b4f76373bb fix(gitea): serve issue search from postgres instead of reindexing on boot
renovate-ci / validate-renovate (push) Successful in 5s
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 3s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 17s
ci / test-backend (push) Successful in 9s
ci / test-frontend (push) Successful in 13s
ci / validate (push) Successful in 2s
ci / build (push) Successful in 7s
cron.rebuild_issue_indexer runs at start, so every gitea pod restart reindexed the whole issue index and read ~7MB/s off the rotational disk for an hour.
2026-09-28 17:01:36 +02:00
forust 74adf38d63 feat(alerts): cover OOM kills, restart loops and evictions
The OOMKilled container behind the immich crash loop was invisible: PodCrashLooping only fires once kubelet has already given up and started the backoff.
2026-09-28 17:01:36 +02:00
forust f5b2f89f38 style(immich): match repo prettier quoting in compose file
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 9s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 16s
ci / test-backend (push) Successful in 9s
ci / test-frontend (push) Successful in 13s
ci / validate (push) Successful in 3s
ci / build (push) Successful in 9s
2026-09-28 17:00:41 +02:00
forust 8e63284240 feat(immich): add self-hosted photo backup with dedicated postgres
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 1s
ci / lint-shellcheck (push) Successful in 3s
ci / lint-prettier (push) Failing after 4s
ci / lint-ruff (push) Successful in 3s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / scan-deps (push) Successful in 20s
ci / test-backend (push) Successful in 8s
ci / validate (push) Successful in 3s
ci / build (push) Skipped
renovate-ci / validate-renovate (push) Successful in 11s
ci / test-frontend (push) Successful in 14s
Server x2, machine learning, valkey and VectorChord postgres on local storage, Traefik routes for external and internal access.
2026-09-28 16:42:05 +02:00
forust 1d81410cd8 fix(traefik): persist plugin storage on a PVC
ci / lint-compose (push) Successful in 4s
ci / lint-actionlint (push) Successful in 3s
ci / lint-shellcheck (push) Successful in 4s
ci / lint-prettier (push) Successful in 5s
ci / lint-ruff (push) Successful in 3s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 3s
ci / scan-deps (push) Successful in 18s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 11s
ci / validate (push) Successful in 4s
renovate-ci / validate-renovate (push) Successful in 1m28s
ci / build (push) Successful in 37s
Mount traefik-plugins PVC at /plugins-storage instead of the chart default emptyDir, so the crowdsec-bouncer download survives node reboots. Without this Traefik starts before the network is ready, the download from plugins.traefik.io times out, plugins get disabled and every route behind the middleware returns 404/503 until a manual restart.
2026-09-28 13:36:46 +02:00
forust 2f891a5d31 fix(postgres): give probes room on an I/O-bound single node
renovate-ci / validate-renovate (push) Canceled after 21s
ci / lint-compose (push) Successful in 8s
ci / lint-actionlint (push) Successful in 3s
ci / lint-shellcheck (push) Successful in 4s
ci / lint-prettier (push) Successful in 4s
ci / lint-ruff (push) Successful in 10s
ci / lint-yaml (push) Successful in 4s
ci / lint-dockerfiles (push) Successful in 4s
ci / scan-deps (push) Successful in 19s
ci / test-backend (push) Successful in 9s
ci / test-frontend (push) Successful in 21s
ci / validate (push) Successful in 5s
ci / build (push) Successful in 22s
pg_isready with the 1s default times out under I/O stall and kubelet kills a healthy postgres mid-recovery; each kill restarts a multi-minute fsync from zero and loops forever. readiness/liveness timeout 5s, liveness threshold 5, startup budget 15min.
2026-09-28 12:35:16 +02:00
forust cda0022d81 fix(renovate): reap finished job pods with ttlSecondsAfterFinished
ci / lint-compose (push) Successful in 5s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 2s
ci / lint-prettier (push) Successful in 4s
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 3s
ci / scan-deps (push) Successful in 17s
ci / test-backend (push) Successful in 7s
ci / test-frontend (push) Successful in 12s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 9m17s
ci / build (push) Failing after 28m20s
History limits never delete manual 'create job --from' runs, so Failed pods accumulated for a week. Keep a day for debugging, reap the rest.
2026-09-28 12:03:26 +02:00
forust dde6b1c743 fix(deploy): recover helm releases from pending-* and skip helm-owned rollbacks
An --atomic upgrade whose own rollback never finishes leaves the release in pending-*, blocking every future run until a human rolls back (loki rev 18/21). Recover automatically before and after each upgrade, and fail loud when recovery does not land on deployed. Also skip helm-managed workloads in rollback_workloads: rollout undo there would step back to the revision --atomic just escaped.
2026-09-28 12:03:26 +02:00
forust 7c4843c88c fix(loki): unblock gateway rollout on a single node
Chart default is required podAntiAffinity on hostname plus RollingUpdate 25%/25%, which is maxUnavailable=0 at replicas=1: the new pod stays Unschedulable while the old one lives, and the old one never leaves while the new one is not Ready. Null the affinity (an empty map deep-merges with the default and keeps the rule) and set maxSurge/maxUnavailable to 1.
2026-09-28 11:33:38 +02:00
forust f9e4623ade fix(k8s): size the remaining workloads against measured use
Finishes the sizing pass over every workload the deploy actually manages. Each
request is at or above the container's p95 over the last seven days, so nothing
is sized below what it is known to use, and each limit is between 1.6x and 5x
the observed max, which is the figure that decides whether a burst gets an
OOMKill.

Some of these go up, and that is the point. adguard was holding 975M against a
500Mi request and netbox 962M against 512Mi, so both sat permanently above
their own request and were standing eviction candidates on a node that has
about 300M of headroom. Raising a request costs scheduler room; leaving it low
costs the pod its place in the queue when the node gets tight.

Others come down. loki ran with a 2Gi limit on 224M, gitea 1.5Gi on 305M, the
authentik worker 1Gi on 305M, and a tail of single-purpose pods -- redis,
glance, the two homepages, cfddns, session-keeper, the netbird dashboard, the
loki gateway -- each reserved 4x to 16x more than they have ever touched.

prometheus gets the opposite treatment: 768Mi/2560Mi, above its p95, because it
compacts its TSDB in place and that is a burst worth budgeting for rather than
throttling.

Two of these limits are close enough to the observed max to be worth watching
rather than trusting: adguard at 1.5x, and its DNS cache grows monotonically, so
the ceiling is a date, not a margin. That was true before this change too; the
pod sizing does not fix it and the cache needs bounding.

CPU limits are untouched throughout. Leaving postgres alone as well: it sits in
an uncommitted file that belongs to other work in progress.

Verified: every request is at or above p95 and every limit above the observed
max across all 74 containers, and 16/16 local gates pass.
2026-09-28 10:38:19 +02:00
forust 2ad4fa1b82 chore(portainer): stop deploying a container manager nothing routes to
Portainer had been running for 111 days with a 512Mi request and a 2Gi limit
against 50M of measured use, on a node that is short of memory. It is a UI over
the Docker socket; nothing in the repo or the cluster depends on it.

The marker goes, not the manifests. `portainer/k8s/active` is what puts these
files in the deploy's manifest set, so without it the next push leaves the
namespace alone and the manifests stay on disk for a one-command return. This
also matters for the smoke stage: that host list is built from the active
directories, so `portainer.forust.xyz` leaves it and the new router check does
not go looking for a route to a service we just retired.

In the cluster the Deployment, the Service and both IngressRoutes are deleted.
The routes go first: leaving an IngressRoute behind a deleted Service keeps a
Traefik router pointing at nothing, which answers 502 while looking perfectly
healthy to the stage that just started checking for routers.

Deliberately kept, so this is reversible rather than destructive: the namespace,
the 2Gi `portainer-data-pvc` and both Certificates stay. Re-enabling is
`git checkout HEAD~1 -- portainer/k8s/active` plus an apply, and no Let's Encrypt
quota is spent reissuing the production certificate.

`glance` still links to `portainer.forust.xyz` and that tile will now be a 404.
Left alone on purpose. The Cloudflare record is manual and cfddns only ever
creates records, so `portainer.forust.xyz` keeps resolving until it is removed
in the dashboard, same as `dockmon.forust.xyz`.

Verified: no Traefik router matches portainer any more, all 19 hosts left in the
smoke list still have a router, and 16/16 local gates pass.
2026-09-28 10:22:51 +02:00
forust 2a4f215546 fix(k8s): bring the over-reserved memory limits down to measured use
Six pods reserved far more memory than they have ever touched. uptime-kuma held
a 3Gi limit against 469M of measured p95, metube 2Gi against 72M, convertx
1.5Gi against 85M, netbird-server 1Gi against 97M, searxng 700Mi against 134M
and bentopdf 700Mi against 4M. Every one of them is a ceiling the scheduler
counts against the node while the memory sits unused.

Requests move down with the limits but never below the measured p95, so none of
these becomes an eviction candidate as a side effect of being right-sized. The
limits keep between 2.2x and 11.6x over the observed max, which is the figure
that decides whether a pod gets OOM-killed during a burst.

Net effect across the six: requests -557M, limits -4.6Gi, all of it ceiling that
was never in use. This is the first change that actually gives memory back.

CPU limits are left exactly as they were. They were not part of the sizing pass,
they are not being hit on a node sitting at 5% CPU, and removing them is a
separate decision from moving memory.

Verified: each limit is above the container's own observed max and each request
is above its p95, and 16/16 local gates pass.
2026-09-28 10:20:24 +02:00
forust 16aaeb60c1 fix(k8s): set requests and limits on the pods that shipped with neither
Eighteen containers had no memory limit at all, so nothing on the node could
bound them. Three of the values files even claimed to set resources: Helm does
not complain about a key it does not recognise, so the block sat there looking
like a limit while the pod ran unbounded.

alloy is the one that mattered. The chart reads `alloy.resources`; the file had
`controller.resources`, so the DaemonSet that tails every pod log on the node
shipped with nothing at all. `kubeStateMetrics` is the same trap in a different
shape -- that is the condition key, the values live under `kube-state-metrics` --
and `configReloader` in the alloy chart sits at the top level rather than under
`alloy`. Each one is verified by rendering the chart and reading the resources
back off the containers, because a values key that is ignored looks exactly
like one that works.

reloader turned out to be set and still wrong: 64Mi request against a measured
p95 of 73M, so the pod ran permanently above its own request and stayed a
standing eviction candidate. That is the pod that restarts every other pod, so
it is the last one that should be evicted. Raised to 96Mi.

Requests are set at p95 throughout, grafana, playwright and alloy included.
Left at the values first proposed they would have sat below their own p95 and
queued for eviction ahead of everything smaller. CPU limits are deliberately
absent: the node is I/O bound at 5% CPU, and CFS throttling would turn disk
wait into runnable-throttled, which is the failure mode that took the node down.

The prometheus and alertmanager configReloader sidecars are left open: chart
86.2.3 does not template the key, so reaching those two containers needs a
postRenderer.

Verified: all four charts render with the resources landing on the intended
containers, and 16/16 local gates pass.
2026-09-28 10:14:52 +02:00
forust a5409edbf2 fix(deploy): fail the smoke stage when Traefik has no route for a host
ci / lint-compose (push) Successful in 5s
ci / lint-actionlint (push) Successful in 2s
ci / lint-shellcheck (push) Successful in 3s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 3s
ci / scan-deps (push) Successful in 15s
renovate-ci / validate-renovate (push) Successful in 1m15s
ci / test-frontend (push) Successful in 14s
ci / validate (push) Successful in 4s
ci / test-backend (push) Failing after 13m28s
ci / build (push) Skipped
The smoke stage treats any HTTP response as proof the service is serving,
which is right -- a 302 to a login or a 404 from a path the app does not
serve still means the chain is intact. But a 404 is not evidence of that on
its own: a router Traefik refused to build answers with exactly the same
404 and nothing behind it.

That is not hypothetical. The crowdsec bouncer is a plugin, and when Traefik
cannot fetch it at startup it disables the plugin without failing, then drops
every router whose chain referenced it. Sixteen routes answered 404 and the
stage printed `ok` for all sixteen, because a dropped router and an unserved
path are indistinguishable from outside.

The Kubernetes objects cannot tell us either: the IngressRoute is still
sitting there looking healthy, the router Traefik built from it is simply not
there. So ask Traefik. api.insecure is already on for the internal entrypoint
and the router list says which hosts it matches right now.

Every probed host has to appear in that list. HTTP routers only -- the TCP
ones match on a HostSNI wildcard and the UDP ones carry no rule at all, both
selected by entrypoint and port, so neither can answer the question. A router
mid-rollout is legitimately absent for a moment, so the list is re-read twice
over 20s; a plugin that failed to load stays absent and waiting cannot rescue
it. An unreadable router list fails the stage rather than skipping the check,
since a check that cannot run is not a passing check.

Verified against the live cluster: all 23 public routes have a router and the
stage passes. With gitea, grafana and uptime removed from that list the
probes still answer and the stage fails on exactly those three.
2026-09-28 09:13:24 +02:00
forust c70d2db3a1 ci: deploy the image the commit built, not whatever the tag points at
Every service tracked the mutable `:prod` tag, so a deploy applied whatever
that tag happened to name at the time rather than the commit it was
deploying. A rollback had no way to state what it was rolling back to, and
two deploys of one commit could land different images.

CI now publishes an immutable `sha-<commit12>` tag beside `:prod` on main,
and re-tags it for every image a push did not rebuild. That re-tag copies
the manifest list, so no layer moves. The deploy resolves the immutable tag
to a digest and pins the workload to it, and only falls back to the moving
tag when the immutable one cannot be resolved -- which it says out loud,
because that fallback is the deploy quietly ceasing to be reproducible from
its own commit.

The image list comes out of the tree with git grep rather than being written
out a second time, so adding a service no longer means keeping two lists in
step.

build also gains the three jobs it was skipping -- scan-deps, test-backend,
test-frontend -- so a change that breaks them cannot be tagged at all. The
two run blocks where a mid-loop failure was survivable now run under
set -euo pipefail: the build loop and the service detector both carried on
past an error and could report a green build having produced nothing.

The registry password moves from run: substitution into an env: block. A
quote, a backtick or a $(...) in the password is parsed as shell before the
command ever runs, and a login that failed that way looked exactly like a
build that failed.

The apply and verify timeouts stay at 45 and 30 minutes. The comments now
record the arithmetic that says so rather than leaving the numbers to be
raised on the next scare: three no-op helm upgrades run 3-5 minutes, one
broken release is a single 10 minute rollback because the loop aborts on
the first failure, and the apply loop itself is about a minute. That is
roughly 15 minutes of work against a 45 minute budget. verify is 32
workloads at 8 wide -- four waves of 300 seconds, 20 minutes -- which
leaves room for two serial rollbacks, and only becomes derivable at 45 once
rollback_workloads is parallelised.
2026-09-28 09:12:01 +02:00
forust 24dd82e801 fix(crowdsec): make the bouncer trust the mobile range independently
The parser-stage whitelist already covers 84.245.64.0/18, so an event
from the phone is dropped before it reaches a bucket and no decision is
ever created for it - confirmed against 72h of traefik access logs, where
the phone shows up as 84.245.120.147, inside that /18. But that left the
bouncer's own ClientTrustedIPs without the range, so the guarantee rested
on a single config. If the parser whitelist ever stops matching, a ban
would be created and then served against the phone, which is the one
thing that must not happen: the address belongs to a carrier, so it comes
back to us by rotation and a 4h ban is not survivable from the device.

ClientTrustedIPs bypasses the bouncer and the decision cache entirely, so
repeating the range there holds even if a decision exists for any reason.
All nine parser-stage ranges are now mirrored in the bouncer, and the
bouncer has no range the parser stage does not know about.

The file header now records that this middleware must be applied together
with a traefik restart. Applying it alone wedges the plugin: in stream
mode handleStreamTicker runs over package-level globals that no
reconfiguration stops, so every route referencing the middleware answers
404 with 'invalid middleware crowdsec-crowdsec-bouncer@kubernetescrd'
until the pod is replaced. Re-applying the prior config does not recover
it and the config is not the cause - NewChecker is a plain net.ParseCIDR
and cannot fail on a valid range. That cost 21 routes down before the
restart requirement was found; recovery is a pod replace, ~35s.

Verified live: middleware applied, traefik restarted, 17 of 20 hosts
serving (the three exceptions are unchanged and unrelated - searxng
returns its own 429, checkmk is down with 503, and one host is
local-only), zero invalid-middleware errors, and the LAPI still shows
/v1/decisions/stream polls at the 15s interval.
2026-09-27 18:52:27 +02:00
forust 11e92fdf4e fix(deploy): bound ssh hangs and retry the stage on transport loss
A connection that died silently used to hang until the job timeout, and
the stage was never re-run. One flaky TCP session cost a whole
45-minute apply, and the symptom - a job that stops mid-output with no
error - is what made the last few deploy failures expensive to read.

ServerAliveInterval/CountMax cap how long a dead peer goes unnoticed at
~60s, ConnectTimeout caps setup. Only exit 255 - ssh's own transport
failures - is retried, up to three attempts with a growing gap. A stage
that fails on its own merits exits with the remote's status, so a real
failure surfaces its own log immediately instead of being repeated three
times over 45 minutes. The stages are declarative applies, so re-running
one that had already committed is harmless.

The stage environment now goes through `env` as separate argv entries
rather than one interpolated string, so nothing in REPO, DEPLOY_SHA or
DEPLOY_SNAPSHOT_DIR is re-split by the remote shell.

Verified against a stubbed ssh: clean run attempts once, a single
transport failure recovers on attempt 2 and exits 0, three failures give
up preserving 255, and a stage failing with 1 or 7 attempts once and
passes the code through unchanged.

Also records why USERBOT_IMAGE stays on the prod tag: render_pinned
rewrites only plain `image:` lines, and this ref is what the panel
injects into the per-instance Deployments it creates, so those instances
track the tag rather than the panel's own resolved digest. The two
panel-created instances currently in the cluster are digest-pinned, so
the panel does accept one either way; the tag is the choice, not a
limitation.
2026-09-27 18:26:25 +02:00
forust b1f98fc148 Revert "fix(compose): stop pointing at the tag the build dropped" for dtek_notif
dtek-notif is being picked up again, so leave its compose alone. It is
also the one image not rebuilt since the build dropped :latest, so
:prod does not exist for it yet - unlike the other six, which resolve.
Keeping this file out of the change also keeps dtek_notif/* out of the
build's changed-service detector, so pushing does not build an image
nobody asked for.
2026-09-27 18:17:32 +02:00
forust 9b91b5847e fix(compose): stop pointing at the tag the build dropped
These five stacks still asked for :latest, but the build stopped pushing
it - on main it only pushes main and prod, on dev only dev. Every one of
these images therefore resolved only because the registry still had a
stale :latest from before that change, and the next time one of them was
actually built the reference would have dangled.

Named for dtek-notif: of the seven, only that one still resolved at
:prod, because it is the only image not rebuilt since the build dropped
:latest - and it is the only one of these five stacks the deploy does not
manage (no `active` marker, and its file is docker-compose.yaml, which
the COMPOSE_STACKS glob does not even match). Touching
dtek_notif/docker-compose.yaml matches dtek_notif/* in the build's
changed-service detector, so pushing this builds it and publishes
:prod for it too.

Verified the other six resolve at :prod in the registry.
2026-09-27 18:11:45 +02:00
forust 892790822d fix(crowdsec): stop the 403 loop that banned our own VPS and runner
Chasing why apply-k8s kept dying mid-run turned up a self-inflicted
ban loop. 585 of 586 LePresidente/http-generic-403-bf alerts in the LAPI
came from 193.181.211.79 - our own VPS - POSTing
/management.ManagementService/GetServerKey, i.e. the NetBird client's
own management call. netbird-server had never seen a single one of them,
so the 403 was not NetBird's: the bouncer was rejecting the request
before it got there. A banned peer keeps retrying, each retry is another
403, and the scenario turns five 403s in ten seconds into a 4h ban, so
the loop kept re-arming the ban it was serving. The hourly janitor step
that deleted those decisions hourly was masking all of it.

* netbird/k8s/ingress.yaml - drop the bouncer from the mesh API routes
  (gRPC-gateway management, signal, relay, /api, /oauth2). A ban there
  locks a peer out of the network it needs to reach anything else, and
  those endpoints authenticate by NetBird token, not by a login form.
  The dashboard keeps the bouncer; it is a real login surface.
  netbird-local was already exempt, so this makes prod match.

* crowdsec-middleware.yaml - CrowdsecMode stream instead of live. live
  blocked on GET /v1/decisions per request, so a burst saturated the
  LAPI and the plugin 403'd IPs that were never banned. v1.3.3 ignores
  UpdateMaxFailure in live, so fail-open is only reachable in stream;
  -1 now means an unreachable LAPI degrades to "no protection" rather
  than "every site 403". 15s poll instead of the 60s default, because
  the runner shares one public IP with the house.

  Also corrects the key name: HTTPTimeoutSeconds, not
  CrowdsecLapiTimeout, which never existed and was being silently
  dropped, leaving the 10s default. Back at 10, not the 2s f7cd75d
  guessed - a pull that times out leaves the ban cache frozen at its
  startup contents, so new bans would silently never apply.

* crowdsec-values.yaml - CIDR allowlisting moves to parsers/s02-enrich,
  where CrowdSec's docs put it: a parser whitelist drops the event before
  it reaches a bucket, so those addresses never become a decision at
  all. The old postoverflow LAN list was checked only after the ban
  existed, which is the window the deploy kept landing in. Added
  100.64.0.0/10, which the RFC 1918 blocks miss and where the
  workstation, the k0s node and the VPS actually live. The DDNS
  home-IP whitelist stays in postoverflows, because resolving a hostname
  is the expensive check the docs reserve that stage for.

  ClientTrustedIPs mirrors that list so the bouncer skips the LAPI
  round-trip entirely for those addresses.

* janitor-cronjob.yaml - drop step 5. The bouncer can no longer
  manufacture 403s, so the only remaining firings of that scenario are
  real scanners, and deleting their decisions hourly was undoing a
  working ban.

* Also lands the LAPI config.yaml.local (SQLite WAL, Central API off,
  bounded flush) that f7cd75d's comments referenced but never included:
  the LAPI was blocked in fsync on its rollback journal, and the CAPI
  resolver held a write transaction while timing out against a host
  this network cannot reach.

Verified in-cluster: no new 403-bf alerts in the 3.5min after applying,
the management endpoint answers 404 from the backend instead of 403 from
the bouncer in 0.17s, and the VPS client reports Management and Signal
connected with 2/2 relays.
2026-09-27 18:09:07 +02:00
378 changed files with 7152 additions and 29048 deletions

No files matched your search

+50
View File
@@ -0,0 +1,50 @@
# EDU ownership handoff
## Status
The EDU ownership handoff is complete. The homelab repository no longer owns
EDU workloads, images, routes, alerts, or deployment selection. The EDU
repository is the only deployment owner: [forust/edu-master](https://git.forust.xyz/forust/edu-master).
Homelab PRs #99 and #105 are merged. PR #105 removed the EDU subtree and its
build, deploy, rollback, verification, route-probe, and registry references.
It also added the serial image build matrix for the homelab services. This
handoff record is the only remaining EDU-specific file in homelab Git.
The dedicated workstation checkout is `/srv/edu-master`, at release
`4f2b2a0e37dc11ac2c75441a15076c178e219d37`. It contains `k8s/active`; root
`active` is absent. The old untracked `/srv/homelab/edu_master` checkout was
moved outside the homelab repository to
`/srv/edu-master-legacy-archive-20261007/edu_master`. Its private files remain
mode `0600` inside an archive directory with mode `0700`. The homelab deploy
checkout has no EDU marker or tracked EDU application/deployment files.
`AUTODEPLOY=false` remains in place for homelab deployment.
## Release evidence
EDU PR #4 merged after its review and CI checks. Main-push CI run 1652 passed
all validation and both image builds. Deploy run 1653 passed for the exact main
SHA above.
The workstation rollout completed for both Deployments. The deployment
verified `/health` and `/live` with HTTP 200, Redis AUTH, session TTL of 1058
seconds, a delivery backlog of zero, and all nine EDU vmalert rules with
matching expressions and healthy evaluation.
The images now run by digest:
- Session keeper: `sha256:998dea51aa3015fd9cabefb0f53b030157a650c3bef72e02fe84f17d5762613d`
- Webinar checker: `sha256:92f3c1fa2bb7f9b4680a9fc76a5b33dfbea8ef3dd9c6490ebc45876fd4c54461`
Redis StatefulSet was unchanged. PVC `redis-data-pvc` remains bound to PV
`pvc-a4f2a79a-363a-4c12-ae91-92cdfc2a0d2e` with capacity 1 GiB. The existing
runtime Secret and Fernet key were preserved during the handoff. Notification
delivery was verified before closeout, as confirmed by the operator. The
deployment did not record downtime.
The release rollback snapshot is
`/home/forust/.local/state/edu-master-deploy/20261007T180541Z-4f2b2a0e37dc11ac2c75441a15076c178e219d37`.
The handoff data snapshot remains at
`/home/forust/.local/state/edu-master-deploy/handoff-20261007T080838Z`.
Both snapshots are outside Git. Do not restore old Redis data unless recovery
requires it. Never delete or recreate the Redis PVC.
+1
View File
@@ -7,4 +7,5 @@ self-hosted-runner:
labels: labels:
- arch - arch
- homelab - homelab
- homelab-pr
- prod - prod
+3
View File
@@ -0,0 +1,3 @@
{
"postgres": ["authentik", "gitea", "immich", "n8n", "netbox", "netronome"]
}
+159
View File
@@ -0,0 +1,159 @@
# Homelab CI/CD
The native Gitea runners run on **vps**; production runs on **workstation**.
Main-branch checks and image builds use `homelab:host`. Pull request and
non-main checks use `homelab-pr:host` under a separate account without Docker
access. The `homelab-pr` runner is registered at User scope for `forust`, so
any repository under that account can schedule jobs that request this label.
Each runner accepts one job at a time; the build waits for every check to pass.
CI and deploy runs also show a summary with
the release SHA, image build or reuse results, deploy mode, selected services,
and image digests. Failed runs keep a summary of completed image builds, stage
results, apply results, and recorded Kubernetes recovery. The final deploy
summary is in the smoke job; earlier jobs show the state observed at that time.
Apply success is separate from health and recovery. Update the installed
workstation controller with `setup-workstation.sh` when no deploy is running.
No job images or Kubernetes credentials are needed on the VPS. Builds use one
pinned BuildKit helper container. CI and deploy are separate workflows.
## Runner installation
Install Docker Engine with Compose and Buildx, Git, Python 3.11+, Bash, curl,
GNU tar/xz, flock and systemd using the host's package manager. Keep the existing
Gitea runner 3.0.2 binary at `/usr/local/bin/gitea-runner`.
From this checkout on the VPS:
```sh
sudo bash .gitea/runner/setup-runner.sh
```
The installer reuses `/var/lib/gitea-runner/.runner` and the existing service.
For a new host, install the same runner binary and register as `gitea-runner`
using the registration token interactively, label `homelab:host`, and working
directory `/var/lib/gitea-runner`; then rerun the installer. Tokens never belong
in this repository or command-line examples.
Pinned tools live in the runner user's `~/.cache/homelab-ci`; CI repairs version
drift there. Installations are locked. Buildx uses only the `homelab-ci` builder,
pushes directly to the registry, and caps retained local cache at 1 GiB with a
2 GiB free-space target. This is not a hard limit on peak build disk usage.
Nothing runs `docker system prune`, removes unrelated images, or deletes volumes.
### Pull request runner
Install the unprivileged host runner on the VPS:
```sh
sudo bash .gitea/runner/setup-pr-runner.sh
```
Get a registration token from the user Actions runner settings. Run the
installer in a terminal. It asks for the token without echoing it, registers the
runner as `homelab-pr` with label `homelab-pr:host`, then enables the service.
The work directory is `/var/lib/gitea-pr-runner`. Confirm that Gitea lists the
runner as User scope before merging the workflow change. An unmatched label can
fall back to the default job image.
Renovate PR validation uses `pull_request_target`, which reads the workflow from
the base branch. It checks out the PR head only after runner selection and runs
that code on `homelab-pr`. Keep this workflow read-only and do not add secrets.
The PR runner has a separate home and tool cache. Do not add it to the `docker`
group or give it access to `/var/run/docker.sock`. It runs repository code from
pull requests, so keep its registration and permissions separate from the
trusted `homelab` runner. This separates users and host permissions, but both
runners still share the VPS kernel and network. Use a disposable VM if PRs from
untrusted external authors must be fully isolated.
## Workstation setup
As the existing SSH deploy user on workstation:
```sh
sudo loginctl enable-linger forust
bash .gitea/runner/setup-workstation.sh
```
The controller uses `/srv/homelab` as the persistent configuration tree and makes
a detached source worktree for each SHA. It never resets `/srv/homelab`, moves
local configuration, renames Compose projects, or changes volume names.
The installer records the current Kubernetes context and cluster UID in
`~/.config/homelab-deploy/environment`. Check these before installing.
Configure Gitea Actions Variables:
- `DEPLOY_HOST`, `DEPLOY_USER`, `DEPLOY_PORT`: the existing VPS-to-workstation SSH endpoint.
- `DEPLOY_KNOWN_HOSTS`: workstation's verified SSH host key entry for that endpoint.
- `AUTODEPLOY`: `false` initially; `true` enables deployment after successful main CI.
Keep `DEPLOY_SSH_KEY`, `REGISTRY_USERNAME` and `REGISTRY_PASSWORD` in Actions
Secrets. Legacy endpoint secrets remain accepted during migration. The Actions
token must have repository read and Actions read access for release downloads.
The deploy user's existing Docker registry authentication remains necessary.
## Releases and deployment
CI publishes `release-<full SHA>` as a Gitea artifact with all three owned image
digests and build input fingerprints. Unchanged images are reused only from a
successful main CI artifact, never from `:prod`. Expired artifacts cause CI to
rebuild images; they block deployment until CI is rerun.
Run deploy from main with `deploy_ref=main` or a checked SHA:
- `full`: required for the first baseline; reconcile all active components.
- `changed`: compare with the last fully successful production deploy.
- `plan`: validate configuration and show selection without changing production resources.
- `refresh_images=true`: explicitly refresh mutable third-party Compose tags.
The manual and automatic paths both require successful CI, a successful build
job and the exact SHA's release artifact. PRs cannot publish images or deploy.
Removed resources are reported and require explicit removal; no automatic prune.
Service dependencies are listed in `.gitea/deploy-dependencies.json`.
A workstation user systemd service holds the deploy lock across validation,
sequential apply, verification and smoke checks. SSH clients only submit/follow:
disconnecting or cancelling the Actions client does not kill production apply.
Retrying the same run ID does not start another apply. `ExecStopPost` recovers
interrupted runs before the unit finishes. Kubernetes rolls back to captured
revisions; configuration and persistent data are not reverted.
## Status and recovery
`--retry` repeats failed verification and smoke checks, never apply. Recovery
keeps a failed deploy out of the successful baseline, even after rollback.
On workstation (replace the numeric ID with Actions run ID and attempt):
```sh
python3 ~/.local/lib/homelab-deploy/controller.py status 123-1
python3 ~/.local/lib/homelab-deploy/controller.py recover 123-1 --retry
journalctl --user -u homelab-deploy@123-1
```
Runs live in `~/.local/state/homelab-deploy/runs`. Compose stores resolved configs
with restricted permissions; these may contain credentials and must never be
uploaded as CI artifacts. Stage logs print the exact manual recovery command
using `compose-before/<stack>.json`, the original project directory and project
name. Compose does not automatically roll back, and Nextcloud AIO's child
containers remain managed by AIO. Preserve its own backups for data recovery.
The controller retains twenty successful/planned runs and preserves failures.
Update the workstation dispatcher only when no deploy is running.
## Validation and migration rollback
```sh
python3 -m unittest discover -s tests -v
bash .gitea/tests/deploy-validation.sh
```
Test on a separate namespace before the initial production `full` run. Check a
failed rollout, interrupted SSH and repeated run ID, and verify that an isolated
service change does not upgrade unrelated Helm releases or Compose stacks.
To roll back the migration, disable autodeploy and finish or recover the remote
run first. Restore the runner config/unit from `.before-<timestamp>` backups,
reload systemd and restart the runner. Restore the prior workflows from Git.
Production data and persistent volumes stay where they were. Do not remove run
state or Compose recovery files until recovery is confirmed.
+11
View File
@@ -0,0 +1,11 @@
[worker.oci]
gc = true
reservedSpace = "256MB"
maxUsedSpace = "1GB"
minFreeSpace = "2GB"
[[worker.oci.gcpolicy]]
reservedSpace = "256MB"
maxUsedSpace = "1GB"
minFreeSpace = "2GB"
all = true
+10
View File
@@ -0,0 +1,10 @@
runner:
file: /var/lib/gitea-runner/.runner
capacity: 1
timeout: 5h
labels:
- homelab:host
cache:
enabled: false
container:
docker_host: unix:///var/run/docker.sock
+18
View File
@@ -0,0 +1,18 @@
[Unit]
Description=Gitea Actions runner
After=network-online.target docker.service
Wants=network-online.target
[Service]
User=gitea-runner
Group=gitea-runner
SupplementaryGroups=docker
WorkingDirectory=/var/lib/gitea-runner
Environment=PATH=/var/lib/gitea-runner/.cache/homelab-ci/bin:/usr/local/bin:/usr/bin:/bin
ExecStart=/usr/local/bin/gitea-runner daemon --config /etc/gitea-runner/config.yaml
Restart=on-failure
RestartSec=5
UMask=0077
[Install]
WantedBy=multi-user.target
+12
View File
@@ -0,0 +1,12 @@
[Unit]
Description=Homelab deploy %i
[Service]
Type=exec
EnvironmentFile=%h/.config/homelab-deploy/environment
ExecStart=/usr/bin/python3 %h/.local/lib/homelab-deploy/controller.py execute %i
ExecStopPost=/usr/bin/python3 %h/.local/lib/homelab-deploy/controller.py recover %i
RuntimeMaxSec=5h
TimeoutStopSec=135min
KillMode=control-group
UMask=0077
+8
View File
@@ -0,0 +1,8 @@
runner:
file: /var/lib/gitea-pr-runner/.runner
capacity: 1
timeout: 5h
labels:
- homelab-pr:host
cache:
enabled: false
+27
View File
@@ -0,0 +1,27 @@
[Unit]
Description=Gitea Actions untrusted pull request runner
After=network-online.target
Wants=network-online.target
[Service]
User=gitea-pr-runner
Group=gitea-pr-runner
WorkingDirectory=/var/lib/gitea-pr-runner
Environment=HOME=/var/lib/gitea-pr-runner
Environment=PATH=/var/lib/gitea-pr-runner/.cache/homelab-ci/bin:/usr/local/bin:/usr/bin:/bin
ExecStart=/usr/local/bin/gitea-runner daemon --config /etc/gitea-pr-runner/config.yaml
Restart=on-failure
RestartSec=5
NoNewPrivileges=yes
PrivateTmp=yes
ProtectSystem=full
ProtectHome=yes
ProtectKernelTunables=yes
ProtectKernelModules=yes
ProtectControlGroups=yes
RestrictSUIDSGID=yes
LockPersonality=yes
UMask=0077
[Install]
WantedBy=multi-user.target
+56
View File
@@ -0,0 +1,56 @@
#!/usr/bin/env bash
# Install a native runner for untrusted PR jobs without Docker access.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
[ "$(id -u)" -eq 0 ] || { echo 'Run with sudo on the runner host' >&2; exit 1; }
for tool in cp cut date getent id install runuser systemctl useradd; do
command -v "$tool" >/dev/null || { echo "Install missing prerequisite: $tool" >&2; exit 1; }
done
command -v /usr/local/bin/gitea-runner >/dev/null || {
echo 'Install gitea-runner 3.0.2 at /usr/local/bin/gitea-runner first' >&2
exit 1
}
id gitea-pr-runner >/dev/null 2>&1 || \
useradd --system --create-home --home-dir /var/lib/gitea-pr-runner --shell /usr/bin/bash gitea-pr-runner
runner_home="$(getent passwd gitea-pr-runner | cut -d: -f6)"
[ "$runner_home" = /var/lib/gitea-pr-runner ] || {
echo 'Unexpected PR runner home; inspect the existing service first' >&2
exit 1
}
case " $(id -nG gitea-pr-runner) " in
*' docker '*)
echo 'The PR runner account must not belong to the docker group' >&2
exit 1
;;
esac
install -d -m 0755 /etc/gitea-pr-runner
stamp="$(date -u +%Y%m%dT%H%M%SZ)"
for existing in /etc/gitea-pr-runner/config.yaml /etc/systemd/system/gitea-pr-runner.service; do
[ ! -f "$existing" ] || cp -p "$existing" "$existing.before-$stamp"
done
install -m 0644 "$here/pr-config.yaml" /etc/gitea-pr-runner/config.yaml
install -m 0644 "$here/pr-runner.service" /etc/systemd/system/gitea-pr-runner.service
if [ ! -f /var/lib/gitea-pr-runner/.runner ]; then
read -r -s -p 'Enter the Gitea repository runner registration token: ' runner_token
printf '\n'
[ -n "$runner_token" ] || { echo 'Runner token is required' >&2; exit 1; }
export GITEA_RUNNER_REGISTRATION_TOKEN="$runner_token"
unset runner_token
runuser --preserve-environment -u gitea-pr-runner -- \
/usr/local/bin/gitea-runner register \
--config /etc/gitea-pr-runner/config.yaml \
--instance https://gitea.forust.xyz \
--name homelab-pr \
--labels homelab-pr:host \
--no-interactive
unset GITEA_RUNNER_REGISTRATION_TOKEN
fi
chmod 0600 /var/lib/gitea-pr-runner/.runner
systemctl daemon-reload
systemctl enable --now gitea-pr-runner.service
systemctl restart gitea-pr-runner.service
echo "PR runner ready. Configuration backups: *.before-$stamp"
+48
View File
@@ -0,0 +1,48 @@
#!/usr/bin/env bash
# Native host runner, with pinned user-space tools and no extra CI images.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
[ "$(id -u)" -eq 0 ] || { echo 'Run with sudo on the runner host' >&2; exit 1; }
for tool in docker curl python3 git tar xz flock runuser systemctl; do
command -v "$tool" >/dev/null || { echo "Install missing prerequisite: $tool" >&2; exit 1; }
done
docker info >/dev/null
docker compose version >/dev/null
docker buildx version >/dev/null
id gitea-runner >/dev/null 2>&1 || useradd --system --create-home --home-dir /var/lib/gitea-runner --shell /usr/bin/bash gitea-runner
# Reuse the established service account and runner registration.
runner_home="$(getent passwd gitea-runner | cut -d: -f6)"
[ "$runner_home" = /var/lib/gitea-runner ] || { echo 'Unexpected runner home; inspect the existing service first' >&2; exit 1; }
runuser -u gitea-runner -- docker info >/dev/null || { echo "The runner user needs access to Docker before setup" >&2; exit 1; }
command -v gitea-runner >/dev/null || { echo 'Install gitea-runner 3.0.2 at /usr/local/bin/gitea-runner first' >&2; exit 1; }
mkdir -p /etc/gitea-runner
stamp="$(date -u +%Y%m%dT%H%M%SZ)"
for existing in /etc/gitea-runner/config.yaml /etc/systemd/system/gitea-runner.service; do
[ ! -f "$existing" ] || cp -p "$existing" "$existing.before-$stamp"
done
scratch="$(mktemp -d)"
trap 'rm -rf "$scratch"' EXIT
chmod 755 "$scratch"
install -m 0644 "$here/../workflows/install-ci-tools.sh" "$here/../workflows/tool-versions.env" "$scratch/"
runuser -u gitea-runner -- bash "$scratch/install-ci-tools.sh"
install -m 0644 "$here/config.yaml" /etc/gitea-runner/config.yaml
python3 - <<'PYLABELS'
import json
from pathlib import Path
registration = Path('/var/lib/gitea-runner/.runner')
if registration.exists():
labels = json.loads(registration.read_text()).get('labels', [])
labels = [label for label in labels if isinstance(label, str) and label.split(':')[0] != 'homelab']
labels.append('homelab:host')
config = Path('/etc/gitea-runner/config.yaml')
config.write_text(config.read_text().replace(' - homelab:host', '\n'.join(' - ' + json.dumps(label) for label in labels)))
PYLABELS
install -m 0644 "$here/gitea-runner.service" /etc/systemd/system/gitea-runner.service
if [ ! -f /var/lib/gitea-runner/.runner ]; then
echo 'Register once as gitea-runner with homelab:host before starting the service.'
exit 0
fi
systemctl daemon-reload
systemctl enable --now gitea-runner.service
systemctl restart gitea-runner.service
echo "Runner ready. Configuration backups: *.before-$stamp"
+32
View File
@@ -0,0 +1,32 @@
#!/usr/bin/env bash
# Run as the existing deploy user on workstation. Never resets the working tree.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
repo="${HOMELAB_REPO:-/srv/homelab}"
for tool in python3 git kubectl helm docker flock timeout; do
command -v "$tool" >/dev/null || { echo "Install missing dependency: $tool" >&2; exit 1; }
done
[ -d "$repo/.git" ] || { echo "Missing deploy checkout: $repo" >&2; exit 1; }
[[ "$repo" =~ ^/[A-Za-z0-9_./-]+$ ]] || { echo 'Deploy path must be absolute and contain no whitespace' >&2; exit 1; }
if [ "$(loginctl show-user "$USER" -p Linger --value)" != yes ]; then
echo "Run once: sudo loginctl enable-linger $USER" >&2
exit 1
fi
config="${XDG_CONFIG_HOME:-$HOME/.config}/homelab-deploy"
mkdir -p "$config" "$HOME/.local/lib/homelab-deploy" "$HOME/.config/systemd/user"
chmod 700 "$config"
if [ ! -f "$config/environment" ]; then
context="$(kubectl config current-context)"
cluster_uid="$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')"
printf 'HOMELAB_REPO=%s\nKUBE_CONTEXT=%s\nEXPECTED_CLUSTER_UID=%s\n' "$repo" "$context" "$cluster_uid" >"$config/environment"
chmod 600 "$config/environment"
fi
# Do not replace a dispatcher while an existing deploy uses it.
if systemctl --user list-units 'homelab-deploy@*' --state=running --no-legend | grep -q .; then
echo 'An existing deploy is running; wait before updating the controller' >&2
exit 1
fi
install -m 0755 "$here/../workflows/deploy-controller.py" "$HOME/.local/lib/homelab-deploy/controller.py"
install -m 0644 "$here/homelab-deploy@.service" "$HOME/.config/systemd/user/homelab-deploy@.service"
systemctl --user daemon-reload
echo 'Controller ready. Run a checked main SHA in full mode for the initial baseline.'
+94
View File
@@ -0,0 +1,94 @@
#!/usr/bin/env bash
# Local regressions only: kubectl is mocked and Docker is used for config parsing.
set -euo pipefail
repo="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
scratch="$(mktemp -d)"
trap 'rm -rf "$scratch"' EXIT
mkdir -p "$scratch/repo/app" "$scratch/repo/postgres" "$scratch/repo/netbird" "$scratch/repo/renovate"
git -C "$scratch/repo" init -q
for file in app/compose.yaml postgres/shared-compose.yaml netbird/client.compose.yaml renovate/renovate-compose.yaml; do
touch "$scratch/repo/$file"
done
git -C "$scratch/repo" add .
# shellcheck source=../workflows/compose-lint.sh
source "$repo/.gitea/workflows/compose-lint.sh"
actual="$(cd "$scratch/repo" && compose_files)"
expected=$'app/compose.yaml\nnetbird/client.compose.yaml\npostgres/shared-compose.yaml\nrenovate/renovate-compose.yaml'
[ "$actual" = "$expected" ] || { echo 'Compose discovery missed a file' >&2; exit 1; }
cat >"$scratch/compose.yaml" <<'YAML'
services:
example:
image: busybox:1.37.0
environment:
REQUIRED: ${HOMELAB_TEST_REQUIRED:?required for this regression}
YAML
unset HOMELAB_TEST_REQUIRED
if validate_compose_file "$scratch/compose.yaml" >"$scratch/config.log" 2>&1; then
echo 'Full Compose validation accepted a missing variable' >&2
exit 1
fi
grep -q 'required for this regression' "$scratch/config.log"
HOMELAB_TEST_REQUIRED=present validate_compose_file "$scratch/compose.yaml"
cat >"$scratch/resources.json" <<'JSON'
{"kind":"List","items":[
{"kind":"Deployment","metadata":{"namespace":"app"},"spec":{"template":{"spec":{
"containers":[{"envFrom":[{"secretRef":{"name":"credentials"}},{"secretRef":{"name":"optional","optional":true}}],"env":[{"valueFrom":{"secretKeyRef":{"name":"credentials","key":"password"}}}]}],
"initContainers":[{"envFrom":[{"secretRef":{"name":"init"}}]}],
"imagePullSecrets":[{"name":"registry"}],
"volumes":[{"secret":{"secretName":"mounted"}},{"projected":{"sources":[{"secret":{"name":"projected"}},{"secret":{"name":"optional-projected","optional":true}}]}}]
}}}},
{"kind":"CronJob","metadata":{},"spec":{"jobTemplate":{"spec":{"template":{"spec":{"containers":[{"envFrom":[{"secretRef":{"name":"cron"}}]}]}}}}}},
{"kind":"IngressRoute","metadata":{"namespace":"app"},"spec":{"tls":{"secretName":"controller-issued-tls"}}}
]}
JSON
actual="$(jq -r -f "$repo/.gitea/workflows/secret-references.jq" "$scratch/resources.json" | sort)"
expected=$'app credentials\napp init\napp mounted\napp projected\napp registry\ndefault cron'
[ "$actual" = "$expected" ] || { echo "Unexpected Secret references: $actual" >&2; exit 1; }
REPO="$repo"
# shellcheck source=../workflows/deploy-lib.sh
source "$repo/.gitea/workflows/deploy-lib.sh"
K8S_MANIFESTS=("$scratch/resources.json")
KUSTOMIZE_APPS=()
# No live cluster access. Reject credentials in app even if they exist elsewhere.
kubectl() {
case "$1" in
create) cat "$scratch/resources.json" ;;
get)
if [ "$3" = credentials ] && [ "$5" = app ]; then
return 1
fi
return 0
;;
*) echo "Unexpected kubectl invocation: $*" >&2; return 1 ;;
esac
}
if check_referenced_secrets >"$scratch/secrets.log"; then
echo 'Namespace-scoped Secret check accepted a missing Secret' >&2
exit 1
fi
grep -q 'MISSING OR UNREADABLE: app/credentials' "$scratch/secrets.log"
# API/rendering errors must not produce an empty reference list and pass.
kubectl() { return 1; }
if ! skip_uninstalled_vmagent_crd "$REPO/prometheus-stack/k8s/vmagent.yaml"; then
echo 'VMAgent preflight did not skip an uninstalled CRD' >&2
exit 1
fi
kubectl() { return 0; }
if skip_uninstalled_vmagent_crd "$REPO/prometheus-stack/k8s/vmagent.yaml"; then
echo 'VMAgent preflight skipped an installed CRD' >&2
exit 1
fi
if skip_uninstalled_vmagent_crd "$REPO/prometheus-stack/k8s/victoria.yaml"; then
echo 'VMAgent preflight skipped an unrelated manifest' >&2
exit 1
fi
kubectl() { return 1; }
if check_referenced_secrets >"$scratch/secrets.log"; then
echo 'Secret check accepted a failed manifest render' >&2
exit 1
fi
printf '%s\n' 'Deploy validation regressions passed.'
+360 -538
View File
File diff suppressed because it is too large. Load diff
+1 -2
View File
@@ -21,8 +21,7 @@
# All committed Compose files, including the ones deploy never starts. # All committed Compose files, including the ones deploy never starts.
compose_files() { compose_files() {
git ls-files \ git ls-files \
'*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \ '*compose.yaml' '*compose.yml'
'*/docker-compose.yaml' '*/docker-compose.yml'
} }
# Prints the flags that turn `docker compose config` into the general check. # Prints the flags that turn `docker compose config` into the general check.
+103
View File
@@ -0,0 +1,103 @@
#!/usr/bin/env python3
"""Resolve Compose images without changing project names or local bind paths."""
import json
import os
import re
import subprocess
import sys
from pathlib import Path
def output(*args, **kwargs):
return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603
def resolve(reference):
if '@sha256:' in reference:
return reference
descriptor = json.loads(
output('docker', 'buildx', 'imagetools', 'inspect', reference, '--format', '{{json .Manifest}}')
)
digest = descriptor['digest']
if not re.fullmatch(r'sha256:[0-9a-f]{64}', digest):
raise ValueError(f'Invalid registry digest for {reference}')
# Strip tag only from the final path segment (registry ports are preserved).
repository = reference.rsplit('/', 1)
repository[-1] = repository[-1].split(':')[0]
return '/'.join(repository) + '@' + digest
def prepare(source_file):
config_repo = Path(os.environ['CONFIG_REPO'])
source_repo = Path(os.environ['REPO'])
directory = Path(os.environ['RUN_DIR'])
relative = source_file.relative_to(source_repo)
project_dir = config_repo / relative.parent
base = ['docker', 'compose', '--project-directory', str(project_dir), '-f', str(source_file)]
config = json.loads(output(*base, 'config', '--format', 'json', cwd=config_repo))
project = config['name']
previous_file = directory / 'previous.json'
previous = json.loads(previous_file.read_text()) if previous_file.exists() else {}
images_file = directory / 'compose-images.json'
locks = json.loads(images_file.read_text()) if images_file.exists() else previous.get('compose-images', {})
release = json.loads((directory / 'release.json').read_text())
before = json.loads(json.dumps(config))
for service, settings in config['services'].items():
reference = settings.get('image')
nextcloud_aio_master = project == 'nextcloud' and service == 'nextcloud-aio-mastercontainer'
if not reference or settings.get('build'):
raise ValueError(f'{project}/{service}: Compose deploy requires a published image')
image_repo = reference.split('@')[0].rsplit('/', 1)
image_repo[-1] = image_repo[-1].split(':')[0]
image_repo = '/'.join(image_repo)
# Nextcloud AIO validates the mastercontainer image reference and rejects
# a digest. Keep its configured tag so AIO can start and manage its stack.
if nextcloud_aio_master:
pinned = reference
elif image_repo in release['images']:
pinned = image_repo + '@' + release['images'][image_repo]
elif os.environ.get('REFRESH_IMAGES') != 'true' and reference in locks:
pinned = locks[reference]
else:
pinned = resolve(reference)
settings['image'] = pinned
locks[reference] = pinned
# Capture what is running, not the current value of its mutable tag.
ids = output(
'docker',
'ps',
'-aq',
'--filter',
f'label=com.docker.compose.project={project}',
'--filter',
f'label=com.docker.compose.service={service}',
).splitlines()
actual = set()
for container in ids:
image_id = output('docker', 'inspect', container, '--format', '{{.Image}}')
digests = json.loads(output('docker', 'image', 'inspect', image_id, '--format', '{{json .RepoDigests}}'))
actual.add(next((d for d in digests or [] if d.split('@')[0] == image_repo), image_id))
if len(actual) > 1:
raise ValueError(f'{project}/{service}: mixed running images, cannot capture one recovery config')
# AIO also rejects a digest in its recovery config. Preserve its tag in
# both deploy and recovery files.
if nextcloud_aio_master:
before['services'][service]['image'] = reference
else:
before['services'][service]['image'] = next(iter(actual)) if actual else reference
for name, data in (('compose', config), ('compose-before', before)):
folder = directory / name
folder.mkdir(mode=0o700, exist_ok=True)
destination = folder / f'{relative.parent.name}.json'
destination.write_text(json.dumps(data, indent=2) + '\n')
destination.chmod(0o600)
images_file.write_text(json.dumps(locks, indent=2) + '\n')
print(f'Compose {project}: images pinned; local paths preserved')
print(
f'Recovery: docker compose --project-directory {project_dir} -p {project} -f {directory}/compose-before/{relative.parent.name}.json up -d --pull never'
)
if __name__ == '__main__':
prepare(Path(sys.argv[1]))
+423
View File
@@ -0,0 +1,423 @@
#!/usr/bin/env python3
"""Durable workstation deployment controller. Install with setup-workstation.sh."""
import argparse
import contextlib
import fcntl
import importlib.util
import json
import math
import os
import re
import shutil
import subprocess
import sys
import time
from pathlib import Path
STATE = Path(os.environ.get('HOMELAB_STATE', Path.home() / '.local/state/homelab-deploy'))
CONFIG_REPO = Path(os.environ.get('HOMELAB_REPO', '/srv/homelab'))
RUN_ID = re.compile(r'[0-9]+-[0-9]+')
def command(*args, **kwargs):
return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603, S607
def atomic_json(path, data):
temporary = path.with_suffix('.tmp')
temporary.write_text(json.dumps(data, indent=2) + '\n')
temporary.chmod(0o600)
temporary.replace(path)
@contextlib.contextmanager
def lock(name):
STATE.mkdir(mode=0o700, parents=True, exist_ok=True)
with (STATE / name).open('a') as stream:
fcntl.flock(stream, fcntl.LOCK_EX)
yield
def load_module(name, path):
spec = importlib.util.spec_from_file_location(name, path)
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def run_directory(run_id):
if not RUN_ID.fullmatch(run_id):
raise ValueError('Run ID must be numeric workflow-id and attempt')
return STATE / 'runs' / run_id
def start(run_id):
payload = sys.stdin.buffer.read(256 * 1024 + 1)
if len(payload) > 256 * 1024:
raise ValueError('Deploy request exceeds 256 KiB')
request = json.loads(payload)
sha = request['release']['sha']
if not re.fullmatch(r'[0-9a-f]{40}', sha) or request['mode'] not in ('changed', 'full', 'plan'):
raise ValueError('Invalid deploy SHA or mode')
if not isinstance(request['refresh_images'], bool):
raise ValueError('refresh_images must be boolean')
directory = run_directory(run_id)
with lock('prepare.lock'):
if (directory / 'request.json').exists():
if json.loads((directory / 'request.json').read_text()) != request:
raise ValueError('Run ID already belongs to a different request')
else:
directory.mkdir(mode=0o700, parents=True, exist_ok=True)
command('git', '-C', str(CONFIG_REPO), 'fetch', '--quiet', 'origin', 'main')
command('git', '-C', str(CONFIG_REPO), 'merge-base', '--is-ancestor', sha, 'origin/main')
if not (directory / 'source').exists():
command('git', '-C', str(CONFIG_REPO), 'worktree', 'add', '--detach', str(directory / 'source'), sha)
if command('git', '-C', str(directory / 'source'), 'rev-parse', 'HEAD') != sha:
raise ValueError('Prepared source does not match deploy SHA')
release_module = load_module('release', directory / 'source/.gitea/workflows/release.py')
release_module.validate_release(request['release'], sha)
atomic_json(directory / 'release.json', request['release'])
atomic_json(directory / 'request.json', request)
if not (directory / 'status.json').exists():
atomic_json(directory / 'status.json', {'state': 'queued', 'stages': {}})
# Starting an existing active or finished ID is idempotent; never re-apply it.
if json.loads((directory / 'status.json').read_text())['state'] == 'queued':
command('systemctl', '--user', 'start', '--no-block', f'homelab-deploy@{run_id}.service')
print(f'Accepted deploy {run_id} ({sha})')
def environment(directory):
request = json.loads((directory / 'request.json').read_text())
return {
**os.environ,
'REPO': str(directory / 'source'),
'CONFIG_REPO': str(CONFIG_REPO),
'RUN_DIR': str(directory),
'DEPLOY_SHA': request['release']['sha'],
'RELEASE_FILE': str(directory / 'release.json'),
'DEPLOY_PLAN': str(directory / 'plan.json'),
'DEPLOY_SNAPSHOT_DIR': str(directory / 'snapshot'),
'REFRESH_IMAGES': str(request['refresh_images']).lower(),
'ROLLOUT_PARALLELISM': '4',
}
def stage(directory, name, budget):
status = json.loads((directory / 'status.json').read_text())
if name in status['stages'] and status['stages'][name].get('result') in ('success', 'failure'):
return status['stages'][name]['result'] == 'success'
started = time.time()
status['stages'][name] = {'result': 'running', 'started': started}
atomic_json(directory / 'status.json', status)
script = directory / 'source/.gitea/workflows/deploy-stage.sh'
with (directory / f'{name}.log').open('a') as log:
# timeout kills the whole stage process group, including children, before recovery.
result = subprocess.run( # noqa: S603, S607
[
shutil.which('timeout') or '/usr/bin/timeout',
'--signal=TERM',
'--kill-after=30s',
str(budget),
'bash',
str(script),
name,
],
env=environment(directory),
stdout=log,
stderr=subprocess.STDOUT,
check=False,
).returncode
status = json.loads((directory / 'status.json').read_text())
status['stages'][name].update(
result='success' if result == 0 else 'failure', exit_code=result, seconds=round(time.time() - started)
)
atomic_json(directory / 'status.json', status)
return result == 0
def make_plan(directory):
source = directory / 'source'
planner = load_module('deploy_plan', source / '.gitea/workflows/deploy-plan.py')
request = json.loads((directory / 'request.json').read_text())
previous = json.loads((STATE / 'last-success.json').read_text()) if (STATE / 'last-success.json').exists() else None
# Helm 4 lists every release status by default and removed the --all flag.
helm = json.loads(command('helm', 'list', '-A', '-o', 'json'))
plan = planner.make_plan(source, CONFIG_REPO, request['release'], previous, request['mode'], helm)
if request['refresh_images']:
plan['selected']['compose'] = plan['active']['compose']
atomic_json(directory / 'plan.json', plan)
if previous:
atomic_json(directory / 'previous.json', previous)
# Local config is deliberately separate from the immutable Git source.
return plan
def finish_success(directory, plan):
# Repeating finalization after a crash is safe while holding deploy.lock.
plan['run_id'] = directory.name
path = directory / 'compose-images.json'
previous = directory / 'previous.json'
plan['compose-images'] = (
json.loads(path.read_text())
if path.exists()
else json.loads(previous.read_text()).get('compose-images', {})
if previous.exists()
else {}
)
atomic_json(STATE / 'last-success.json', plan)
status = json.loads((directory / 'status.json').read_text())
status['state'] = 'success'
atomic_json(directory / 'status.json', status)
try:
retain_completed(directory)
except (OSError, subprocess.CalledProcessError) as error:
print(f'Retention deferred: {error}', flush=True)
def recover(directory, retry=False):
status = json.loads((directory / 'status.json').read_text())
if status['state'] in ('success', 'planned'):
return
completed = ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke')
if all(status['stages'].get(name, {}).get('result') == 'success' for name in completed):
finish_success(directory, json.loads((directory / 'plan.json').read_text()))
return
if retry:
for name in ('verify-k8s', 'smoke'):
if status['stages'].get(name, {}).get('result') == 'failure':
del status['stages'][name]
atomic_json(directory / 'status.json', status)
snapshot = directory / 'snapshot/current'
if snapshot.exists():
stage(directory, 'verify-k8s', 7200)
stage(directory, 'smoke', 600)
status = json.loads((directory / 'status.json').read_text())
status['state'] = 'failure'
atomic_json(directory / 'status.json', status)
def execute(run_id):
directory = run_directory(run_id)
with lock('deploy.lock'):
status = json.loads((directory / 'status.json').read_text())
if status['state'] != 'queued':
return
# A crashed predecessor must be recovered before another apply begins.
for other in (STATE / 'runs').iterdir():
if (
other != directory
and (other / 'status.json').exists()
and json.loads((other / 'status.json').read_text())['state'] == 'running'
):
raise ValueError(f'Interrupted deploy {other.name}; run recover first')
status['state'] = 'running'
atomic_json(directory / 'status.json', status)
phase = 'plan'
try:
plan = make_plan(directory)
print(
json.dumps({'selected': plan['selected'], 'helm': plan['helm'], 'manual_removals': plan['removed']}),
flush=True,
)
phase = 'doctor'
if not stage(directory, 'doctor', 600):
raise RuntimeError('Preflight failed')
phase = 'validate'
if not stage(directory, 'validate', 1200):
raise RuntimeError('Validation failed')
if json.loads((directory / 'request.json').read_text())['mode'] == 'plan':
status = json.loads((directory / 'status.json').read_text())
status['state'] = 'planned'
atomic_json(directory / 'status.json', status)
return
# Budget includes both rollout checks and rollback waves, plus API overhead.
phase = 'Recovery budget'
count = int(
command(
'bash',
str(directory / 'source/.gitea/workflows/deploy-stage.sh'),
'workload-count',
env=environment(directory),
)
)
verify_budget = max(600, 2 * math.ceil(count / 4) * 300 + 120)
if verify_budget > 7200:
raise ValueError('More than two hours of recovery required; split this deploy')
phase = 'apply-k8s'
k8s_ok = stage(directory, 'apply-k8s', 2700)
phase = 'apply-compose'
compose_ok = stage(directory, 'apply-compose', 1800) if k8s_ok else False
phase = 'verify-k8s'
verify_ok = stage(directory, 'verify-k8s', verify_budget)
phase = 'smoke'
smoke_ok = stage(directory, 'smoke', 600)
if not all((k8s_ok, compose_ok, verify_ok, smoke_ok)):
raise RuntimeError('Deploy failed; inspect stage logs and recovery report')
phase = 'Save the successful baseline'
finish_success(directory, plan)
except Exception as error:
status = json.loads((directory / 'status.json').read_text())
status['failure_stage'] = next(
(name for name, result in status['stages'].items() if result.get('result') == 'failure'), phase
)
atomic_json(directory / 'status.json', status)
with (directory / 'controller.log').open('a') as stream:
stream.write(f'{error}\n')
recover(directory)
raise
def retain_completed(current):
finished = []
for directory in (STATE / 'runs').iterdir():
status_file = directory / 'status.json'
if status_file.exists() and json.loads(status_file.read_text())['state'] in ('success', 'planned'):
finished.append(directory)
for directory in sorted(finished, key=lambda p: p.stat().st_mtime, reverse=True)[20:]:
if directory == current:
continue
command('git', '-C', str(CONFIG_REPO), 'worktree', 'remove', '--force', str(directory / 'source'))
shutil.rmtree(directory)
def follow(run_id, phase):
directory = run_directory(run_id)
groups = {
'apply': ('doctor', 'validate', 'apply-k8s', 'apply-compose'),
'verify': ('verify-k8s',),
'smoke': ('smoke',),
}
names = groups[phase]
offsets = {}
while True:
status = json.loads((directory / 'status.json').read_text())
for name in (*names, 'controller'):
path = directory / f'{name}.log'
if path.exists():
with path.open() as stream:
stream.seek(offsets.get(name, 0))
content = stream.read()
if content:
print(content, end='', flush=True)
offsets[name] = stream.tell()
stages = status['stages']
if all(stages.get(name, {}).get('result') in ('success', 'failure') for name in names):
return all(stages[name]['result'] == 'success' for name in names)
if status['state'] in ('success', 'failure', 'planned'):
return status['state'] in ('success', 'planned')
time.sleep(3)
def summary(run_id):
directory = run_directory(run_id)
request = json.loads((directory / 'request.json').read_text())
release = request['release']
plan_file = directory / 'plan.json'
lines = [
f'## Deploy `{release["sha"]}`',
'',
f'- Mode: `{request["mode"]}`',
f'- Refresh third-party images: `{request["refresh_images"]}`',
]
status = json.loads((directory / 'status.json').read_text())
if status.get('failure_stage'):
lines.append(f'- Failed stage: **{status["failure_stage"]}**')
lines.extend(
[
'',
f'- Observed run state: **{status["state"]}**',
'',
'### Stage results',
'| Stage | Result | Exit code |',
'| --- | --- | --- |',
]
)
for name in ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke'):
stage_result = status['stages'].get(name, {})
lines.append(f'| {name} | {stage_result.get("result", "not started")} | {stage_result.get("exit_code", "—")} |')
lines.extend(['', '### Apply and Helm recovery results'])
events_file = directory / 'apply-events.jsonl'
events = []
if events_file.exists():
for line in events_file.read_text().splitlines():
try:
events.append(json.loads(line))
except json.JSONDecodeError:
lines.append('- An operation record is incomplete. Check the stage log.')
latest = {(event['action'], event['target']): event['result'] for event in events}
lines.extend(f'- `{action}` `{target}`: **{result}**' for (action, target), result in latest.items())
if not latest:
lines.append('- No apply results were recorded.')
lines.append('- A completed apply does not confirm health. See verification and smoke results.')
lines.extend(['', '### Kubernetes recovery'])
pointer = directory / 'snapshot/current'
failed = Path(pointer.read_text().strip()) / 'failed-workloads' if pointer.exists() else None
if failed and failed.exists():
contents = failed.read_text()
counts = dict(re.findall(r'^(ROLLED_BACK|UNRECOVERED)=([0-9]+)$', contents, re.MULTILINE))
if not contents.strip():
lines.append('- No failed workloads were recorded. See the verification result above.')
elif counts:
lines.append(f'- Workloads restored: **{counts.get("ROLLED_BACK", "unknown")}**')
lines.append(f'- Workloads that need manual recovery: **{counts.get("UNRECOVERED", "unknown")}**')
else:
lines.append('- Rollback has no recorded result yet. Check the verification log.')
else:
lines.append('- No workload rollback was recorded. This does not confirm health.')
lines.append('- Compose requires manual recovery. Use the saved command in the apply log.')
if not plan_file.exists():
lines.extend(['', 'Plan was not created. Check the controller log.'])
print('\n'.join(lines))
return
plan = json.loads(plan_file.read_text())
lines.extend(['', '### Selected services'])
count = 0
for kind, services in plan['selected'].items():
for service in services:
lines.append(f'- `{kind}`: `{service}`')
count += 1
if not count:
lines.append('- None')
lines.extend(['', '### Selected Helm releases'])
lines.extend(f'- `{release}`' for release in plan.get('helm', []))
if not plan.get('helm'):
lines.append('- None')
lines.extend(['', '### Images pinned in the checked release'])
lines.extend(f'- `{image}@{digest}`' for image, digest in sorted(release['images'].items()))
lines.extend(['', '### Removed resources requiring manual review'])
lines.extend(f'- `{item}`' for item in plan.get('removed', []))
if not plan.get('removed'):
lines.append('- None')
print('\n'.join(lines))
def main():
os.umask(0o077)
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow', 'summary'))
parser.add_argument('run_id')
parser.add_argument('phase', nargs='?', choices=('apply', 'verify', 'smoke'))
parser.add_argument('--retry', action='store_true', help='Retry failed recovery checks; never repeat apply')
args = parser.parse_args()
directory = run_directory(args.run_id)
if args.action == 'start':
start(args.run_id)
elif args.action == 'execute':
execute(args.run_id)
elif args.action == 'recover':
with lock('deploy.lock'):
recover(directory, retry=args.retry)
elif args.action == 'status':
print((directory / 'status.json').read_text())
if (directory / 'plan.json').exists():
plan = json.loads((directory / 'plan.json').read_text())
print(json.dumps({k: plan[k] for k in ('sha', 'selected', 'helm', 'removed')}, indent=2))
elif args.action == 'summary':
summary(args.run_id)
elif not follow(args.run_id, args.phase):
sys.exit(1)
if __name__ == '__main__':
main()
File diff suppressed because it is too large. Load diff
+124
View File
@@ -0,0 +1,124 @@
#!/usr/bin/env python3
"""Calculate selected components against the last fully successful deploy."""
import hashlib
import json
import re
import subprocess
from pathlib import Path
def output(*args, **kwargs):
return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603
def tracked(repo):
return output('git', '-C', str(repo), 'ls-files').splitlines()
def helm_releases(repo):
text = (repo / '.gitea/workflows/deploy-lib.sh').read_text()
return [line.split('|') for line in re.findall(r'^ "([^"\n]+\|[^"\n]+)"$', text, re.MULTILINE)]
def inventory(repo):
files = tracked(repo)
k8s = sorted(
{f.split('/k8s/')[0] for f in files if '/k8s/' in f and (repo / f.split('/k8s/')[0] / 'k8s/active').is_file()}
)
compose = sorted(
{
str(Path(f).parent)
for f in files
if Path(f).name in ('compose.yaml', 'compose.yml') and (repo / Path(f).parent / 'active').is_file()
}
)
return {'k8s': k8s, 'compose': compose}
def file_hash(path):
return hashlib.sha256(path.read_bytes()).hexdigest() if path.is_file() else 'missing'
def make_plan(repo, config_repo, release, previous, mode, live_helm):
active = inventory(repo)
all_services = set(active['k8s'] + active['compose'])
helm_inputs = {}
helm_selected = []
for name, chart, namespace, version, values, marker in helm_releases(repo):
if not (repo / marker).is_file():
continue
value_path = repo / values if (repo / values).is_file() else config_repo / values
if not value_path.is_file():
raise ValueError(f'Missing Helm values: {values}')
stamp = hashlib.sha256(f'{chart}|{version}|{file_hash(value_path)}'.encode()).hexdigest()
helm_inputs[name] = stamp
live = next((h for h in live_helm if h['name'] == name and h['namespace'] == namespace), None)
if (
mode == 'full'
or previous is None
or previous.get('helm_inputs', {}).get(name) != stamp
or live is None
or live.get('status') != 'deployed'
or live.get('chart') != f'{chart.split("/")[-1]}-{version}'
):
helm_selected.append(name)
local_inputs = {}
for service in all_services:
candidates = [config_repo / service / '.env']
if service in active['compose']:
candidates.append(config_repo / '.env')
cfg = config_repo / service / 'config'
if cfg.is_dir():
candidates.extend(
p for p in cfg.rglob('*') if p.is_file() and p.suffix in ('.yaml', '.yml', '.json', '.conf')
)
local_inputs[service] = hashlib.sha256(
'\n'.join(f'{p.relative_to(config_repo)}:{file_hash(p)}' for p in sorted(candidates)).encode()
).hexdigest()
if previous is None:
if mode == 'changed':
raise ValueError('No successful baseline; run deploy in full mode first')
changed = set(all_services)
removed = []
else:
paths = output('git', '-C', str(repo), 'diff', '--name-only', previous['sha'], release['sha']).splitlines()
changed = {path.split('/')[0] for path in paths}
if any(path.startswith('.gitea/') for path in paths):
changed |= all_services
changed |= {s for s in all_services if previous.get('local_inputs', {}).get(s) != local_inputs[s]}
for file in tracked(repo):
service = file.split('/')[0]
if service not in all_services or not file.endswith(('.yaml', '.yml')):
continue
text = (repo / file).read_text()
if any(
image in text and previous.get('images', {}).get(image) != digest
for image, digest in release['images'].items()
):
changed.add(service)
removed = sorted(
set(previous.get('active', {}).get('k8s', []) + previous.get('active', {}).get('compose', []))
- all_services
)
removed += [path for path in paths if '/k8s/' in path and not (repo / path).exists()]
if mode == 'full':
changed = set(all_services)
dependencies = json.loads((repo / '.gitea/deploy-dependencies.json').read_text())
while True:
expanded = changed | {dependent for service in changed for dependent in dependencies.get(service, [])}
if expanded == changed:
break
changed = expanded
return {
'version': 1,
'sha': release['sha'],
'images': release['images'],
'active': active,
'selected': {kind: sorted(set(services) & changed) for kind, services in active.items()},
'helm': helm_selected,
'helm_inputs': helm_inputs,
'local_inputs': local_inputs,
'removed': sorted(set(removed)),
'full_smoke': mode == 'full' or 'traefik' in changed,
}
+10
View File
@@ -0,0 +1,10 @@
#!/usr/bin/env bash
set -euo pipefail
source "${REPO:?}/.gitea/workflows/deploy-lib.sh"
case "${1:?stage required}" in
workload-count)
select_manifests >/dev/null
selected_workload_refs | sort -u | wc -l
;;
*) run_stage "$1" ;;
esac
+102 -113
View File
@@ -1,151 +1,140 @@
name: deploy name: deploy
on: on:
# Deploy only what CI already validated. workflow_run is used instead of
# workflow_dispatch so a red lint/validate run can never reach the cluster.
workflow_run: workflow_run:
workflows: [ci] workflows: [ci]
branches: [main]
types: [completed] types: [completed]
workflow_dispatch: workflow_dispatch:
inputs:
deploy_ref:
description: "Commit already checked by successful main CI (main or SHA)"
default: main
required: true
deploy_mode:
description: "First deploy requires full; plan changes no production resources"
type: choice
options: [changed, full, plan]
default: changed
refresh_images:
description: "Explicitly refresh mutable third-party Compose tags"
type: boolean
default: false
# The deploy jobs read the tree, then reach the cluster over SSH with the
# deploy key. The Actions token itself is not part of that path, so it gets
# read-only contents and no more.
permissions: permissions:
contents: read contents: read
actions: read
concurrency: concurrency:
group: deploy-main group: deploy-main
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
# takes the verify job down with it, so a superseded deploy would leave the
# cluster half-applied and unchecked — the exact failure the verify job exists
# to catch. kubectl apply and docker compose up are both idempotent, so letting
# the older run finish and then deploying the newer commit costs little.
cancel-in-progress: false cancel-in-progress: false
env: env:
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} DEPLOY_HOST: ${{ vars.DEPLOY_HOST || secrets.DEPLOY_HOST }}
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }} DEPLOY_PORT: ${{ vars.DEPLOY_PORT || secrets.DEPLOY_PORT }}
DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_USER: ${{ vars.DEPLOY_USER || secrets.DEPLOY_USER }}
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} DEPLOY_KNOWN_HOSTS: ${{ vars.DEPLOY_KNOWN_HOSTS }}
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the DEPLOY_RUN_ID: ${{ github.run_id }}-${{ github.run_attempt || 1 }}
# finished ci run checked. Pin the exact validated commit instead, so a push DEPLOY_MODE: ${{ inputs.deploy_mode || 'changed' }}
# landing mid-deploy cannot make the workstation deploy something else. Also REFRESH_IMAGES: ${{ inputs.refresh_images && 'true' || 'false' }}
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
# which falls back to the current origin/main.
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
jobs: jobs:
preflight: gate:
if: >- if: >-
github.event_name != 'workflow_run' || github.ref == 'refs/heads/main' &&
(github.event.workflow_run.conclusion == 'success' && (vars.AUTODEPLOY == 'true' || github.event_name == 'workflow_dispatch') &&
github.event.workflow_run.head_branch == 'main') (github.event_name != 'workflow_run' ||
runs-on: [self-hosted, linux, arch, homelab, prod] (github.event.workflow_run.conclusion == 'success' && github.event.workflow_run.head_branch == 'main'))
runs-on: homelab
timeout-minutes: 10 timeout-minutes: 10
outputs:
sha: ${{ steps.release.outputs.sha }}
steps: steps:
- name: Checkout repository - name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
- name: Fetch and reset workstation fetch-depth: 0
shell: bash - name: Check successful CI and download the exact commit release
id: release
env:
GITEA_TOKEN: ${{ github.token }}
DEPLOY_REF: ${{ inputs.deploy_ref || 'main' }}
EVENT_SHA: ${{ github.event.workflow_run.head_sha }}
run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA"
- name: Submit durable deploy to workstation
run: bash .gitea/workflows/ssh-run.sh start
- name: Write the request result
if: always()
env:
REQUEST_RESULT: ${{ job.status }}
CHECKED_SHA: ${{ steps.release.outputs.sha }}
run: | run: |
set -euo pipefail if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
./.gitea/workflows/ssh-run.sh preflight printf '## Deploy request\n\n- Result: **%s**\n- Checked commit: %s\n- Mode: %s\n' "$REQUEST_RESULT" "${CHECKED_SHA:-not checked}" "$DEPLOY_MODE" >>"$GITHUB_STEP_SUMMARY"
if [ "$REQUEST_RESULT" != success ]; then
echo 'Open the failed step log. If SSH submission failed, check the remote controller state.' >>"$GITHUB_STEP_SUMMARY"
fi
fi
validate: apply:
needs: [preflight] needs: [gate]
runs-on: [self-hosted, linux, arch, homelab, prod] runs-on: homelab
timeout-minutes: 20 timeout-minutes: 120
steps: steps:
- name: Checkout repository - name: Checkout checked commit
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
- name: Dry-run manifests and check Secrets ref: ${{ needs.gate.outputs.sha }}
shell: bash - name: Follow validation and sequential Kubernetes / Compose apply
run: bash .gitea/workflows/ssh-run.sh apply
- name: Write the deploy result
if: always()
run: | run: |
set -euo pipefail if [ -f .gitea/workflows/ssh-run.sh ]; then
./.gitea/workflows/ssh-run.sh validate bash .gitea/workflows/ssh-run.sh summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY"
fi
apply-k8s: verify:
needs: [validate] needs: [gate, apply]
runs-on: [self-hosted, linux, arch, homelab, prod] if: always() && needs.gate.result == 'success'
# Apply only, no verification, so this is just the work itself: snapshot, runs-on: homelab
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the timeout-minutes: 130
# apply loop. Verification has its own job and its own budget.
timeout-minutes: 45
steps: steps:
- name: Checkout repository - name: Checkout checked commit
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
- name: Apply Kubernetes manifests ref: ${{ needs.gate.outputs.sha }}
shell: bash - name: Follow workload verification and recovery
run: bash .gitea/workflows/ssh-run.sh verify
- name: Write the deploy result
if: always()
run: | run: |
set -euo pipefail if [ -f .gitea/workflows/ssh-run.sh ]; then
./.gitea/workflows/ssh-run.sh apply-k8s bash .gitea/workflows/ssh-run.sh summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY"
fi
apply-compose:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Redeploy docker compose stacks
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh apply-compose
# Watches the workloads this deploy changed and rolls back the ones that never
# became healthy. Runs even when the apply jobs failed, timed out or were
# cancelled — that is the whole point of splitting it out. `always()` is what
# lets it start after a failed dependency; the needs on apply-compose are a
# barrier, so verification begins only once both applies are done.
verify-k8s:
needs: [apply-k8s, apply-compose]
if: >-
always() &&
needs.apply-k8s.result != 'skipped' &&
needs.apply-compose.result != 'skipped'
runs-on: [self-hosted, linux, arch, homelab, prod]
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Verify workloads and roll back on failure
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh verify-k8s
# Asks the public route of every active service whether it is actually
# serving, which the rollout check above structurally cannot: a pod can
# converge and still be crash-looping, or be listening on a port no Service
# points at, or answer 500.
#
# `always()` for the same reason verify-k8s has it, and it runs after that job
# specifically because a rollback is when a route most needs re-checking. The
# needs is a barrier, not a filter: whether verify-k8s passed, failed or was
# cancelled, the probes are what say whether the cluster is serving, and
# suppressing them on a rollback would hide the one run where the answer
# matters most.
smoke: smoke:
needs: [verify-k8s] needs: [gate, verify]
if: always() && needs.verify-k8s.result != 'skipped' if: always() && needs.gate.result == 'success'
runs-on: [self-hosted, linux, arch, homelab, prod] runs-on: homelab
timeout-minutes: 10 timeout-minutes: 15
steps: steps:
- name: Checkout repository - name: Checkout checked commit
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
- name: Probe the public route of every active service ref: ${{ needs.gate.outputs.sha }}
shell: bash - name: Follow public route checks
run: bash .gitea/workflows/ssh-run.sh smoke
- name: Write the deploy result
if: always()
run: | run: |
set -euo pipefail if [ -f .gitea/workflows/ssh-run.sh ]; then
./.gitea/workflows/ssh-run.sh smoke bash .gitea/workflows/ssh-run.sh summary
elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY"
fi
+85 -33
View File
@@ -13,9 +13,18 @@ here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=tool-versions.env # shellcheck source=tool-versions.env
. "$here/tool-versions.env" . "$here/tool-versions.env"
TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}" TOOLS_DIR="${TOOLS_DIR:-${XDG_CACHE_HOME:-$HOME/.cache}/homelab-ci}"
BIN_DIR="$TOOLS_DIR/bin" BIN_DIR="$TOOLS_DIR/bin"
mkdir -p "$BIN_DIR" mkdir -p "$BIN_DIR"
# A runner may accept overlapping workflows even though each workflow is sequential.
exec 9>"$TOOLS_DIR/install.lock"
flock -w 300 9
export UV_TOOL_DIR="$TOOLS_DIR/uv-tools"
export UV_CACHE_DIR="$TOOLS_DIR/uv-cache"
# The just-installed tools must resolve inside this script too: callers only
# prepend BIN_DIR to PATH after the script exits, so a bare `uv` below would
# miss the binary install_uv just placed (exit 127 on a clean runner).
export PATH="$BIN_DIR:$PATH"
arch="$(uname -m)" arch="$(uname -m)"
# Upstream projects disagree on arch spelling: kubeconform and actionlint use # Upstream projects disagree on arch spelling: kubeconform and actionlint use
@@ -44,7 +53,7 @@ esac
fetch() { fetch() {
# fetch <url> <dest> # fetch <url> <dest>
if command -v curl >/dev/null 2>&1; then if command -v curl >/dev/null 2>&1; then
curl -sSLf --retry 3 -o "$2" "$1" curl -sSLf --connect-timeout 15 --max-time 120 --retry 3 -o "$2" "$1"
elif command -v wget >/dev/null 2>&1; then elif command -v wget >/dev/null 2>&1; then
wget -q -O "$2" "$1" wget -q -O "$2" "$1"
else else
@@ -53,24 +62,44 @@ fetch() {
fi fi
} }
# resolve <command>
# Absolute path to use for invoking a tool: the copy in BIN_DIR when present,
# otherwise the name for PATH lookup. Every version check and every in-script
# invocation goes through this, so a tool missing from both places reads as
# "not installed" instead of dying with 127 under `set -e`.
resolve() {
if [ -x "$BIN_DIR/$1" ]; then
printf '%s' "$BIN_DIR/$1"
else
printf '%s' "$1"
fi
}
# installed_version <command> # installed_version <command>
# Prints the version of an already-installed tool, or nothing. Each tool spells # Prints the version of an already-installed tool, or nothing. Each tool spells
# its version flag differently, hence the case. # its version flag differently, hence the case.
installed_version() { installed_version() {
local out local bin out
bin="$(resolve "$1")"
if ! command -v "$bin" >/dev/null 2>&1; then
return 0
fi
case "$1" in case "$1" in
kubeconform) out="$("$1" -v 2>/dev/null | head -1 || true)" ;; kubeconform) out="$("$bin" -v 2>/dev/null | head -1 || true)" ;;
*) out="$("$1" --version 2>/dev/null | head -1 || true)" ;; *) out="$("$bin" --version 2>/dev/null | head -1 || true)" ;;
esac esac
printf '%s' "$out" printf '%s' "$out"
} }
# at_version <command> <expected> # at_version <command> <expected>
at_version() { at_version() {
case "$(installed_version "$1")" in local version expected="${2#v}"
*"$2"*) return 0 ;; version="$(installed_version "$1")"
*) return 1 ;; if [[ "$version" =~ (^|[^0-9.])v?([0-9]+(\.[0-9]+)+) ]]; then
esac [ "${BASH_REMATCH[2]}" = "$expected" ]
else
return 1
fi
} }
install_kubeconform() { install_kubeconform() {
@@ -99,6 +128,15 @@ install_shellcheck() {
rm -rf "$tmp" rm -rf "$tmp"
} }
install_jq() {
if at_version jq "${JQ_VERSION}"; then
return 0
fi
fetch "https://github.com/jqlang/jq/releases/download/jq-${JQ_VERSION}/jq-linux-${goarch}" \
"$BIN_DIR/jq"
chmod 0755 "$BIN_DIR/jq"
}
install_uv() { install_uv() {
if at_version uv "${UV_VERSION}"; then if at_version uv "${UV_VERSION}"; then
return 0 return 0
@@ -130,7 +168,7 @@ install_uv_tool() {
return 0 return 0
fi fi
install_uv install_uv
UV_TOOL_BIN_DIR="$BIN_DIR" uv tool install --force "$1==$2" >/dev/null UV_TOOL_BIN_DIR="$BIN_DIR" "$BIN_DIR/uv" tool install --force "$1==$2" >/dev/null
} }
install_ruff() { install_ruff() {
@@ -146,6 +184,7 @@ install_pip_audit() {
} }
install_prettier() { install_prettier() {
install_node
if at_version prettier "${PRETTIER_VERSION}"; then if at_version prettier "${PRETTIER_VERSION}"; then
return 0 return 0
fi fi
@@ -206,28 +245,41 @@ install_actionlint() {
rm -rf "$tmp" rm -rf "$tmp"
} }
wanted=("$@") main() {
if [ "${#wanted[@]}" -eq 0 ]; then wanted=("$@")
wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint) if [ "${#wanted[@]}" -eq 0 ]; then
fi wanted=(node jq kubeconform shellcheck actionlint prettier ruff yamllint hadolint)
fi
for tool in "${wanted[@]}"; do for tool in "${wanted[@]}"; do
case "$tool" in case "$tool" in
kubeconform) install_kubeconform ;; kubeconform) install_kubeconform ;;
shellcheck) install_shellcheck ;; shellcheck) install_shellcheck ;;
actionlint) install_actionlint ;; jq) install_jq ;;
prettier) install_prettier ;; actionlint) install_actionlint ;;
ruff) install_ruff ;; prettier) install_prettier ;;
yamllint) install_yamllint ;; ruff) install_ruff ;;
pip-audit) install_pip_audit ;; yamllint) install_yamllint ;;
hadolint) install_hadolint ;; pip-audit) install_pip_audit ;;
node) install_node ;; hadolint) install_hadolint ;;
uv) install_uv ;; node) install_node ;;
*) uv) install_uv ;;
echo "install-ci-tools: unknown tool: $tool" >&2 *)
exit 1 echo "install-ci-tools: unknown tool: $tool" >&2
;; exit 1
esac ;;
done esac
done
printf '%s\n' "$BIN_DIR" for old in "$BIN_DIR"/node-* "$BIN_DIR"/prettier-*; do
[ -d "$old" ] || continue
case "$(basename "$old")" in
"node-$NODE_VERSION"|"prettier-$PRETTIER_VERSION") ;;
*) rm -rf "$old" ;;
esac
done
if [ -x "$BIN_DIR/uv" ]; then "$BIN_DIR/uv" cache prune >/dev/null; fi
printf '%s\n' "$BIN_DIR"
}
if [ "${BASH_SOURCE[0]}" = "$0" ]; then main "$@"; fi
+516
View File
@@ -0,0 +1,516 @@
#!/usr/bin/env python3
"""CI release artifacts and the SHA-specific Gitea deployment gate (stdlib only)."""
import argparse
import hashlib
import io
import itertools
import json
import os
import re
import shutil
import subprocess
import sys
import tempfile
import urllib.error
import urllib.parse
import urllib.request
import zipfile
from pathlib import Path
SHA = re.compile(r'[0-9a-f]{40}')
DIGEST = re.compile(r'sha256:[0-9a-f]{64}')
IMAGES = {
'error-pages': ('errorpages', 'errorpages/Dockerfile'),
'forust-homepage': ('homepages', 'homepages/Dockerfile.forust'),
'xdfnx-homepage': ('homepages', 'homepages/Dockerfile.xdfnx'),
}
def command(*args, **kwargs):
"""Arguments are passed directly to the executable, never to a shell."""
return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603, S607
def validate_release(data, sha=None):
if data.get('version') != 1 or not SHA.fullmatch(data.get('sha', '')):
raise ValueError('Invalid release version or SHA')
if sha is not None and data['sha'] != sha:
raise ValueError('Release SHA does not match the checked CI commit')
expected = {f'gcr.forust.xyz/forust/{name}' for name in IMAGES}
if set(data.get('images', {})) != expected:
raise ValueError('Release must contain all owned images')
if not all(DIGEST.fullmatch(value) for value in data['images'].values()):
raise ValueError('Release has an invalid image digest')
if set(data.get('inputs', {})) != expected or not all(
re.fullmatch(r'[0-9a-f]{64}', value) for value in data['inputs'].values()
):
raise ValueError('Release has invalid build input fingerprints')
return data
class NoRedirect(urllib.request.HTTPRedirectHandler):
def redirect_request(self, _req, _fp, _code, _msg, _headers, _newurl):
return None
class Gitea:
def __init__(self):
self.origin = os.environ['GITHUB_SERVER_URL'].rstrip('/')
if urllib.parse.urlsplit(self.origin).scheme != 'https':
raise ValueError('Gitea API must use HTTPS')
self.repository = os.environ['GITHUB_REPOSITORY']
if not re.fullmatch(r'[\w.-]+/[\w.-]+', self.repository):
raise ValueError('Invalid Gitea repository')
self.token = os.environ['GITEA_TOKEN']
self.base = f'{self.origin}/api/v1/repos/{self.repository}'
def request(self, url, *, archive=False):
if not url.startswith(self.base + '/'):
raise ValueError('Refusing to send the Actions token to another origin')
req = urllib.request.Request(url, headers={'Authorization': f'token {self.token}'}) # noqa: S310 -- HTTPS origin validated above
opener = urllib.request.build_opener(NoRedirect())
try:
response = opener.open(req, timeout=30) # noqa: S310
except urllib.error.HTTPError as error:
if not archive or error.code not in (301, 302, 303, 307, 308):
raise RuntimeError(f'Gitea API returned HTTP {error.code}') from None
target = urllib.parse.urljoin(url, error.headers['Location'])
if urllib.parse.urlsplit(target).scheme != 'https':
raise ValueError('Artifact redirect must use HTTPS') from None
# Signed storage redirects must never receive the Gitea token.
response = urllib.request.urlopen(target, timeout=30) # noqa: S310
with response:
payload = response.read(8 * 1024 * 1024 + 1)
if len(payload) > 8 * 1024 * 1024:
raise ValueError('Gitea response exceeds 8 MiB')
return payload if archive else json.loads(payload)
def pages(self, path, key, **params):
for page in range(1, 101):
query = urllib.parse.urlencode({**params, 'page': page, 'limit': 50})
data = self.request(f'{self.base}/{path}?{query}')
entries = data[key]
yield from entries
if len(entries) < 50:
return
raise RuntimeError('Gitea pagination limit exceeded')
def successful_runs(self, sha=None):
params = {'branch': 'main', 'status': 'success', 'exclude_pull_requests': 'true'}
if sha:
params['head_sha'] = sha
for run in self.pages('actions/workflows/ci.yaml/runs', 'workflow_runs', **params):
if (
run.get('status') == 'completed'
and run.get('conclusion') == 'success'
and run.get('head_branch') == 'main'
and run.get('event') in ('push', 'workflow_dispatch')
and (run.get('repository') or {}).get('full_name') == self.repository
and (run.get('head_repository') or run.get('repository') or {}).get('full_name') == self.repository
and (sha is None or run.get('head_sha') == sha)
):
yield run
def release(self, run):
sha = run['head_sha']
jobs = list(self.pages(f'actions/runs/{run["id"]}/jobs', 'jobs'))
# A green workflow with a skipped build must not authorize a deploy.
if not any(job.get('name') == 'build' and job.get('conclusion') == 'success' for job in jobs):
raise ValueError('CI build job did not succeed')
artifacts = self.request(f'{self.base}/actions/runs/{run["id"]}/artifacts')['artifacts']
matching = [a for a in artifacts if a['name'] == f'release-{sha}' and not a.get('expired')]
if len(matching) != 1:
raise ValueError('CI release artifact is missing, expired or ambiguous; rerun CI')
blob = self.request(f'{self.base}/actions/artifacts/{matching[0]["id"]}/zip', archive=True)
with zipfile.ZipFile(io.BytesIO(blob)) as archive:
files = [entry for entry in archive.infolist() if not entry.is_dir()]
if len(files) != 1 or files[0].filename != 'release.json' or files[0].file_size > 256 * 1024:
raise ValueError('Unexpected release archive contents')
return validate_release(json.loads(archive.read(files[0])), sha)
def fingerprint(context, dockerfile):
tree = command('git', 'ls-tree', '-r', 'HEAD', '--', context, dockerfile, '.gitea/workflows/release.py')
return hashlib.sha256(tree.encode()).hexdigest()
def gate(output, requested_ref, event_sha):
command('git', 'fetch', '--quiet', 'origin', 'main')
if event_sha:
if not SHA.fullmatch(event_sha):
raise ValueError('Invalid workflow_run SHA')
sha = event_sha
else:
if requested_ref == 'main':
requested_ref = 'origin/main'
sha = command('git', 'rev-parse', '--verify', '--end-of-options', f'{requested_ref}^{{commit}}')
if not SHA.fullmatch(sha):
raise ValueError('Invalid deploy SHA')
command('git', 'merge-base', '--is-ancestor', sha, 'origin/main')
api = Gitea()
runs = list(api.successful_runs(sha))
if not runs:
raise ValueError(f'No successful main CI for {sha}; run CI before deploying')
release = api.release(max(runs, key=lambda run: run['id']))
output.write_text(json.dumps(release, indent=2) + '\n')
if os.environ.get('GITHUB_OUTPUT'):
with Path(os.environ['GITHUB_OUTPUT']).open('a') as stream:
stream.write(f'sha={sha}\n')
print(f'CI gate accepted {sha}')
def prepare_images(output):
sha = command('git', 'rev-parse', 'HEAD')
if sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha):
raise ValueError('Build checkout does not match GITHUB_SHA')
api = Gitea()
previous = None
for run in sorted(itertools.islice(api.successful_runs(), 50), key=lambda item: item['id'], reverse=True):
if str(run['id']) == os.environ.get('GITHUB_RUN_ID'):
continue
try:
previous = api.release(run)
break
except ValueError:
# Expired artifacts only cost a rebuild; mutable tags are never a fallback.
continue
targets = []
for name, (context, dockerfile) in IMAGES.items():
image = f'gcr.forust.xyz/forust/{name}'
inputs = fingerprint(context, dockerfile)
old_digest = (previous or {}).get('images', {}).get(image)
targets.append(
{
'name': name,
'image': image,
'context': context,
'dockerfile': dockerfile,
'inputs': inputs,
'reuse_digest': old_digest if (previous or {}).get('inputs', {}).get(image) == inputs else None,
}
)
output.write_text(json.dumps({'sha': sha, 'targets': targets}, indent=2) + '\n')
if os.environ.get('GITHUB_OUTPUT'):
with Path(os.environ['GITHUB_OUTPUT']).open('a') as stream:
stream.write('matrix=' + json.dumps({'include': targets}, separators=(',', ':')) + '\n')
print(f'Prepared {len(targets)} image jobs; {sum(t["reuse_digest"] is None for t in targets)} require builds')
def checked_plan(path):
data = json.loads(path.read_text())
sha = command('git', 'rev-parse', 'HEAD')
if data.get('sha') != sha or sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha):
raise ValueError('Image plan does not match the checked source commit')
targets = data.get('targets', [])
if sorted(t['name'] for t in targets) != sorted(IMAGES):
raise ValueError('Image plan must contain each owned image once')
for target in targets:
name = target['name']
context, dockerfile = IMAGES[name]
if (target['context'], target['dockerfile'], target['image']) != (
context,
dockerfile,
f'gcr.forust.xyz/forust/{name}',
) or target['inputs'] != fingerprint(context, dockerfile):
raise ValueError('Image plan has invalid build inputs')
if target['reuse_digest'] is not None and not DIGEST.fullmatch(target['reuse_digest']):
raise ValueError('Image plan has an invalid reuse digest')
return data
def build_images(output, report, name, plan):
data = checked_plan(plan)
sha = data['sha']
target = next(t for t in data['targets'] if t['name'] == name)
context, dockerfile = IMAGES[name]
docker_config = tempfile.mkdtemp(prefix='homelab-registry-')
builder_config = Path.home() / '.cache/homelab-ci/buildx'
builder_config.mkdir(parents=True, exist_ok=True)
env = {**os.environ, 'DOCKER_CONFIG': docker_config, 'BUILDX_CONFIG': str(builder_config)}
try:
report['phase'] = 'Registry login'
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
'login',
'gcr.forust.xyz',
'-u',
os.environ['REGISTRY_USERNAME'],
'--password-stdin',
],
input=os.environ['REGISTRY_PASSWORD'],
text=True,
check=True,
env=env,
)
report['phase'] = 'Prepare the builder'
builder = 'homelab-ci'
versions = dict(
re.findall(r'^([A-Z_]+)="([^"\n]+)"$', Path('.gitea/workflows/tool-versions.env').read_text(), re.MULTILINE)
)
image = versions['BUILDKIT_IMAGE']
signature = builder_config / 'homelab-ci-image'
exists = (
subprocess.run( # noqa: S603
[shutil.which('docker') or '/usr/bin/docker', 'buildx', 'inspect', builder],
capture_output=True,
env=env,
).returncode
== 0
)
if exists and (not signature.exists() or signature.read_text().strip() != image):
command('docker', 'buildx', 'rm', '--keep-state', builder, env=env)
exists = False
if not exists:
command(
'docker',
'buildx',
'create',
'--name',
builder,
'--driver',
'docker-container',
'--driver-opt',
f'image={image}',
'--buildkitd-config',
'.gitea/runner/buildkitd.toml',
env=env,
)
signature.write_text(image + '\n')
release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}}
report['images'] = release['images']
report['phase'] = f'Build or reuse {name}'
report['current'] = name
image = f'gcr.forust.xyz/forust/{name}'
inputs = target['inputs']
old_digest = target['reuse_digest']
exists = False
if old_digest:
exists = (
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
'buildx',
'imagetools',
'inspect',
f'{image}@{old_digest}',
],
capture_output=True,
env=env,
timeout=60,
).returncode
== 0
)
if exists:
print(f'Reuse {name}: inputs unchanged')
digest = old_digest
report['reused'].append(name)
else:
print(f'Build {name}', flush=True)
metadata = Path(docker_config) / 'metadata.json'
command(
'docker',
'buildx',
'build',
'--builder',
builder,
'--platform',
'linux/amd64',
'--provenance=false',
'--cache-from',
f'type=registry,ref={image}:buildcache',
'--cache-to',
f'type=registry,ref={image}:buildcache,mode=max',
'--output',
f'type=image,name={image},push-by-digest=true,name-canonical=true,push=true',
'--metadata-file',
str(metadata),
'--file',
dockerfile,
context,
env=env,
)
digest = json.loads(metadata.read_text())['containerimage.digest']
report['built'].append(name)
release['images'][image] = digest
release['inputs'][image] = inputs
if not DIGEST.fullmatch(digest):
raise ValueError('Image job returned an invalid digest')
output.write_text(json.dumps(release, indent=2) + '\n')
report['current'] = None
report['phase'] = 'Release file saved'
finally:
# Cleanup errors must neither leak credentials nor mask the original build error.
try:
subprocess.run( # noqa: S603
[
shutil.which('docker') or '/usr/bin/docker',
'buildx',
'prune',
'--builder',
'homelab-ci',
'--force',
'--max-used-space',
'1gb',
],
env=env,
timeout=60,
)
except (OSError, subprocess.TimeoutExpired):
print('CI builder cache cleanup deferred', flush=True)
finally:
shutil.rmtree(docker_config)
def write_summary(lines):
path = os.environ.get('GITHUB_STEP_SUMMARY')
if path:
try:
with Path(path).open('a') as stream:
stream.write('\n'.join(lines) + '\n\n')
except OSError:
print('WARNING: cannot write the job summary')
def check_summary():
lines = [
f'## {os.environ["SUMMARY_CHECK"]}',
'',
f'- Commit: `{os.environ.get("GITHUB_SHA", "unknown")}`',
f'- Result: **{os.environ["SUMMARY_RESULT"]}**',
]
if os.environ.get('SUMMARY_FAILED_STEP'):
lines.append(f'- Failed step: {os.environ["SUMMARY_FAILED_STEP"]}')
if os.environ['SUMMARY_RESULT'] != 'success':
lines.append('- Open the failed step log for the error details.')
write_summary(lines)
def build(output, name, plan):
report = {'phase': 'Check the source commit', 'current': None, 'built': [], 'reused': [], 'images': {}}
result = 'failure'
try:
build_images(output, report, name, plan)
result = 'success'
finally:
lines = [
f'## Image release `{os.environ.get("GITHUB_SHA", "unknown")}`',
'',
f'- Result: **{result}**',
f'- Last stage: {report["phase"]}',
]
if result == 'failure':
lines.append('- No release from this build can be deployed. Open the failed step log.')
if report['current']:
lines.append(f'- Image at the failure: `{report["current"]}`')
for title, key in (('Built', 'built'), ('Reused from successful CI', 'reused')):
lines.extend(['', f'### {title}'])
lines.extend(f'- `{name}`' for name in report[key])
if not report[key]:
lines.append('- None')
lines.extend(['', '### Completed image digests'])
lines.extend(f'- `{image}@{digest}`' for image, digest in report['images'].items())
if not report['images']:
lines.append('- None')
write_summary(lines)
def render(stream, destination):
release = validate_release(json.loads(Path(os.environ['RELEASE_FILE']).read_text()), os.environ['DEPLOY_SHA'])
image_line = re.compile(
r"^(\s*(?:-\s*)?image:\s*)(['\"]?)(gcr\.forust\.xyz/forust/[\w.-]+)(?::[\w.-]+|@sha256:[0-9a-f]{64})\2(\s*(?:#.*)?)$"
)
rendered = []
for line in stream:
match = image_line.fullmatch(line.rstrip('\n'))
if match:
prefix, quote, image, tail = match.groups()
if image not in release['images']:
raise ValueError(f'Owned image missing from checked release: {image}')
line = f'{prefix}{quote}{image}@{release["images"][image]}{quote}{tail}\n'
elif re.match(r'\s*(?:-\s*)?image:', line) and 'gcr.forust.xyz/forust/' in line:
raise ValueError('Unsupported owned image syntax; refusing to apply a mutable tag')
rendered.append(line)
destination.writelines(rendered)
def finalize_images(output, fragments, plan):
data = checked_plan(plan)
sha = data['sha']
release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}}
for name in IMAGES:
fragment = json.loads((fragments / f'image-{name}' / 'image.json').read_text())
image = f'gcr.forust.xyz/forust/{name}'
if fragment.get('sha') != sha or fragment.get('version') != 1 or set(fragment.get('images', {})) != {image}:
raise ValueError('Image job artifact is missing or belongs to another commit')
target = next(t for t in data['targets'] if t['name'] == name)
if fragment.get('inputs') != {image: target['inputs']}:
raise ValueError('Image artifact does not match the build plan')
release['images'].update(fragment['images'])
release['inputs'].update(fragment['inputs'])
validate_release(release, sha)
# Only a complete set of successful image jobs can publish the release tags.
docker_config = tempfile.mkdtemp(prefix='homelab-registry-')
env = {**os.environ, 'DOCKER_CONFIG': docker_config}
try:
subprocess.run( # noqa: S603, S607
[
shutil.which('docker') or '/usr/bin/docker',
'login',
'gcr.forust.xyz',
'-u',
os.environ['REGISTRY_USERNAME'],
'--password-stdin',
],
input=os.environ['REGISTRY_PASSWORD'],
text=True,
check=True,
env=env,
)
for image, digest in release['images'].items():
command(
'docker',
'buildx',
'imagetools',
'create',
'--prefer-index=false',
'--tag',
f'{image}:sha-{sha}',
f'{image}@{digest}',
env=env,
timeout=90,
)
output.write_text(json.dumps(release, indent=2) + '\n')
finally:
shutil.rmtree(docker_config)
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('action', choices=('prepare', 'image', 'finalize', 'gate', 'render', 'check-summary'))
parser.add_argument('--output', type=Path, default=Path('release.json'))
parser.add_argument('--ref', default='main')
parser.add_argument('--event-sha', default='')
parser.add_argument('--image', choices=IMAGES)
parser.add_argument('--plan', type=Path, default=Path('build-plan.json'))
parser.add_argument('--fragments', type=Path, default=Path('artifacts'))
args = parser.parse_args()
if args.action == 'check-summary':
check_summary()
elif args.action == 'render':
render(sys.stdin, sys.stdout)
elif args.action == 'gate':
gate(args.output, args.ref, args.event_sha)
elif args.action == 'prepare':
prepare_images(args.output)
elif args.action == 'image':
if not args.image:
parser.error('--image is required')
build(args.output, args.image, args.plan)
else:
finalize_images(args.output, args.fragments, args.plan)
if __name__ == '__main__':
main()
+41 -17
View File
@@ -1,10 +1,26 @@
name: renovate-ci name: renovate-ci
on: on:
pull_request: # Read the workflow from the trusted base branch. PR code runs only on the
# unprivileged runner selected below.
pull_request_target:
paths:
- "renovate/**"
- ".gitea/workflows/renovate-ci.yaml"
- ".gitea/workflows/sync-renovate-configmap.sh"
- ".gitea/workflows/compose-lint.sh"
- ".gitea/workflows/install-ci-tools.sh"
- ".gitea/workflows/tool-versions.env"
push: push:
branches: branches:
- main - main
paths:
- "renovate/**"
- ".gitea/workflows/renovate-ci.yaml"
- ".gitea/workflows/sync-renovate-configmap.sh"
- ".gitea/workflows/compose-lint.sh"
- ".gitea/workflows/install-ci-tools.sh"
- ".gitea/workflows/tool-versions.env"
workflow_dispatch: workflow_dispatch:
permissions: permissions:
@@ -12,37 +28,47 @@ permissions:
jobs: jobs:
validate-renovate: validate-renovate:
runs-on: [self-hosted, linux, arch, homelab] runs-on: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' && 'homelab' || 'homelab-pr' }}
timeout-minutes: 20 timeout-minutes: 20
steps: steps:
- name: Checkout repository - name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
ref: ${{ github.event_name == 'pull_request_target' && github.event.pull_request.head.sha || github.sha }}
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag, # renovate/k8s/cronjob.yaml is the single source of truth for the version.
# so the same version that runs in the cluster is the one validated here. - name: Resolve the deployed Renovate version
- name: Resolve the deployed Renovate image
id: image id: image
shell: bash shell: bash
run: | run: |
set -euo pipefail set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \ image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)" renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then if [[ ! "$image" =~ ^renovate/renovate:([0-9]+\.[0-9]+\.[0-9]+)$ ]]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml" echo "::error::expected a pinned renovate/renovate semantic version in renovate/k8s/cronjob.yaml"
exit 1 exit 1
fi fi
echo "using $image" version="${BASH_REMATCH[1]}"
echo "image=$image" >> "$GITHUB_OUTPUT" echo "using Renovate $version"
printf 'version=%s\n' "$version" >> "$GITHUB_OUTPUT"
- name: Validate Renovate repository config - name: Prepare pinned validation tools
shell: bash shell: bash
run: | run: |
set -euo pipefail set -euo pipefail
docker run --rm \ tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform node)"
-v "$PWD/renovate:/opt/renovate:ro" \ echo "$tools_dir" >> "$GITHUB_PATH"
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \ - name: Validate Renovate repository config
renovate-config-validator /opt/renovate/renovate.json shell: bash
env:
RENOVATE_VERSION: ${{ steps.image.outputs.version }}
run: |
set -euo pipefail
npm_cache="$(mktemp -d "${RUNNER_TEMP:-/tmp}/renovate-npm-cache.XXXXXXXX")"
trap 'rm -rf "$npm_cache"' EXIT
NPM_CONFIG_CACHE="$npm_cache" RENOVATE_CONFIG_FILE="$PWD/renovate/renovate.json" \
npm exec --yes --package="renovate@${RENOVATE_VERSION}" -- renovate-config-validator
# The CronJob cannot read the repository, so renovate/k8s/configmap.yaml # The CronJob cannot read the repository, so renovate/k8s/configmap.yaml
# carries an inlined copy of the config. Fail if it no longer matches. # carries an inlined copy of the config. Fail if it no longer matches.
@@ -56,8 +82,6 @@ jobs:
shell: bash shell: bash
run: | run: |
set -euo pipefail set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
export PATH="$tools_dir:$PATH"
kubeconform \ kubeconform \
-strict \ -strict \
-ignore-missing-schemas \ -ignore-missing-schemas \
+13 -7
View File
@@ -32,11 +32,14 @@ concurrency:
jobs: jobs:
run-renovate: run-renovate:
runs-on: [self-hosted, linux, arch, homelab] if: github.ref == 'refs/heads/main'
runs-on: homelab
timeout-minutes: 60 timeout-minutes: 60
steps: steps:
- name: Checkout repository - name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
ref: refs/heads/main
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag. # renovate/k8s/cronjob.yaml is the single source of truth for the image tag.
# Reading it here means this workflow validates and runs the exact version # Reading it here means this workflow validates and runs the exact version
@@ -48,21 +51,23 @@ jobs:
set -euo pipefail set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \ image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)" renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then if [[ ! "$image" =~ ^renovate/renovate:[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml" echo "::error::expected a pinned renovate/renovate semantic version in renovate/k8s/cronjob.yaml"
exit 1 exit 1
fi fi
echo "using $image" echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT" printf 'image=%s\n' "$image" >> "$GITHUB_OUTPUT"
- name: Validate Renovate config - name: Validate Renovate config
shell: bash shell: bash
env:
RENOVATE_IMAGE: ${{ steps.image.outputs.image }}
run: | run: |
set -euo pipefail set -euo pipefail
docker run --rm \ docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \ -v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \ -e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \ "$RENOVATE_IMAGE" \
renovate-config-validator renovate-config-validator
- name: Run Renovate - name: Run Renovate
@@ -73,6 +78,7 @@ jobs:
RENOVATE_REPOSITORIES: ${{ inputs.repositories }} RENOVATE_REPOSITORIES: ${{ inputs.repositories }}
RENOVATE_DRY_RUN: ${{ inputs.dry_run && 'full' || '' }} RENOVATE_DRY_RUN: ${{ inputs.dry_run && 'full' || '' }}
LOG_LEVEL: ${{ inputs.log_level }} LOG_LEVEL: ${{ inputs.log_level }}
RENOVATE_IMAGE: ${{ steps.image.outputs.image }}
run: | run: |
set -euo pipefail set -euo pipefail
@@ -81,7 +87,7 @@ jobs:
docker run --rm \ docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \ -v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-e RENOVATE_PLATFORM=gitea \ -e RENOVATE_PLATFORM=gitea \
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \ -e RENOVATE_ENDPOINT=https://git.forust.xyz/api/v1 \
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \ -e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \ -e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \ -e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
@@ -89,4 +95,4 @@ jobs:
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \ -e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
-e RENOVATE_BASE_DIR=/tmp/renovate \ -e RENOVATE_BASE_DIR=/tmp/renovate \
-e LOG_LEVEL="${LOG_LEVEL:-info}" \ -e LOG_LEVEL="${LOG_LEVEL:-info}" \
"${{ steps.image.outputs.image }}" "$RENOVATE_IMAGE"
+13
View File
@@ -0,0 +1,13 @@
# kubectl emits a List for files containing multiple resources.
(if .kind == "List" then .items[] else . end)
| (.metadata.namespace // "default") as $ns
| [
(.. | objects
| (.secretRef? // empty), (.secretKeyRef? // empty), (.secret? // empty)
| select(.optional != true)
| .name // .secretName // empty),
(.. | objects | .imagePullSecrets[]?.name)
]
| unique[]
| select(. != null and . != "")
| "\($ns) \(.)"
+64 -24
View File
@@ -1,30 +1,70 @@
#!/usr/bin/env bash #!/usr/bin/env bash
# usage: ssh-run.sh <stage> # The SSH client submits once and follows durable stages on workstation.
# Runs one deploy-lib.sh stage on the workstation over SSH.
set -euo pipefail set -euo pipefail
: "${DEPLOY_HOST:?missing DEPLOY_HOST}" : "${DEPLOY_HOST:?missing DEPLOY_HOST}"
: "${DEPLOY_USER:?missing DEPLOY_USER}" : "${DEPLOY_USER:?missing DEPLOY_USER}"
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}" : "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
: "${DEPLOY_KNOWN_HOSTS:?configure pinned DEPLOY_KNOWN_HOSTS}"
deploy_port="${DEPLOY_PORT:-22}" : "${DEPLOY_RUN_ID:?missing DEPLOY_RUN_ID}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}" [[ "$DEPLOY_USER" =~ ^[A-Za-z_][A-Za-z0-9_.-]*$ ]] || exit 1
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)" [[ "$DEPLOY_HOST" =~ ^[A-Za-z0-9_.:-]+$ ]] || exit 1
[[ "$DEPLOY_RUN_ID" =~ ^[0-9]+-[0-9]+$ ]] || exit 1
# The private key is written to a per-run directory that is removed on exit, so a [[ "${DEPLOY_PORT:-22}" =~ ^[0-9]+$ ]] || exit 1
# failed or cancelled job cannot leave deploy credentials in the runner's temp
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")" key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
trap 'rm -rf "$key_dir"' EXIT INT TERM trap 'rm -rf "$key_dir"' EXIT
chmod 700 "$key_dir"
ssh_key="$key_dir/deploy_key" printf '%s\n' "$DEPLOY_KEY" >"$key_dir/key"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key" printf '%s\n' "$DEPLOY_KNOWN_HOSTS" >"$key_dir/known_hosts"
chmod 600 "$ssh_key" chmod 600 "$key_dir/key" "$key_dir/known_hosts"
ssh_opts=(-i "$key_dir/key" -p "${DEPLOY_PORT:-22}" -o BatchMode=yes -o StrictHostKeyChecking=yes
ssh -i "$ssh_key" -p "$deploy_port" \ -o "UserKnownHostsFile=$key_dir/known_hosts" -o ConnectTimeout=15
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \ -o ServerAliveInterval=15 -o ServerAliveCountMax=4)
"${DEPLOY_USER}@${DEPLOY_HOST}" \ controller=.local/lib/homelab-deploy/controller.py
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-} STAGE=$1 bash -se" <<'EOF' case "${1:?start, apply, verify, smoke or summary required}" in
source "$REPO/.gitea/workflows/deploy-lib.sh" start)
run_stage "$STAGE" python3 - <<'PY' >"$key_dir/request.json"
EOF import json
import os
from pathlib import Path
release = json.loads(Path('release.json').read_text())
print(json.dumps({'release': release, 'mode': os.environ.get('DEPLOY_MODE', 'changed'),
'refresh_images': os.environ.get('REFRESH_IMAGES', 'false') == 'true'}))
PY
for attempt in 1 2 3; do
rc=0
# shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables.
ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" start "$DEPLOY_RUN_ID" <"$key_dir/request.json" || rc=$?
[ "$rc" -eq 0 ] && exit 0
[ "$rc" -eq 255 ] || exit "$rc"
sleep 5
done
exit "$rc"
;;
apply|verify|smoke)
result=0
for attempt in 1 2 3; do
rc=0
# shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables.
ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$?
[ "$rc" -eq 0 ] && break
[ "$rc" -eq 255 ] || { result="$rc"; break; }
echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)"
if [ "$attempt" -eq 3 ]; then result=255; break; fi
sleep 5
done
exit "$result"
;;
summary)
if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then
rc=0
# shellcheck disable=SC2029 # The run ID is validated above.
ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" summary "$DEPLOY_RUN_ID" >"$key_dir/deploy-summary.md" || rc=$?
if [ "$rc" -eq 0 ]; then
cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" || echo "WARNING: cannot write the deploy summary"
else
echo 'Deploy summary is unavailable. The SSH connection failed or the controller did not respond. Check the job log.' >>"$GITHUB_STEP_SUMMARY" || true
fi
fi
;;
*) echo "Unknown SSH operation: $1" >&2; exit 1 ;;
esac
+6
View File
@@ -31,3 +31,9 @@ UV_VERSION="0.12.17"
# so the tree that gets tested is the tree that gets built. Renovate keeps this # so the tree that gets tested is the tree that gets built. Renovate keeps this
# in step with the Dockerfile's node: tag via the "node runtime" group. # in step with the Dockerfile's node: tag via the "node runtime" group.
NODE_VERSION="22.23.3" NODE_VERSION="22.23.3"
# Secret-reference regression tests parse rendered Kubernetes objects.
JQ_VERSION="1.8.1"
# BuildKit is the only auxiliary CI container; jobs themselves stay on the host.
BUILDKIT_IMAGE="moby/buildkit:v0.33.1"
-1
View File
@@ -94,7 +94,6 @@ replacements.txt
.idea .idea
# Temp files # Temp files
edu_master/temp/
temp/* temp/*
# Local-only tooling scratch space (pinned CI tools, verification scripts) # Local-only tooling scratch space (pinned CI tools, verification scripts)
tmp/ tmp/
+1 -1
View File
@@ -31,7 +31,7 @@ services:
- "traefik.http.routers.adguard-dev.entrypoints=websecure" - "traefik.http.routers.adguard-dev.entrypoints=websecure"
- "traefik.http.routers.adguard-dev.tls=true" - "traefik.http.routers.adguard-dev.tls=true"
# DoH Router # DoH Router
- "traefik.http.routers.dns-over-https.rule=(Host(`dns.forust.xyz` || Host(`adguard.forust.xyz`)) && PathPrefix(`/dns-query`))" - "traefik.http.routers.dns-over-https.rule=(Host(`dns.forust.xyz`) || Host(`adguard.forust.xyz`)) && PathPrefix(`/dns-query`)"
- "traefik.http.routers.dns-over-https.entrypoints=websecure" - "traefik.http.routers.dns-over-https.entrypoints=websecure"
- "traefik.http.routers.dns-over-https.tls.certresolver=letsencrypt" - "traefik.http.routers.dns-over-https.tls.certresolver=letsencrypt"
+12 -3
View File
@@ -51,6 +51,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: adguard-deployment name: adguard-deployment
namespace: adguard namespace: adguard
spec: spec:
@@ -58,12 +60,12 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: adguard app: adguard
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
app: adguard app: adguard
annotations:
reloader.stakater.com/auto: "true"
spec: spec:
containers: containers:
- name: adguard - name: adguard
@@ -73,7 +75,7 @@ spec:
memory: "1.5Gi" memory: "1.5Gi"
cpu: "300m" cpu: "300m"
requests: requests:
memory: "500Mi" memory: "512Mi"
cpu: "50m" cpu: "50m"
ports: ports:
- containerPort: 3000 - containerPort: 3000
@@ -82,6 +84,13 @@ spec:
name: dns name: dns
- containerPort: 853 - containerPort: 853
name: dot name: dot
readinessProbe:
tcpSocket:
port: dns
initialDelaySeconds: 5
periodSeconds: 5
successThreshold: 1
failureThreshold: 3
volumeMounts: volumeMounts:
- name: adguard-data - name: adguard-data
mountPath: /opt/adguardhome/work mountPath: /opt/adguardhome/work
-3
View File
@@ -9,9 +9,6 @@ spec:
routes: routes:
- match: Host(`dns.forust.xyz`) - match: Host(`dns.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: adguard-service - name: adguard-service
port: 3000 port: 3000
+13 -5
View File
@@ -27,6 +27,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: authentik-server-deployment name: authentik-server-deployment
namespace: authentik namespace: authentik
spec: spec:
@@ -34,6 +36,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: authentik-server app: authentik-server
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -52,8 +56,8 @@ spec:
- containerPort: 9000 - containerPort: 9000
resources: resources:
requests: requests:
memory: "700Mi" memory: "768Mi"
cpu: "300m" cpu: "100m"
limits: limits:
memory: "1.5Gi" memory: "1.5Gi"
cpu: "1000m" cpu: "1000m"
@@ -61,6 +65,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: authentik-worker-deployment name: authentik-worker-deployment
namespace: authentik namespace: authentik
spec: spec:
@@ -68,6 +74,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: authentik-worker app: authentik-worker
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -86,8 +94,8 @@ spec:
name: authentik-secrets name: authentik-secrets
resources: resources:
requests: requests:
memory: "512Mi" memory: "320Mi"
cpu: "300m" cpu: "100m"
limits: limits:
memory: "1Gi" memory: "768Mi"
cpu: "700m" cpu: "700m"
-3
View File
@@ -9,9 +9,6 @@ spec:
routes: routes:
- match: Host(`auth.forust.xyz`) - match: Host(`auth.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: authentik-server-service - name: authentik-server-service
port: 9000 port: 9000
+6 -2
View File
@@ -1,6 +1,8 @@
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: cfddns name: cfddns
labels: labels:
app: cfddns app: cfddns
@@ -9,6 +11,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: cfddns app: cfddns
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -22,10 +26,10 @@ spec:
imagePullPolicy: Always imagePullPolicy: Always
resources: resources:
requests: requests:
memory: "20Mi" memory: "32Mi"
cpu: "30m" cpu: "30m"
limits: limits:
memory: "64Mi" memory: "128Mi"
cpu: "50m" cpu: "50m"
envFrom: envFrom:
- secretRef: - secretRef:
+4
View File
@@ -17,6 +17,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: checkmk-deployment name: checkmk-deployment
namespace: checkmk namespace: checkmk
spec: spec:
@@ -24,6 +26,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: checkmk app: checkmk
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
-3
View File
@@ -9,9 +9,6 @@ spec:
routes: routes:
- match: Host(`cmk.forust.xyz`) - match: Host(`cmk.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: checkmk-service - name: checkmk-service
port: 5000 port: 5000
+7 -3
View File
@@ -1,6 +1,8 @@
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: cloudflared name: cloudflared
labels: labels:
app: cloudflared app: cloudflared
@@ -9,6 +11,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: cloudflared app: cloudflared
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -16,7 +20,7 @@ spec:
spec: spec:
containers: containers:
- name: cloudflared - name: cloudflared
image: cloudflare/cloudflared:2026.9.3 image: cloudflare/cloudflared:2026.10.0
imagePullPolicy: IfNotPresent imagePullPolicy: IfNotPresent
args: args:
- tunnel - tunnel
@@ -30,8 +34,8 @@ spec:
key: TUNNEL_TOKEN key: TUNNEL_TOKEN
resources: resources:
requests: requests:
memory: "32Mi" memory: "128Mi"
cpu: "30m" cpu: "30m"
limits: limits:
memory: "128Mi" memory: "256Mi"
cpu: "200m" cpu: "200m"
+1 -1
View File
@@ -1,7 +1,7 @@
services: services:
convertx: convertx:
container_name: convertx container_name: convertx
image: ghcr.io/c4illin/convertx:v0.18.0 image: ghcr.io/c4illin/convertx:v0.19.0
restart: unless-stopped restart: unless-stopped
ports: ports:
- "9992:3000" - "9992:3000"
+3 -2
View File
@@ -31,12 +31,13 @@ spec:
name: bentopdf name: bentopdf
ports: ports:
- containerPort: 8080 - containerPort: 8080
# p95 4M, max 11M over 7 days. Was 50Mi/700Mi.
resources: resources:
requests: requests:
memory: "50Mi" memory: "32Mi"
cpu: "50m" cpu: "50m"
ephemeral-storage: "100Mi" ephemeral-storage: "100Mi"
limits: limits:
memory: "700Mi" memory: "128Mi"
cpu: "700m" cpu: "700m"
ephemeral-storage: "5Gi" ephemeral-storage: "5Gi"
+8 -3
View File
@@ -13,6 +13,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: convertx-deployment name: convertx-deployment
namespace: converters namespace: converters
spec: spec:
@@ -20,13 +22,15 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: convertx app: convertx
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
app: convertx app: convertx
spec: spec:
containers: containers:
- image: ghcr.io/c4illin/convertx:v0.18.0 - image: ghcr.io/c4illin/convertx:v0.19.0
name: convertx name: convertx
envFrom: envFrom:
- configMapRef: - configMapRef:
@@ -38,13 +42,14 @@ spec:
volumeMounts: volumeMounts:
- mountPath: /data - mountPath: /data
name: data name: data
# p95 85M, max 136M over 7 days, spikes while converting. Was 250Mi/1.5Gi.
resources: resources:
requests: requests:
memory: "250Mi" memory: "128Mi"
cpu: "100m" cpu: "100m"
limits: limits:
cpu: "1500m" cpu: "1500m"
memory: "1.5Gi" memory: "512Mi"
volumes: volumes:
- name: data - name: data
persistentVolumeClaim: persistentVolumeClaim:
-24
View File
@@ -1,24 +0,0 @@
apiVersion: traefik.io/v1alpha1
kind: Middleware
metadata:
name: crowdsec-bouncer
namespace: crowdsec
spec:
plugin:
crowdsec-bouncer:
enabled: true
LogLevel: INFO
CrowdsecMode: live
CrowdsecLapiScheme: http
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
# LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on
# `GET /v1/decisions?ip=...&banned=true` before the request reaches
# the backend, and fails CLOSED (403) if the lookup exceeds the
# timeout. Unset, the fork defaults to 10s, which is an eternity for
# a request path: a single slow LAPI (idle 1.3-7.4s here) turned
# every request into a 10s hang and then a self-inflicted 403.
# 2s keeps the fail-closed path fast and bounded; with the LAPI
# resourced properly (see crowdsec-values.yaml) the lookup is
# sub-100ms and this budget is never hit.
CrowdsecLapiTimeout: "2s"
+97 -20
View File
@@ -16,6 +16,12 @@ agent:
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
- name: DISABLE_COLLECTIONS - name: DISABLE_COLLECTIONS
value: crowdsecurity/sshd value: crowdsecurity/sshd
# Bans on 401/403 bursts hurt more than they protect: with L3 enforcement
# a false positive cuts the IP off everything (SSH included), and past
# incidents show legit automation (deploy runner, mesh peers, registry
# pulls) tripping this probe. Probing/XSS/SQLi/CVE scenarios stay.
- name: DISABLE_SCENARIOS
value: crowdsecurity/http-generic-bf
metrics: metrics:
enabled: true enabled: true
serviceMonitor: serviceMonitor:
@@ -58,26 +64,15 @@ config:
reason: "Mobile IP whitelist" reason: "Mobile IP whitelist"
cidr: cidr:
- "84.245.64.0/18" - "84.245.64.0/18"
# CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage.
postoverflows: # A parser whitelist discards the event before it reaches a bucket, so
s01-whitelist: # these addresses never produce an overflow and never become a decision.
home-dynamic-ip.yaml: | # A postoverflow whitelist is checked only *after* the ban exists, and
name: forust/home-dynamic-ip # the bouncer answers 403 for as long as it does - which is a window we
description: "Whitelist home dynamic IP" # do not want the deploy sitting in.
whitelist: local-network.yaml: |
reason: "Home dynamic IP" name: forust/local-network
expression: description: "Whitelist loopback, private and VPN networks"
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
# The hairpin-NAT address of the router (192.168.88.1) is what the
# Gitea Actions runner presents to Traefik - it is NOT the home
# dynamic IP, so the whitelist above did not cover it. During a
# deploy the runner POSTs to the Actions API many times a second;
# a single 403 storm was enough to earn it a 4h ban and break every
# later job. Whitelisting the whole LAN also covers phones and
# tablets browsing over 192.168.88.0/24.
lan.yaml: |
name: forust/lan
description: "Whitelist local network"
whitelist: whitelist:
reason: "Local network" reason: "Local network"
cidr: cidr:
@@ -85,6 +80,88 @@ config:
- "10.0.0.0/8" - "10.0.0.0/8"
- "172.16.0.0/12" - "172.16.0.0/12"
- "192.168.0.0/16" - "192.168.0.0/16"
# CGNAT range (RFC 6598). The workstation and the k0s node live
# here on WireGuard, and 100.64.0.0/10 is not covered by the
# RFC 1918 blocks above.
- "100.64.0.0/10"
- "169.254.0.0/16"
- "fc00::/7"
- "fe80::/10"
vps-whitelist.yaml: |
name: forust/vps-whitelist
description: "Whitelist static VPS"
whitelist:
reason: "VPS"
ip:
- "193.181.211.79"
postoverflows:
s01-whitelist:
# The one whitelist that has to stay here: resolving a hostname is a
# network call, and the docs put expensive lookups in postoverflows on
# purpose - it runs only when a bucket actually overflows.
# ddns.forust.xyz is the public home address, not a private one, so
# forust/local-network does not cover it.
home-dynamic-ip.yaml: |
name: forust/home-dynamic-ip
description: "Whitelist home dynamic IP"
whitelist:
reason: "Home dynamic IP"
expression:
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
# LAPI-only main config override, merged over config.yaml. NOTE: the
# chart's own default for this key is REPLACED, not merged, so its
# auto_registration block is repeated verbatim below - drop it and the
# agent can no longer register itself.
config.yaml.local: |
api:
server:
auto_registration: # Activate if not using TLS for authentication
enabled: true
token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart)
allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster
- "127.0.0.1/32"
- "192.168.0.0/16"
- "10.0.0.0/8"
- "172.16.0.0/12"
# This homelab has no egress to console.crowdsec.cloud: DNS does
# not resolve. The LAPI kept trying anyway ("Signal push: N
# signals to push", "capi metrics: sending" every 10s) and each
# attempt sat on a resolver timeout WHILE HOLDING A WRITE
# TRANSACTION, which is what kept stalling per-request decision
# lookups even with WAL enabled. Nothing to share and nothing to
# pull - turn the Central API off instead of letting it block the
# only database writer we have.
online_client:
sharing: false
pull:
community: false
blocklists: false
disable_usage_metrics_export: true
db_config:
# SQLite without WAL serialises every reader behind the writer's
# rollback journal, and the LAPI writes constantly: the agent pushes
# Traefik alerts read from Loki, the metrics collector counts
# decisions, the bouncer touches "last pull" on every request.
# Symptom: decision lookups taking 10-30s (and a second connection
# that could not even open the database) while the LAPI sat at 28m
# CPU - the process was blocked in fsync, not computing. Every
# bouncer-protected request then blew through the plugin timeout and
# fail-closed with 403, on every site at once.
# The PVC is local-path-retain (hostPath), not a network share, so
# WAL is safe here; the crowdsec docs recommend it for exactly this
# ("allowing more concurrency in SQLite that will improve
# performances in most scenarios").
use_wal: true
# Keeps the alert table bounded. At the 5000/7d default the file
# reached 54MB in 15 days off the Traefik access log alone, and the
# metrics collector counts decisions on a timer; a smaller working
# set means fewer full scans. Crowdsec only prunes - SQLite never
# shrinks the file, so the size stays until a manual VACUUM.
flush:
max_items: 1000
max_age: 24h
lapi: lapi:
env: env:
+8 -20
View File
@@ -30,10 +30,14 @@
# 3. ensure the static machine exists, recreating it with the # 3. ensure the static machine exists, recreating it with the
# Secret password if missing (agent retry loops reconnect # Secret password if missing (agent retry loops reconnect
# on their own - same name + same password); # on their own - same name + same password);
# 4. prune bouncer entries idle for 30d; # 4. prune bouncer entries idle for 30d.
# 5. delete decisions from LePresidente/http-generic-403-bf, a hub #
# scenario that bans an IP for 4h after 5 POST-403s in 10s and # It used to also delete LePresidente/http-generic-403-bf decisions hourly.
# therefore bans us for our own bouncer's fail-closed 403s. # That was a workaround for the bouncer failing closed on a slow LAPI and
# 403-ing the deploy runner into a 4h ban. The bouncer now polls decisions
# into a cache and never blocks on an unreachable LAPI, so it cannot
# manufacture those 403s any more, and the scenario only fires against real
# scanners - deleting their decisions hourly was undoing a working ban.
# #
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml): # Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml # kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
@@ -196,19 +200,3 @@ spec:
fi fi
echo "== 4. prune stale bouncers (no pull for 30d) ==" echo "== 4. prune stale bouncers (no pull for 30d) =="
$LAPI_EXEC cscli bouncers prune -d 720h --force $LAPI_EXEC cscli bouncers prune -d 720h --force
echo "== 5. drop http-403-bf decisions (4h self-bans) =="
# `LePresidente/http-generic-403-bf` (hub item
# crowdsecurity/http-generic-bf v0.9) bans any source IP
# after 5 POSTs answered 403 within 10s, for 4h. That
# includes 403s this homelab generates ITSELF (any
# bouncer fail-closed, any app CSRF/rate-limit 403), and a
# 4h ban on the runner/home IP silently breaks deploys and
# browsing. The scenario cannot be removed per-scenario -
# it is baked into a hub item, and disabling the whole
# base-http-scenarios collection would drop ~40 useful
# detections. Instead we keep the detection and drop its
# decisions hourly; the LAN/home whitelists in
# crowdsec-values.yaml handle the legit sources, so this
# only ever hits real scanners (who are re-banned anyway).
$LAPI_EXEC cscli decisions delete \
--scenario LePresidente/http-generic-403-bf --all || true
-2
View File
@@ -18,8 +18,6 @@ spec:
- match: Host(`dockmon.forust.xyz`) - match: Host(`dockmon.forust.xyz`)
kind: Rule kind: Rule
middlewares: middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
- name: security-headers@file - name: security-headers@file
services: services:
- name: dockmon-service - name: dockmon-service
+1 -1
View File
@@ -1,7 +1,7 @@
services: services:
downtify: downtify:
container_name: downtify container_name: downtify
image: ghcr.io/henriquesebastiao/downtify:3.1.0 image: ghcr.io/henriquesebastiao/downtify:3.4.0
restart: unless-stopped restart: unless-stopped
# ports: # ports:
# - '7077:8000' # - '7077:8000'
+3 -1
View File
@@ -20,6 +20,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: downtify app: downtify
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -27,7 +29,7 @@ spec:
spec: spec:
containers: containers:
- name: downtify - name: downtify
image: ghcr.io/henriquesebastiao/downtify:3.1.0 image: ghcr.io/henriquesebastiao/downtify:3.4.0
ports: ports:
- containerPort: 8000 - containerPort: 8000
volumeMounts: volumeMounts:
-2
View File
@@ -10,8 +10,6 @@ spec:
- match: Host(`downtify.forust.xyz`) - match: Host(`downtify.forust.xyz`)
kind: Rule kind: Rule
middlewares: middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
- name: security-chain@file - name: security-chain@file
services: services:
- name: downtify-service - name: downtify-service
-13
View File
@@ -1,13 +0,0 @@
FROM python:3.9-alpine
WORKDIR /app
# Установка зависимостей
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
# Копирование кода
COPY main.py .
COPY .env .
# Запуск бота
CMD ["python", "-u", "main.py"]
-373
View File
@@ -1,373 +0,0 @@
Mozilla Public License Version 2.0
==================================
1. Definitions
--------------
1.1. "Contributor"
means each individual or legal entity that creates, contributes to
the creation of, or owns Covered Software.
1.2. "Contributor Version"
means the combination of the Contributions of others (if any) used
by a Contributor and that particular Contributor's Contribution.
1.3. "Contribution"
means Covered Software of a particular Contributor.
1.4. "Covered Software"
means Source Code Form to which the initial Contributor has attached
the notice in Exhibit A, the Executable Form of such Source Code
Form, and Modifications of such Source Code Form, in each case
including portions thereof.
1.5. "Incompatible With Secondary Licenses"
means
(a) that the initial Contributor has attached the notice described
in Exhibit B to the Covered Software; or
(b) that the Covered Software was made available under the terms of
version 1.1 or earlier of the License, but not also under the
terms of a Secondary License.
1.6. "Executable Form"
means any form of the work other than Source Code Form.
1.7. "Larger Work"
means a work that combines Covered Software with other material, in
a separate file or files, that is not Covered Software.
1.8. "License"
means this document.
1.9. "Licensable"
means having the right to grant, to the maximum extent possible,
whether at the time of the initial grant or subsequently, any and
all of the rights conveyed by this License.
1.10. "Modifications"
means any of the following:
(a) any file in Source Code Form that results from an addition to,
deletion from, or modification of the contents of Covered
Software; or
(b) any new file in Source Code Form that contains any Covered
Software.
1.11. "Patent Claims" of a Contributor
means any patent claim(s), including without limitation, method,
process, and apparatus claims, in any patent Licensable by such
Contributor that would be infringed, but for the grant of the
License, by the making, using, selling, offering for sale, having
made, import, or transfer of either its Contributions or its
Contributor Version.
1.12. "Secondary License"
means either the GNU General Public License, Version 2.0, the GNU
Lesser General Public License, Version 2.1, the GNU Affero General
Public License, Version 3.0, or any later versions of those
licenses.
1.13. "Source Code Form"
means the form of the work preferred for making modifications.
1.14. "You" (or "Your")
means an individual or a legal entity exercising rights under this
License. For legal entities, "You" includes any entity that
controls, is controlled by, or is under common control with You. For
purposes of this definition, "control" means (a) the power, direct
or indirect, to cause the direction or management of such entity,
whether by contract or otherwise, or (b) ownership of more than
fifty percent (50%) of the outstanding shares or beneficial
ownership of such entity.
2. License Grants and Conditions
--------------------------------
2.1. Grants
Each Contributor hereby grants You a world-wide, royalty-free,
non-exclusive license:
(a) under intellectual property rights (other than patent or trademark)
Licensable by such Contributor to use, reproduce, make available,
modify, display, perform, distribute, and otherwise exploit its
Contributions, either on an unmodified basis, with Modifications, or
as part of a Larger Work; and
(b) under Patent Claims of such Contributor to make, use, sell, offer
for sale, have made, import, and otherwise transfer either its
Contributions or its Contributor Version.
2.2. Effective Date
The licenses granted in Section 2.1 with respect to any Contribution
become effective for each Contribution on the date the Contributor first
distributes such Contribution.
2.3. Limitations on Grant Scope
The licenses granted in this Section 2 are the only rights granted under
this License. No additional rights or licenses will be implied from the
distribution or licensing of Covered Software under this License.
Notwithstanding Section 2.1(b) above, no patent license is granted by a
Contributor:
(a) for any code that a Contributor has removed from Covered Software;
or
(b) for infringements caused by: (i) Your and any other third party's
modifications of Covered Software, or (ii) the combination of its
Contributions with other software (except as part of its Contributor
Version); or
(c) under Patent Claims infringed by Covered Software in the absence of
its Contributions.
This License does not grant any rights in the trademarks, service marks,
or logos of any Contributor (except as may be necessary to comply with
the notice requirements in Section 3.4).
2.4. Subsequent Licenses
No Contributor makes additional grants as a result of Your choice to
distribute the Covered Software under a subsequent version of this
License (see Section 10.2) or under the terms of a Secondary License (if
permitted under the terms of Section 3.3).
2.5. Representation
Each Contributor represents that the Contributor believes its
Contributions are its original creation(s) or it has sufficient rights
to grant the rights to its Contributions conveyed by this License.
2.6. Fair Use
This License is not intended to limit any rights You have under
applicable copyright doctrines of fair use, fair dealing, or other
equivalents.
2.7. Conditions
Sections 3.1, 3.2, 3.3, and 3.4 are conditions of the licenses granted
in Section 2.1.
3. Responsibilities
-------------------
3.1. Distribution of Source Form
All distribution of Covered Software in Source Code Form, including any
Modifications that You create or to which You contribute, must be under
the terms of this License. You must inform recipients that the Source
Code Form of the Covered Software is governed by the terms of this
License, and how they can obtain a copy of this License. You may not
attempt to alter or restrict the recipients' rights in the Source Code
Form.
3.2. Distribution of Executable Form
If You distribute Covered Software in Executable Form then:
(a) such Covered Software must also be made available in Source Code
Form, as described in Section 3.1, and You must inform recipients of
the Executable Form how they can obtain a copy of such Source Code
Form by reasonable means in a timely manner, at a charge no more
than the cost of distribution to the recipient; and
(b) You may distribute such Executable Form under the terms of this
License, or sublicense it under different terms, provided that the
license for the Executable Form does not attempt to limit or alter
the recipients' rights in the Source Code Form under this License.
3.3. Distribution of a Larger Work
You may create and distribute a Larger Work under terms of Your choice,
provided that You also comply with the requirements of this License for
the Covered Software. If the Larger Work is a combination of Covered
Software with a work governed by one or more Secondary Licenses, and the
Covered Software is not Incompatible With Secondary Licenses, this
License permits You to additionally distribute such Covered Software
under the terms of such Secondary License(s), so that the recipient of
the Larger Work may, at their option, further distribute the Covered
Software under the terms of either this License or such Secondary
License(s).
3.4. Notices
You may not remove or alter the substance of any license notices
(including copyright notices, patent notices, disclaimers of warranty,
or limitations of liability) contained within the Source Code Form of
the Covered Software, except that You may alter any license notices to
the extent required to remedy known factual inaccuracies.
3.5. Application of Additional Terms
You may choose to offer, and to charge a fee for, warranty, support,
indemnity or liability obligations to one or more recipients of Covered
Software. However, You may do so only on Your own behalf, and not on
behalf of any Contributor. You must make it absolutely clear that any
such warranty, support, indemnity, or liability obligation is offered by
You alone, and You hereby agree to indemnify every Contributor for any
liability incurred by such Contributor as a result of warranty, support,
indemnity or liability terms You offer. You may include additional
disclaimers of warranty and limitations of liability specific to any
jurisdiction.
4. Inability to Comply Due to Statute or Regulation
---------------------------------------------------
If it is impossible for You to comply with any of the terms of this
License with respect to some or all of the Covered Software due to
statute, judicial order, or regulation then You must: (a) comply with
the terms of this License to the maximum extent possible; and (b)
describe the limitations and the code they affect. Such description must
be placed in a text file included with all distributions of the Covered
Software under this License. Except to the extent prohibited by statute
or regulation, such description must be sufficiently detailed for a
recipient of ordinary skill to be able to understand it.
5. Termination
--------------
5.1. The rights granted under this License will terminate automatically
if You fail to comply with any of its terms. However, if You become
compliant, then the rights granted under this License from a particular
Contributor are reinstated (a) provisionally, unless and until such
Contributor explicitly and finally terminates Your grants, and (b) on an
ongoing basis, if such Contributor fails to notify You of the
non-compliance by some reasonable means prior to 60 days after You have
come back into compliance. Moreover, Your grants from a particular
Contributor are reinstated on an ongoing basis if such Contributor
notifies You of the non-compliance by some reasonable means, this is the
first time You have received notice of non-compliance with this License
from such Contributor, and You become compliant prior to 30 days after
Your receipt of the notice.
5.2. If You initiate litigation against any entity by asserting a patent
infringement claim (excluding declaratory judgment actions,
counter-claims, and cross-claims) alleging that a Contributor Version
directly or indirectly infringes any patent, then the rights granted to
You by any and all Contributors for the Covered Software under Section
2.1 of this License shall terminate.
5.3. In the event of termination under Sections 5.1 or 5.2 above, all
end user license agreements (excluding distributors and resellers) which
have been validly granted by You or Your distributors under this License
prior to termination shall survive termination.
************************************************************************
* *
* 6. Disclaimer of Warranty *
* ------------------------- *
* *
* Covered Software is provided under this License on an "as is" *
* basis, without warranty of any kind, either expressed, implied, or *
* statutory, including, without limitation, warranties that the *
* Covered Software is free of defects, merchantable, fit for a *
* particular purpose or non-infringing. The entire risk as to the *
* quality and performance of the Covered Software is with You. *
* Should any Covered Software prove defective in any respect, You *
* (not any Contributor) assume the cost of any necessary servicing, *
* repair, or correction. This disclaimer of warranty constitutes an *
* essential part of this License. No use of any Covered Software is *
* authorized under this License except under this disclaimer. *
* *
************************************************************************
************************************************************************
* *
* 7. Limitation of Liability *
* -------------------------- *
* *
* Under no circumstances and under no legal theory, whether tort *
* (including negligence), contract, or otherwise, shall any *
* Contributor, or anyone who distributes Covered Software as *
* permitted above, be liable to You for any direct, indirect, *
* special, incidental, or consequential damages of any character *
* including, without limitation, damages for lost profits, loss of *
* goodwill, work stoppage, computer failure or malfunction, or any *
* and all other commercial damages or losses, even if such party *
* shall have been informed of the possibility of such damages. This *
* limitation of liability shall not apply to liability for death or *
* personal injury resulting from such party's negligence to the *
* extent applicable law prohibits such limitation. Some *
* jurisdictions do not allow the exclusion or limitation of *
* incidental or consequential damages, so this exclusion and *
* limitation may not apply to You. *
* *
************************************************************************
8. Litigation
-------------
Any litigation relating to this License may be brought only in the
courts of a jurisdiction where the defendant maintains its principal
place of business and such litigation shall be governed by laws of that
jurisdiction, without reference to its conflict-of-law provisions.
Nothing in this Section shall prevent a party's ability to bring
cross-claims or counter-claims.
9. Miscellaneous
----------------
This License represents the complete agreement concerning the subject
matter hereof. If any provision of this License is held to be
unenforceable, such provision shall be reformed only to the extent
necessary to make it enforceable. Any law or regulation which provides
that the language of a contract shall be construed against the drafter
shall not be used to construe this License against a Contributor.
10. Versions of the License
---------------------------
10.1. New Versions
Mozilla Foundation is the license steward. Except as provided in Section
10.3, no one other than the license steward has the right to modify or
publish new versions of this License. Each version will be given a
distinguishing version number.
10.2. Effect of New Versions
You may distribute the Covered Software under the terms of the version
of the License under which You originally received the Covered Software,
or under the terms of any subsequent version published by the license
steward.
10.3. Modified Versions
If you create software not governed by this License, and you want to
create a new license for such software, you may create and use a
modified version of this License if you rename the license and remove
any references to the name of the license steward (except to note that
such modified license differs from this License).
10.4. Distributing Source Code Form that is Incompatible With Secondary
Licenses
If You choose to distribute Source Code Form that is Incompatible With
Secondary Licenses under the terms of this version of the License, the
notice described in Exhibit B of this License must be attached.
Exhibit A - Source Code Form License Notice
-------------------------------------------
This Source Code Form is subject to the terms of the Mozilla Public
License, v. 2.0. If a copy of the MPL was not distributed with this
file, You can obtain one at https://mozilla.org/MPL/2.0/.
If it is not possible or desirable to put the notice in a particular
file, then You may include the notice in a location (such as a LICENSE
file in a relevant directory) where a recipient would be likely to look
for such a notice.
You may add additional accurate notices of copyright ownership.
Exhibit B - "Incompatible With Secondary Licenses" Notice
---------------------------------------------------------
This Source Code Form is "Incompatible With Secondary Licenses", as
defined by the Mozilla Public License, v. 2.0.
-15
View File
@@ -1,15 +0,0 @@
services:
dtek_notif:
build:
context: .
dockerfile: Dockerfile
image: gcr.forust.xyz/forust/dtek-notif:latest
pull_policy: build
restart: unless-stopped
environment:
- TZ=Europe/Kyiv
dns:
- 1.1.1.1
- 8.8.8.8
networks:
- default
-748
View File
@@ -1,748 +0,0 @@
import asyncio
import contextlib
import logging
import os
from datetime import datetime, timedelta
import requests
from aiogram import Bot, Dispatcher
from aiogram.filters import Command
from aiogram.types import KeyboardButton, Message
from aiogram.utils.keyboard import ReplyKeyboardBuilder
from bs4 import BeautifulSoup
from dotenv import load_dotenv
# Загрузка переменных окружения
load_dotenv()
# Настройки
TELEGRAM_TOKEN = os.getenv('TELEGRAM_TOKEN', 'YOUR_TOKEN_HERE')
ALLOWED_CHAT_IDS = list(map(int, os.getenv('ALLOWED_CHAT_IDS', '').split(','))) if os.getenv('ALLOWED_CHAT_IDS') else []
CHECK_INTERVAL = int(os.getenv('CHECK_INTERVAL', '120'))
# Параметры для запроса
VOE_CITY_ID = int(os.getenv('VOE_CITY_ID', 'VOE_CITY_ID'))
VOE_STREET_ID = int(os.getenv('VOE_STREET_ID', 'VOE_STREET_ID'))
VOE_HOUSE_ID = int(os.getenv('VOE_HOUSE_ID', 'VOE_HOUSE_ID'))
# Настройка логирования
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# Глобальные переменные
bot = Bot(token=TELEGRAM_TOKEN)
dp = Dispatcher()
last_schedule: list[dict] | None = None
last_notification_time: dict[str, datetime] = {}
# ============================================================================
# УТИЛИТЫ
# ============================================================================
def format_time_duration(minutes: int) -> str:
"""Форматирует время из минут в часы и минуты"""
hours = minutes // 60
mins = minutes % 60
if hours == 0:
return f'{mins}м'
elif mins == 0:
return f'{hours}ч'
return f'{hours}ч {mins}м'
def get_day_statistics(day_blocks: list[dict]) -> dict[str, int]:
"""Получает статистику по дню"""
total_minutes = 0
confirmed_minutes = 0
possible_minutes = 0
for block in day_blocks:
for half in [block['first_half'], block['second_half']]:
if half['status'] == 'off':
total_minutes += 30
if half['confirmed']:
confirmed_minutes += 30
else:
possible_minutes += 30
return {'total': total_minutes, 'confirmed': confirmed_minutes, 'possible': possible_minutes}
# ============================================================================
# ПАРСИНГ ДАННЫХ
# ============================================================================
def parse_html(html: str) -> list[dict]:
"""Парсит HTML с графиком отключений (логика от 15.11.2024)"""
soup = BeautifulSoup(html, 'html.parser')
cells = soup.select('.disconnection-detailed-table-cell.cell')
schedule = []
current_hour = 0
current_day = 0
for cell in cells:
if 'legend' in cell.get('class', []) or 'head' in cell.get('class', []):
continue
cell_classes = cell.get('class', [])
# ПРоверка статуса отключения на весь час
full_hour_off = 'has_disconnection' in cell_classes and 'full_hour' in cell_classes
hour_block = cell.select_one('.hour_block')
if not hour_block:
continue
# Проверка подтверждённости отключения для всего часа
cell_confirmed = None
if 'confirm_1' in cell_classes:
cell_confirmed = True
elif 'confirm_0' in cell_classes:
cell_confirmed = False
# Проверка половин часа
left = hour_block.select_one('.half.left')
right = hour_block.select_one('.half.right')
def parse_half(half, is_full_hour_off: bool, cell_confirmed: bool | None = None) -> dict:
"""Парсит половину часа"""
if not half:
return {'status': 'on', 'queue': None, 'confirmed': None}
half_classes = half.get('class', [])
# Если вся ячейка full_hour - используем статус ячейки
if is_full_hour_off:
return {'status': 'off', 'queue': None, 'confirmed': cell_confirmed}
# Определяем статус половины
if 'has_disconnection' in half_classes:
status = 'off'
elif 'no_disconnection' in half_classes:
status = 'on'
else:
status = 'on' # По умолчанию считаем включенным
# Если выключено - ищем подробности
queue = None
confirmed = None
if status == 'off':
disconnection_div = half.select_one('.disconnection')
if disconnection_div:
# Ищем номер черги в title
if disconnection_div.has_attr('title'):
title = disconnection_div['title']
if 'Номер черги' in title or 'Номер черги:' in title:
with contextlib.suppress(BaseException):
queue = title.split(':')[-1].strip()
# Определяем подтверждение
disc_classes = disconnection_div.get('class', [])
if 'disconnection_confirm_1' in disc_classes:
confirmed = True
elif 'disconnection_confirm_0' in disc_classes:
confirmed = False
return {'status': status, 'queue': queue, 'confirmed': confirmed}
first_half_data = parse_half(left, full_hour_off, cell_confirmed)
second_half_data = parse_half(right, full_hour_off, cell_confirmed)
schedule.append(
{
'hour': current_hour,
'day': current_day,
'first_half': first_half_data,
'second_half': second_half_data,
}
)
current_hour += 1
if current_hour >= 24:
current_hour = 0
current_day += 1
return schedule
def get_voe_html(city_id: int, street_id: int, house_id: int) -> str:
"""Получает HTML с сайта VOE"""
url = 'https://www.voe.com.ua/disconnection/detailed?ajax_form=1&_wrapper_format=drupal_ajax'
headers = {
'Content-Type': 'application/x-www-form-urlencoded; charset=UTF-8',
'X-Requested-With': 'XMLHttpRequest',
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
}
data = {
'search_type': 0,
'city_id': city_id,
'street_id': street_id,
'house_id': house_id,
'form_build_id': 'form-Irv5aHw1R2FT_Ik2apyHOZ47hTH5xPNH_LQnBrmpSTc',
'form_id': 'disconnection_detailed_search_form',
'_triggering_element_name': 'search',
'_triggering_element_value': 'Показати',
'_drupal_ajax': 1,
}
try:
response = requests.post(url, headers=headers, data=data, timeout=10)
response.raise_for_status()
resp_json = response.json()
insert_html = next((item['data'] for item in resp_json if item.get('command') == 'insert'), None)
if not insert_html:
raise ValueError('HTML не найден в ответе')
return insert_html
except requests.exceptions.RequestException as e:
logger.error(f'Ошибка запроса VOE: {e}')
raise
# ============================================================================
# ФОРМАТИРОВАНИЕ СООБЩЕНИЙ
# ============================================================================
def get_main_keyboard():
"""Создает главную клавиатуру"""
builder = ReplyKeyboardBuilder()
builder.row(KeyboardButton(text='📊 Графік'), KeyboardButton(text='🔄 Оновити'))
builder.row(KeyboardButton(text='📅 Сьогодні'), KeyboardButton(text='📅 Завтра'))
builder.row(KeyboardButton(text='ℹ️ Про бота'))
return builder.as_markup(resize_keyboard=True)
def format_schedule_message(schedule: list[dict], days_to_show: int = 2) -> str:
"""Форматирует полный график на несколько дней"""
lines = [
'⚡️ <b>Графік відключень світла</b>',
f'🕐 Оновлено: {datetime.now().strftime("%d.%m.%Y %H:%M:%S")}',
'─' * 30,
'',
]
start_date = datetime.now()
for day in range(min(days_to_show, 2)):
day_blocks = [b for b in schedule if b['day'] == day]
if not day_blocks:
continue
date_str = (start_date + timedelta(days=day)).strftime('%d.%m.%Y')
day_name = '🌅 <b>Сьогодні</b>' if day == 0 else '🌄 <b>Завтра</b>'
lines.append(f'{day_name} ({date_str})')
# Статистика
stats = get_day_statistics(day_blocks)
if stats['total'] > 0:
lines.append(f'⏱ Всього: <code>{format_time_duration(stats["total"])}</code>')
if stats['confirmed'] > 0:
lines.append(f'🔴 Підтверджено: <code>{format_time_duration(stats["confirmed"])}</code>')
if stats['possible'] > 0:
lines.append(f'🟠 Можливо: <code>{format_time_duration(stats["possible"])}</code>')
else:
lines.append('🟢 <b>Відключень немає!</b>')
lines.append('')
# Детальный список отключений
disconnections = []
current_status = None
start_time = None
current_confirmed = None
current_queue = None
for block in day_blocks:
hour = block['hour']
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
time_str = f'{hour:02d}:00' if half_idx == 0 else f'{hour:02d}:30'
if half['status'] == 'off':
if current_status != 'off':
start_time = time_str
current_confirmed = half['confirmed']
current_queue = half['queue']
current_status = 'off'
else:
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч{current_queue})' if current_queue else ''
disconnections.append(f'{icon} <code>{start_time} - {time_str}</code>{queue_text}')
current_status = half['status']
# Если день закончился на отключении
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч{current_queue})' if current_queue else ''
next_hour = (day_blocks[-1]['hour'] + 1) % 24
end_time = f'{next_hour:02d}:00'
disconnections.append(f'{icon} <code>{start_time} - {end_time}</code>{queue_text}')
if disconnections:
for idx, disc in enumerate(disconnections, 1):
lines.append(f'{idx}. {disc}')
lines.append('')
lines.append('<i>🔴 = підтверджено • 🟠 = можливо • 🟢 = світло</i>')
return '\n'.join(lines)
def format_single_day_schedule(schedule: list[dict], day: int) -> str:
"""Форматирует график на один день"""
day_blocks = [b for b in schedule if b['day'] == day]
if not day_blocks:
return '❌ Немає даних для цього дня'
start_date = datetime.now()
date_str = (start_date + timedelta(days=day)).strftime('%d.%m.%Y')
day_name = '🟠 <b>Сьогодні</b>' if day == 0 else '🔶 <b>Завтра</b>'
lines = [f'{day_name} • {date_str}', '']
# Статистика
lines.append('<b>📊 Статистика</b>')
stats = get_day_statistics(day_blocks)
if stats['total'] == 0:
lines.append('└ 🟢 <b>Відключень немає!</b>')
else:
total_time = format_time_duration(stats['total'])
lines.append(f'├ ⏱ Всього: <code>{total_time}</code>')
if stats['confirmed'] > 0:
confirmed_time = format_time_duration(stats['confirmed'])
lines.append(f'├ 🔴 Підтверджено: <code>{confirmed_time}</code>')
if stats['possible'] > 0:
possible_time = format_time_duration(stats['possible'])
lines.append(f'└ 🟠 Можливо: <code>{possible_time}</code>')
else:
lines.append('└ 🟢 Решта часу світло')
lines.append('')
# Детальный список отключений
disconnections = []
current_status = None
start_time = None
current_confirmed = None
current_queue = None
for block in day_blocks:
hour = block['hour']
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
time_str = f'{hour:02d}:00' if half_idx == 0 else f'{hour:02d}:30'
if half['status'] == 'off':
if current_status != 'off':
start_time = time_str
current_confirmed = half['confirmed']
current_queue = half['queue']
current_status = 'off'
else:
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч.{current_queue})' if current_queue else ''
disconnections.append(f'{icon} <code>{start_time} - {time_str}</code>{queue_text}')
current_status = half['status']
# Если день закончился на отключении
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч.{current_queue})' if current_queue else ''
next_hour = (day_blocks[-1]['hour'] + 1) % 24
end_time = f'{next_hour:02d}:00'
disconnections.append(f'{icon} <code>{start_time} - {end_time}</code>{queue_text}')
if disconnections:
lines.append('<b>⚡️ Розклад відключень</b>')
for idx, disc in enumerate(disconnections, 1):
lines.append(f'{idx}. {disc}')
lines.append('')
lines.append('<i>🔴 підтверджено • 🟠 можливо • 🟢 світло</i>')
return '\n'.join(lines)
def schedules_differ(old_schedule: list[dict] | None, new_schedule: list[dict] | None) -> bool:
"""Проверяет отличия между графиками"""
if old_schedule is None or new_schedule is None:
return True
if len(old_schedule) != len(new_schedule):
return True
for old, new in zip(old_schedule, new_schedule, strict=False):
if old['day'] >= 2:
break
if old['first_half'] != new['first_half'] or old['second_half'] != new['second_half']:
return True
return False
# ============================================================================
# УВЕДОМЛЕНИЯ
# ============================================================================
async def send_to_all_users(message_text: str, parse_mode: str = 'HTML'):
"""Отправляет сообщение всем пользователям"""
if not ALLOWED_CHAT_IDS:
logger.warning('Нет допущенных ID чатов для отправки уведомлений')
return
for chat_id in ALLOWED_CHAT_IDS:
try:
await bot.send_message(chat_id, message_text, parse_mode=parse_mode)
logger.info(f'✅ Сообщение отправлено пользователю {chat_id}')
except Exception as e:
logger.error(f'❌ Ошибка отправки пользователю {chat_id}: {e}')
await asyncio.sleep(0.5)
async def check_schedule():
"""Проверяет график и отправляет уведомления"""
global last_schedule
try:
logger.info('🔍 Проверка графика...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
new_schedule = parse_html(html)
if schedules_differ(last_schedule, new_schedule):
logger.info('✨ Обнаружены изменения!')
message = format_schedule_message(new_schedule, days_to_show=2)
if last_schedule is not None:
await send_to_all_users(f'🔄 <b>Графік оновлено!</b>\n\n{message}')
last_schedule = new_schedule
else:
logger.info('✓ Графік без змін')
except Exception as e:
logger.error(f'❌ Ошибка при проверке графика: {e}')
async def check_upcoming_disconnections():
"""Проверяет предстоящие события и отправляет предупреждения за 5 минут"""
global last_notification_time
if last_schedule is None:
return
now = datetime.now()
today_blocks = [b for b in last_schedule if b['day'] == 0]
# Создаем список всех переходов (off -> on или on -> off)
transitions = []
prev_status = None
for block in today_blocks:
hour = block['hour']
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
minute = 0 if half_idx == 0 else 30
time_str = f'{hour:02d}:{minute:02d}'
current_status = half['status']
# Если статус изменился - это переход
if prev_status is not None and prev_status != current_status:
transitions.append(
{
'hour': hour,
'minute': minute,
'time_str': time_str,
'from_status': prev_status,
'to_status': current_status,
'confirmed': half.get('confirmed'),
'queue': half.get('queue'),
}
)
prev_status = current_status
# Проверяем переходы
for transition in transitions:
event_time = now.replace(hour=transition['hour'], minute=transition['minute'], second=0, microsecond=0)
time_until = (event_time - now).total_seconds() / 60
notification_key = f'{transition["hour"]}:{transition["minute"]}_{transition["to_status"]}'
# Если за 5 минут до события (±1 минута) и еще не отправляли
if 4 <= time_until <= 6:
# Проверяем, не отправляли ли уже уведомление сегодня
if notification_key in last_notification_time:
last_notif_time = last_notification_time[notification_key]
if last_notif_time.date() == now.date():
continue # Уже отправляли сегодня
# Переход на ОТКЛЮЧЕНИЕ (on -> off)
if transition['from_status'] == 'on' and transition['to_status'] == 'off':
icon = '🔴' if transition['confirmed'] else '🟠'
status = 'підтверджено' if transition['confirmed'] else 'можливе'
queue_info = f' (Черга {transition["queue"]})' if transition['queue'] else ''
warning = (
f'⚠️ <b>УВАГА! ВІДКЛЮЧЕННЯ</b>\n\n'
f'Через ~5 хвилин\n'
f'Час: <code>{transition["time_str"]}</code>\n'
f'Статус: {icon} {status}{queue_info}'
)
await send_to_all_users(warning)
last_notification_time[notification_key] = now
logger.info(f'📢 Відправлено попередження про ВІДКЛЮЧЕННЯ в {transition["time_str"]}')
# Переход на ВКЛЮЧЕНИЕ (off -> on)
elif transition['from_status'] == 'off' and transition['to_status'] == 'on':
warning = (
f'✅ <b>УВАГА! ВКЛЮЧЕННЯ</b>\n\n'
f'Через ~5 хвилин буде світло\n'
f'Час: <code>{transition["time_str"]}</code>'
)
await send_to_all_users(warning)
last_notification_time[notification_key] = now
logger.info(f'📢 Відправлено попередження про ВКЛЮЧЕННЯ в {transition["time_str"]}')
async def monitoring_loop():
"""Основной цикл мониторинга"""
await check_schedule()
while True:
try:
await asyncio.sleep(CHECK_INTERVAL)
await check_schedule()
await check_upcoming_disconnections()
except Exception as e:
logger.error(f'Ошибка в цикле мониторинга: {e}')
await asyncio.sleep(5)
# ============================================================================
# ОБРАБОТЧИКИ КОМАНД
# ============================================================================
@dp.message(Command('start'))
async def cmd_start(message: Message):
"""Обработчик /start"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу до цього бота.')
return
await message.answer(
'👋 <b>Ласкаво просимо!</b>\n\n'
'🤖 <b>Бот для моніторингу графіку відключень світла</b>\n\n'
'✨ <b>Можливості:</b>\n'
'• 📊 Перегляд графіку на сьогодні і завтра\n'
'• 🔔 Автоматичні сповіщення за 5 хвилин до подій\n'
'• 🔄 Моніторинг змін графіку\n\n'
'Використовуйте кнопки нижче 👇',
parse_mode='HTML',
reply_markup=get_main_keyboard(),
)
@dp.message(lambda msg: msg.text == 'ℹ️ Про бота')
async def cmd_info(message: Message):
"""Показывает информацию о боте"""
if message.chat.id not in ALLOWED_CHAT_IDS:
return
await message.answer(
'<b>ℹ️ Про бота</b>\n\n'
'🚀 <b>Версія:</b> 2.2 (Стабільна)\n\n'
'📝 <b>Реліз-ноути:</b>\n'
'├ 15.11.2024: Адаптація під оновлену логіку сайту VOE\n'
'├ Виправлено парсинг half.left та half.right\n'
'├ Покращено визначення підтвердження відключень\n'
'└ Оптимізовано обробку статусу для всієї години\n\n'
'⚡ <b>Функціональність:</b>\n'
'├ Моніторинг графіку 24/7\n'
'├ Сповіщення за 5 хвилин\n'
'├ Детальна статистика дня\n'
'└ Красива візуалізація\n\n'
'🔐 <b>Безпека:</b> Використовуються .env файли\n'
'💾 <b>Джерело:</b> voe.com.ua',
parse_mode='HTML',
reply_markup=get_main_keyboard(),
)
@dp.message(Command('schedule'))
async def cmd_schedule(message: Message):
"""Показывает полный график"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('⏳ Завантаження графіку...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
schedule = parse_html(html)
text = format_schedule_message(schedule, days_to_show=2)
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message(Command('today'))
async def cmd_today(message: Message):
"""Показывает график на сегодня"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('⏳ Завантаження графіку сьогодні...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
schedule = parse_html(html)
text = format_single_day_schedule(schedule, 0)
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message(Command('tomorrow'))
async def cmd_tomorrow(message: Message):
"""Показывает график на завтра"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('⏳ Завантаження графіку завтра...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
schedule = parse_html(html)
text = format_single_day_schedule(schedule, 1)
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
# await message.answer("❌ Функція тимчасово недоступна. Чекаємо на оновлення сайту", parse_mode="HTML", reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message(Command('check'))
async def cmd_check(message: Message):
"""Принудительная проверка графика"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('🔄 <b>Перевіряю графік...</b>', parse_mode='HTML')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
new_schedule = parse_html(html)
prefix = (
'✅ <b>Знайдено зміни!</b>\n\n'
if schedules_differ(last_schedule, new_schedule)
else '✓ <b>Графік без змін</b>\n\n'
)
result = prefix + format_schedule_message(new_schedule, days_to_show=2)
await message.answer(result, parse_mode='HTML', reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message()
async def handle_text(message: Message):
"""Обработчик текстовых сообщений и кнопок"""
if message.chat.id not in ALLOWED_CHAT_IDS:
return
text = message.text
# Кнопка "Графік"
if text == '📊 Графік':
await cmd_schedule(message)
# Кнопка "Сьогодні"
elif text == '📅 Сьогодні':
await cmd_today(message)
# Кнопка "Завтра"
elif text == '📅 Завтра':
await cmd_tomorrow(message)
# Кнопка "Оновити"
elif text == '🔄 Оновити':
await cmd_check(message)
# Кнопка "Про бота"
elif text == 'ℹ️ Про бота':
await cmd_info(message)
# Неизвестная команда
else:
await message.answer(
'❓ <b>Команда не розпізнана</b>\n\n'
'Використовуйте кнопки на клавіатурі або команди:\n'
'/start • /today • /tomorrow • /schedule • /check',
parse_mode='HTML',
reply_markup=get_main_keyboard(),
)
# ============================================================================
# ГЛАВНАЯ ФУНКЦИЯ
# ============================================================================
async def main():
"""Главная функция"""
logger.info('=' * 50)
logger.info('ЗАПУСК БОТА V2.2 (stable 2.2, 15.11.2025)')
logger.info('=' * 50)
if not TELEGRAM_TOKEN or os.getenv('TELEGRAM_TOKEN', 'YOUR_TOKEN_HERE') == TELEGRAM_TOKEN:
logger.error('❌ TELEGRAM_TOKEN не конфігурований! Напишіть токен в .env файл')
return
if not ALLOWED_CHAT_IDS:
logger.error('❌ ALLOWED_CHAT_IDS не конфігуровані! Напишіть ID в .env файл')
return
logger.info(f'📌 Allowed chat ids: {ALLOWED_CHAT_IDS}')
logger.info(f'⏱ Інтервал перевірки: {CHECK_INTERVAL} сек')
logger.info('=' * 50)
# Запускаем мониторинг
monitoring_task = asyncio.create_task(monitoring_loop())
try:
await dp.start_polling(bot)
except KeyboardInterrupt:
logger.info('⏹ Бот зупинений користувачем')
finally:
monitoring_task.cancel()
await bot.session.close()
logger.info('✓ Підключення закрито')
if __name__ == '__main__':
try:
asyncio.run(main())
except KeyboardInterrupt:
logger.info('⏹ Завершено')
-7
View File
@@ -1,7 +0,0 @@
[project]
name = "dtek-notif"
version = "0.1.0"
description = "Add your description here"
readme = "README.md"
requires-python = ">=3.13"
dependencies = []
-5
View File
@@ -1,5 +0,0 @@
requests>=2.31.0
beautifulsoup4>=4.12.0
aiogram>=3.3.0
python-dotenv>=1.0.0
aiohttp>=3.9.0
-14
View File
@@ -1,14 +0,0 @@
EDU_LOGIN=your_edu_login_here
EDU_PASSWORD=your_edu_password_here
EDU_URL_LOGIN=https://edu.edu.vn.ua/user/login
EDU_URL_VERIFY=https://edu.edu.vn.ua/course/userlist
PHPSESSID_INTERVAL=10
USER_AGENT="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36"
WEBINAR_URL=https://edu.edu.vn.ua/webinar/useractive
WEBINAR_CHECK_INTERVAL=60
REDIS_HOST=redis
REDIS_PORT=6379
PLAYWRIGHT_WS=ws://playwright-service:3000/ws
TZ=Europe/Kyiv
WEBINAR_TELEGRAM_TOKEN=your_telegram_bot_token_here
WEBINAR_ADMIN_ID=123456789
-1
View File
@@ -1 +0,0 @@
1.56.0
-49
View File
@@ -1,49 +0,0 @@
services:
redis:
image: redis:8.10.2-alpine
restart: unless-stopped
volumes:
- redis-data:/data
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 5s
timeout: 3s
retries: 5
playwright-service:
image: mcr.microsoft.com/playwright:v1.56.0-jammy
restart: unless-stopped
command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws
session-keeper:
build: ./phpsessid-bot
image: gcr.forust.xyz/forust/session-keeper:latest
pull_policy: build
env_file: .env
restart: unless-stopped
depends_on:
redis:
condition: service_healthy
healthcheck:
test: ["CMD-SHELL", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
interval: 30s
timeout: 5s
retries: 10
start_period: 60s
webinar-checker:
build: ./webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:latest
pull_policy: build
env_file: .env
restart: unless-stopped
depends_on:
redis:
condition: service_healthy
session-keeper:
condition: service_healthy
playwright-service:
condition: service_started
volumes:
redis-data:
-96
View File
@@ -1,96 +0,0 @@
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: edu-master-webinar
namespace: edu-master
labels:
release: prometheus-stack
spec:
groups:
- name: edu_master.webinar
rules:
# No successful webinar check for 5m (~2-3 missed 2-min checks).
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
# The last_success > 0 guard is mandatory: checker.py initialises
# last_success to 0, so without it `time() - 0` equals the current epoch
# and humanizeDuration renders ~20722d on every pod restart. Keep the
# duration expression on the left so $value stays the real gap.
- alert: WebinarCheckerNoSuccessfulCheck
expr: |
((time() - webinar_check_last_success_timestamp_seconds) > 300)
and (webinar_check_last_success_timestamp_seconds > 0)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 2m
labels:
severity: critical
annotations:
summary: "Webinar checker has no successful check for 5m"
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
# Checks are running but none has ever succeeded since pod start.
# Split out from the rule above so a zeroed gauge never feeds
# humanizeDuration.
- alert: WebinarCheckerNeverSucceeded
expr: |
(webinar_check_last_success_timestamp_seconds == 0)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker has never completed a successful check"
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
- alert: WebinarCheckerConsecutiveFailures
expr: |
webinar_check_consecutive_failures >= 3
for: 5m
labels:
severity: critical
annotations:
summary: "Webinar checker failing consecutively"
description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
- alert: WebinarCheckerScrapeDown
expr: |
absent(webinar_check_last_run_timestamp_seconds) == 1
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker metrics missing"
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
- alert: EduPhpsessidMissing
expr: |
edu_phpsessid_present == 0
for: 10m
labels:
severity: critical
annotations:
summary: "EDU_PHPSESSID missing"
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
# Hard deps: checker and playwright deployments unavailable.
- alert: WebinarCheckerDeploymentDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker deployment unavailable"
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
- alert: PlaywrightServiceDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Playwright service unavailable"
description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it."
-58
View File
@@ -1,58 +0,0 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: playwright-service
namespace: edu-master
labels:
app: edu-master-playwright
spec:
replicas: 1
selector:
matchLabels:
app: edu-master-playwright
template:
metadata:
labels:
app: edu-master-playwright
spec:
containers:
- name: playwright
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
image: mcr.microsoft.com/playwright:v1.56.0-jammy
imagePullPolicy: IfNotPresent
command:
- npx
- -y
- playwright@1.56.0
- run-server
- --port
- "3000"
- --path
- /ws
ports:
- containerPort: 3000
readinessProbe:
tcpSocket:
port: 3000
initialDelaySeconds: 5
periodSeconds: 10
timeoutSeconds: 3
livenessProbe:
tcpSocket:
port: 3000
initialDelaySeconds: 15
periodSeconds: 20
timeoutSeconds: 3
---
apiVersion: v1
kind: Service
metadata:
name: playwright-service
namespace: edu-master
spec:
selector:
app: edu-master-playwright
ports:
- name: ws
port: 3000
targetPort: 3000
-75
View File
@@ -1,75 +0,0 @@
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: redis
namespace: edu-master
labels:
app: edu-master-redis
spec:
serviceName: redis
replicas: 1
selector:
matchLabels:
app: edu-master-redis
template:
metadata:
labels:
app: edu-master-redis
spec:
containers:
- name: redis
image: redis:8.10.2-alpine
imagePullPolicy: IfNotPresent
ports:
- containerPort: 6379
volumeMounts:
- name: redis-data
mountPath: /data
resources:
requests:
cpu: 25m
memory: 64Mi
limits:
cpu: 250m
memory: 256Mi
readinessProbe:
exec:
command: ["redis-cli", "ping"]
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 3
livenessProbe:
exec:
command: ["redis-cli", "ping"]
initialDelaySeconds: 10
periodSeconds: 10
timeoutSeconds: 3
volumes:
- name: redis-data
persistentVolumeClaim:
claimName: redis-data-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: redis-data-pvc
namespace: edu-master
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 1Gi
---
apiVersion: v1
kind: Service
metadata:
name: redis
namespace: edu-master
spec:
selector:
app: edu-master-redis
ports:
- name: redis
port: 6379
targetPort: 6379
@@ -1,50 +0,0 @@
# One-time Job to migrate redis state from docker compose to k8s (maintenance window).
# The .example file is not applied by the deploy pipeline (mask *.example.yaml).
#
# Runbook:
# 1. docker compose -f <repo>/edu_master/compose.yaml stop # SIGTERM -> redis will flush dump.rdb
# 2. docker run --rm -v edu_master_redis-data:/data \
# -v /tmp/edu-master-backup:/backup \
# redis:alpine sh -c "cp /data/dump.rdb /backup/ && ls -la /backup"
# 3. kubectl apply -f edu_master/k8s/namespace.yaml
# 4. kubectl apply -f <only the PVC from redis.yaml> # seed must come BEFORE redis pod starts
# 5. kubectl apply -f edu_master/k8s/restore-seed-job.yaml.example
# kubectl wait --for=condition=complete job/redis-restore-seed -n edu-master --timeout=120s
# 6. kubectl delete job redis-restore-seed -n edu-master
# 7. kubectl apply -f edu_master/k8s/ -R # apply remaining manifests
apiVersion: batch/v1
kind: Job
metadata:
name: redis-restore-seed
namespace: edu-master
spec:
backoffLimit: 2
ttlSecondsAfterFinished: 3600
template:
spec:
restartPolicy: Never
containers:
- name: seed
image: redis:alpine
command:
- /bin/sh
- -ec
- |
ls -la /backup
cp /backup/dump.rdb /data/dump.rdb
chmod 644 /data/dump.rdb
ls -la /data
volumeMounts:
- name: redis-data
mountPath: /data
- name: backup
mountPath: /backup
readOnly: true
volumes:
- name: redis-data
persistentVolumeClaim:
claimName: redis-data-pvc
- name: backup
hostPath:
path: /tmp/edu-master-backup
type: DirectoryOrCreate
-29
View File
@@ -1,29 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: edu-master-secrets
namespace: edu-master
type: Opaque
stringData:
# Session keeper credentials
KEEPER_LOGIN: ""
KEEPER_PASSWORD: ""
KEEPER_INTERVAL: "10"
# EDU links
EDU_URL_BASE: "https://edu.edu.vn.ua"
EDU_URL_LOGIN: "/user/login"
EDU_URL_COURSES: "/course/userlist"
EDU_URL_WEBINAR: "/webinar/useractive"
# Playwright
USER_AGENT: ""
PLAYWRIGHT_WS: "ws://playwright-service:3000/ws"
# Webinar-checker
WEBINAR_TELEGRAM_TOKEN: ""
WEBINAR_ADMIN_ID: ""
WEBINAR_CHECK_INTERVAL: "60"
# Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml)
METRICS_PORT: "8000"
# Database
REDIS_HOST: "redis"
REDIS_PORT: "6379"
TZ: "Europe/Kyiv"
-15
View File
@@ -1,15 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: webinar-checker
namespace: edu-master
labels:
app: edu-master-webinar-checker
spec:
selector:
app: edu-master-webinar-checker
ports:
- name: metrics
port: 8000
targetPort: metrics
protocol: TCP
-51
View File
@@ -1,51 +0,0 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: session-keeper
namespace: edu-master
labels:
app: edu-master-session-keeper
spec:
replicas: 1
selector:
matchLabels:
app: edu-master-session-keeper
template:
metadata:
labels:
app: edu-master-session-keeper
spec:
initContainers:
- name: wait-redis
image: redis:8.10.2-alpine
command:
- /bin/sh
- -ec
- |
i=0
until redis-cli -h redis ping | grep -q PONG; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: redis not ready" && exit 1
sleep 2
done
echo "redis is ready"
containers:
- name: session-keeper
image: gcr.forust.xyz/forust/session-keeper:prod
envFrom:
- secretRef:
name: edu-master-secrets
resources:
requests:
cpu: 25m
memory: 96Mi
limits:
cpu: 250m
memory: 256Mi
readinessProbe:
exec:
command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
initialDelaySeconds: 15
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 10
-73
View File
@@ -1,73 +0,0 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: webinar-checker
namespace: edu-master
labels:
app: edu-master-webinar-checker
spec:
replicas: 1
selector:
matchLabels:
app: edu-master-webinar-checker
template:
metadata:
labels:
app: edu-master-webinar-checker
spec:
# Enforces dependency order like compose depends_on:
# redis healthy -> session-keeper healthy (EXISTS EDU_PHPSESSID) -> playwright started
initContainers:
- name: wait-deps
image: redis:8.10.2-alpine
command:
- /bin/sh
- -ec
- |
i=0
until redis-cli -h redis ping | grep -q PONG; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: redis not ready" && exit 1
sleep 2
done
echo "redis ok"
until [ "$(redis-cli -h redis EXISTS EDU_PHPSESSID)" = "1" ]; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: no PHPSESSID (session-keeper down?)" && exit 1
sleep 2
done
echo "PHPSESSID ok"
until nc -z playwright-service 3000; do
i=$((i+1))
[ "$i" -ge 300 ] && echo "TIMEOUT: playwright-service not reachable" && exit 1
sleep 2
done
echo "playwright ok"
containers:
- name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:prod
ports:
- name: metrics
containerPort: 8000
protocol: TCP
readinessProbe:
httpGet:
path: /health
port: metrics
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 12
initialDelaySeconds: 10
envFrom:
- secretRef:
name: edu-master-secrets
env:
- name: TZ
value: "Europe/Kyiv"
resources:
requests:
cpu: "50m"
memory: "128Mi"
limits:
cpu: "600m"
memory: "512Mi"
-15
View File
@@ -1,15 +0,0 @@
FROM python:3.11-slim
WORKDIR /app
# Install system dependencies
RUN apt-get update && apt-get install -y --no-install-recommends redis-tools && rm -rf /var/lib/apt/lists/*
# Install dependencies
RUN pip install --no-cache-dir requests==2.32.3 redis==5.2.1
# Copy application code
COPY . .
# Run the bot
CMD ["python", "bot.py"]
-132
View File
@@ -1,132 +0,0 @@
import logging
import os
import time
from datetime import datetime
import redis
import requests
# Configure logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# Load configuration (adapted to .env keys)
def _env(key, default=None):
v = os.getenv(key, default)
if isinstance(v, str) and len(v) >= 2 and ((v[0] == '"' and v[-1] == '"') or (v[0] == "'" and v[-1] == "'")):
return v[1:-1]
return v
LOGIN = _env('KEEPER_LOGIN')
PASSWORD = _env('KEEPER_PASSWORD')
EDU_BASE = _env('EDU_URL_BASE', 'https://edu.edu.vn.ua')
EDU_LOGIN_PATH = _env('EDU_URL_LOGIN', '/user/login')
EDU_COURSES_PATH = _env('EDU_URL_COURSES', '/course/userlist')
URL_LOGIN = f'{EDU_BASE.rstrip("/")}/{EDU_LOGIN_PATH.lstrip("/")}'
URL_VERIFY = f'{EDU_BASE.rstrip("/")}/{EDU_COURSES_PATH.lstrip("/")}'
INTERVAL = int(_env('KEEPER_INTERVAL', 10))
USER_AGENT = _env(
'USER_AGENT',
'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36',
)
REDIS_HOST = _env('REDIS_HOST', 'redis')
REDIS_PORT = int(_env('REDIS_PORT', 6379))
SUCCESS_FILE = '/tmp/last_success' # noqa: S108
def touch_success_file():
"""Updates the timestamp of the success file for healthchecks."""
try:
with open(SUCCESS_FILE, 'w') as f:
f.write(str(datetime.now().timestamp()))
except Exception as e:
logger.error(f'Failed to touch success file: {e}')
def main():
logger.info('Starting Session Keeper Bot')
# Connect to Redis
try:
redis_client = redis.Redis(host=REDIS_HOST, port=REDIS_PORT, decode_responses=True)
redis_client.ping()
logger.info(f'Connected to Redis at {REDIS_HOST}:{REDIS_PORT}')
except Exception as e:
logger.error(f'Failed to connect to Redis: {e}')
return
session = requests.Session()
# Set headers
headers = {
'User-Agent': USER_AGENT,
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
'Accept-Language': 'en-US,en;q=0.9',
'Cache-Control': 'max-age=0',
'Upgrade-Insecure-Requests': '1',
'Sec-Fetch-Site': 'same-origin',
'Sec-Fetch-Mode': 'navigate',
'Sec-Fetch-User': '?1',
'Sec-Fetch-Dest': 'document',
'Sec-Ch-Ua': '"Not_A Brand";v="99", "Chromium";v="142"',
'Sec-Ch-Ua-Mobile': '?0',
'Sec-Ch-Ua-Platform': '"Linux"',
'Accept-Encoding': 'gzip, deflate, br',
'Priority': 'u=0, i',
}
session.headers.update(headers)
while True:
try:
logger.info('Attempting login...')
# Login payload
payload = {'login': LOGIN, 'password': PASSWORD}
# Perform Login
# Note: The user request shows a POST to /user/login with form data
# We need to make sure we handle the PHPSESSID correctly.
# If we already have a PHPSESSID, requests will send it.
login_response = session.post(URL_LOGIN, data=payload, allow_redirects=True)
logger.info(f'Login Response Status: {login_response.status_code}')
logger.info(f'Cookies after login: {session.cookies.get_dict()}')
# Verify Session
logger.info('Verifying session...')
verify_response = session.get(URL_VERIFY, allow_redirects=False)
logger.info(f'Verify Response Status: {verify_response.status_code}')
if verify_response.status_code == 200:
logger.info('Session verification SUCCESS (200 OK).')
touch_success_file()
# Save PHPSESSID to Redis
phpsessid = session.cookies.get('PHPSESSID')
if phpsessid:
try:
redis_client.set('EDU_PHPSESSID', phpsessid)
logger.info(f'Saved PHPSESSID to Redis: {phpsessid}')
except Exception as e:
logger.error(f'Failed to save PHPSESSID to Redis: {e}')
elif verify_response.status_code == 302:
logger.warning('Session verification FAILED (302 Redirect). Session might be invalid.')
else:
logger.warning(f'Session verification returned unexpected status: {verify_response.status_code}')
except Exception as e:
logger.error(f'An error occurred: {e}')
logger.info(f'Sleeping for {INTERVAL} minutes...')
time.sleep(INTERVAL * 60)
if __name__ == '__main__':
main()
-13
View File
@@ -1,13 +0,0 @@
FROM python:3.11-slim
WORKDIR /app
# renovate: datasource=pypi depName=playwright versioning=pep440
ARG PLAYWRIGHT_VERSION=1.56.0
# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
COPY checker.py .
CMD ["python", "checker.py"]
File diff suppressed because it is too large. Load diff
+1 -1
View File
@@ -1,7 +1,7 @@
services: services:
errorpage: errorpage:
build: . build: .
image: gcr.forust.xyz/forust/error-pages:latest image: gcr.forust.xyz/forust/error-pages:prod
pull_policy: build pull_policy: build
container_name: error-pages container_name: error-pages
restart: unless-stopped restart: unless-stopped
+9
View File
@@ -20,6 +20,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: error-pages app: error-pages
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -28,6 +30,13 @@ spec:
containers: containers:
- name: error-pages - name: error-pages
image: gcr.forust.xyz/forust/error-pages:prod image: gcr.forust.xyz/forust/error-pages:prod
# p95 6M, max 10M, no limit before.
resources:
requests:
cpu: "10m"
memory: "32Mi"
limits:
memory: "128Mi"
ports: ports:
- containerPort: 80 - containerPort: 80
readinessProbe: readinessProbe:
+6 -3
View File
@@ -1,6 +1,6 @@
services: services:
server: server:
image: docker.gitea.com/gitea:1.27.3 image: docker.gitea.com/gitea:28.0.0
container_name: gitea container_name: gitea
restart: always restart: always
environment: environment:
@@ -13,9 +13,12 @@ services:
- GITEA__database__PASSWD=gitea - GITEA__database__PASSWD=gitea
- GITEA__database__NAME=gitea - GITEA__database__NAME=gitea
# Server # Server
- GITEA__server__ROOT_URL=https://gitea.forust.xyz - GITEA__server__ROOT_URL=https://git.forust.xyz
- GITEA__server__SSH_DOMAIN=gitssh.forust.xyz - GITEA__server__SSH_DOMAIN=gitssh.forust.xyz
- GITEA__server__SSH_PORT=2221 - GITEA__server__SSH_PORT=2221
# Pin 28.0 defaults explicitly (see k8s/config.yaml for rationale)
- GITEA__service__DISABLE_REGISTRATION=true
- GITEA__actions__RUN_RETENTION_DAYS=90
# Mailer # Mailer
- GITEA__mailer__ENABLED=true - GITEA__mailer__ENABLED=true
- GITEA__mailer__FROM=${SERVICE_EMAIL} - GITEA__mailer__FROM=${SERVICE_EMAIL}
@@ -34,7 +37,7 @@ services:
- "traefik.http.services.gitea.loadbalancer.server.port=3000" - "traefik.http.services.gitea.loadbalancer.server.port=3000"
# Prod Router # Prod Router
- "traefik.http.routers.gitea.rule=Host(`gitea.forust.xyz`)" - "traefik.http.routers.gitea.rule=Host(`git.forust.xyz`) || Host(`gitea.forust.xyz`)"
- "traefik.http.routers.gitea.entrypoints=websecure" - "traefik.http.routers.gitea.entrypoints=websecure"
- "traefik.http.routers.gitea.tls.certresolver" - "traefik.http.routers.gitea.tls.certresolver"
# Local Router # Local Router
+1
View File
@@ -8,6 +8,7 @@ spec:
dnsNames: dnsNames:
- gcr.forust.xyz - gcr.forust.xyz
- gitea.forust.xyz - gitea.forust.xyz
- git.forust.xyz
issuerRef: issuerRef:
name: letsencrypt-prod name: letsencrypt-prod
kind: ClusterIssuer kind: ClusterIssuer
+13 -2
View File
@@ -4,11 +4,14 @@ metadata:
name: gitea-config name: gitea-config
namespace: gitea namespace: gitea
data: data:
GITEA__server__DOMAIN: "gitea.forust.xyz" GITEA__server__ROOT_URL: "https://git.forust.xyz"
GITEA__server__ROOT_URL: "https://gitea.forust.xyz"
GITEA__server__SSH_DOMAIN: "gitssh.forust.xyz" GITEA__server__SSH_DOMAIN: "gitssh.forust.xyz"
GITEA__server__SSH_PORT: "2221" GITEA__server__SSH_PORT: "2221"
GITEA__service__DISABLE_REGISTRATION: "true"
GITEA__actions__RUN_RETENTION_DAYS: "90"
GITEA__database__DB_TYPE: "postgres" GITEA__database__DB_TYPE: "postgres"
GITEA__database__HOST: "postgres.database.svc.cluster.local:5432" GITEA__database__HOST: "postgres.database.svc.cluster.local:5432"
GITEA__database__NAME: "gitea" GITEA__database__NAME: "gitea"
@@ -17,6 +20,14 @@ data:
GITEA__mailer__ENABLED: "false" GITEA__mailer__ENABLED: "false"
GITEA__metrics__ENABLED: "true"
# No code/issue search needed: bleve reindexes the whole issue index on
# every pod restart (cron.rebuild_issue_indexer RUN_AT_START) and hammers
# the rotational disk for an hour. "db" serves issue search from postgres.
GITEA__indexer__ISSUE_INDEXER_TYPE: "db"
GITEA__indexer__REPO_INDEXER_ENABLED: "false"
GITEA__log__logger.access.MODE: "console, file" GITEA__log__logger.access.MODE: "console, file"
USER_UID: "1000" USER_UID: "1000"
USER_GID: "1000" USER_GID: "1000"
+10 -4
View File
@@ -3,6 +3,8 @@ kind: Service
metadata: metadata:
name: gitea-service name: gitea-service
namespace: gitea namespace: gitea
labels:
app: gitea
spec: spec:
selector: selector:
app: gitea app: gitea
@@ -17,6 +19,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: gitea-deployment name: gitea-deployment
namespace: gitea namespace: gitea
spec: spec:
@@ -24,6 +28,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: gitea app: gitea
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -31,7 +37,7 @@ spec:
spec: spec:
containers: containers:
- name: gitea - name: gitea
image: gitea/gitea:1.27.3 image: gitea/gitea:28.0.0
envFrom: envFrom:
- configMapRef: - configMapRef:
name: gitea-config name: gitea-config
@@ -47,10 +53,10 @@ spec:
mountPath: /data mountPath: /data
resources: resources:
requests: requests:
memory: "512Mi" memory: "320Mi"
cpu: "300m" cpu: "100m"
limits: limits:
memory: "1.5Gi" memory: "1Gi"
cpu: "1300m" cpu: "1300m"
volumes: volumes:
- name: gitea-data - name: gitea-data
+3 -15
View File
@@ -7,24 +7,12 @@ spec:
entryPoints: entryPoints:
- websecure - websecure
routes: routes:
- match: Host(`gitea.forust.xyz`) # Metrics are scraped directly through the cluster Service.
- match: (Host(`gitea.forust.xyz`) || Host(`git.forust.xyz`)) && !PathPrefix(`/metrics`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: gitea-service - name: gitea-service
port: 3000 port: 3000
# Registry route: NO crowdsec-bouncer.
# The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on
# *every* request. A deploy burst (runner Action API polls, `docker
# manifest inspect` per own image, containerd pulls, smoke probes) fires
# hundreds of parallel registry calls; LAPI saturation pushed the lookup
# past the plugin timeout, and the bouncer fail-closed with 403 - which
# containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod.
# This route only serves authenticated OCI traffic (registry tokens,
# basic-auth already handled by gitea) and scanners get nothing useful
# from /v2, so there is no bruteforce surface to protect here.
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`) - match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
kind: Rule kind: Rule
services: services:
@@ -42,7 +30,7 @@ spec:
entryPoints: entryPoints:
- websecure - websecure
routes: routes:
- match: Host(`gitea.workstation.internal`) || Host(`gitea.gigaforust.internal`) - match: (Host(`gitea.workstation.internal`) || Host(`gitea.gigaforust.internal`)) || (Host(`git.workstation.internal`) || Host(`git.gigaforust.internal`))
kind: Rule kind: Rule
services: services:
- name: gitea-service - name: gitea-service
+16
View File
@@ -0,0 +1,16 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: gitea
namespace: gitea
labels:
release: prometheus-stack
spec:
selector:
matchLabels:
app: gitea
endpoints:
- port: http
path: /metrics
interval: 30s
scrapeTimeout: 10s
View File
Whitespace-only changes.
+2 -2
View File
@@ -177,8 +177,8 @@ data:
# url: https://gitssh.forust.xyz # url: https://gitssh.forust.xyz
# - title: gcr.forust.xyz # - title: gcr.forust.xyz
# url: https://gcr.forust.xyz/v2/ # url: https://gcr.forust.xyz/v2/
- title: gitea.forust.xyz - title: git.forust.xyz
url: https://gitea.forust.xyz url: https://git.forust.xyz
- title: nextcloud.forust.xyz - title: nextcloud.forust.xyz
url: https://nextcloud.forust.xyz url: https://nextcloud.forust.xyz
- title: mc.forust.xyz - title: mc.forust.xyz
+7 -3
View File
@@ -13,6 +13,8 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: glance-deployment name: glance-deployment
namespace: glance namespace: glance
spec: spec:
@@ -20,6 +22,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: glance app: glance
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -57,17 +61,17 @@ spec:
resources: resources:
requests: requests:
cpu: "50m" cpu: "50m"
memory: "64Mi" memory: "32Mi"
limits: limits:
cpu: "200m" cpu: "200m"
memory: "256Mi" memory: "128Mi"
volumes: volumes:
- name: glance-config - name: glance-config
configMap: configMap:
name: glance-config name: glance-config
- name: glance-assets - name: glance-assets
configMap: configMap:
name: glance-config name: glance-assets
- name: docker-socket - name: docker-socket
hostPath: hostPath:
path: /var/run/docker.sock path: /var/run/docker.sock
+1 -1
View File
@@ -7,7 +7,7 @@
# # Dev server_url # # Dev server_url
# server_url: https://hs.dev_internal_domain.internal # server_url: https://hs.dev_internal_domain.internal
listen_addr: 0.0.0.0:8080 listen_addr: 0.0.0.0:8080
metrics_listen_addr: 127.0.0.1:9090 metrics_listen_addr: 0.0.0.0:9090
grpc_listen_addr: 127.0.0.1:50443 grpc_listen_addr: 127.0.0.1:50443
grpc_allow_insecure: false grpc_allow_insecure: false
noise: noise:
@@ -3,6 +3,8 @@ kind: Service
metadata: metadata:
name: headscale-server-external name: headscale-server-external
namespace: headscale namespace: headscale
labels:
app: headscale
spec: spec:
ports: ports:
- port: 8080 - port: 8080
-8
View File
@@ -23,17 +23,11 @@ spec:
port: 8080 port: 8080
- match: Host(`hs.forust.xyz`) && PathPrefix(`/admin`) - match: Host(`hs.forust.xyz`) && PathPrefix(`/admin`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: headscale-ui-external - name: headscale-ui-external
port: 80 port: 80
- match: Host(`hs.forust.xyz`) && PathPrefix(`/metrics`) - match: Host(`hs.forust.xyz`) && PathPrefix(`/metrics`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: headscale-server-external - name: headscale-server-external
port: 9090 port: 9090
@@ -53,8 +47,6 @@ spec:
kind: Rule kind: Rule
middlewares: middlewares:
- name: headplane-prefix - name: headplane-prefix
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: headplane-external - name: headplane-external
port: 3000 port: 3000
+18
View File
@@ -0,0 +1,18 @@
apiVersion: operator.victoriametrics.com/v1beta1
kind: VMServiceScrape
metadata:
name: headscale
namespace: headscale
labels:
release: prometheus-stack
spec:
# The external Service has a manually managed EndpointSlice, not Endpoints.
discoveryRole: endpointslice
selector:
matchLabels:
app: headscale
endpoints:
- port: metrics
path: /metrics
interval: 30s
scrapeTimeout: 10s
+4
View File
@@ -0,0 +1,4 @@
SECRET_ENCRYPTION_KEY="REPLACE_ME"
TZ="Europe/Bratislava"
PUID="1000"
PGID="1000"
+38
View File
@@ -0,0 +1,38 @@
services:
homarr:
container_name: homarr
image: ghcr.io/homarr-labs/homarr:v2.2.0
restart: unless-stopped
volumes:
- ./appdata:/appdata
- /var/run/docker.sock:/var/run/docker.sock:ro
- ./kubeconfig:/app/config/kubeconfig:ro
env_file: .env
ports:
- 80:7575
- 81:3000
environment:
- TZ=${TZ:-Europe/Bratislava}
- TURBO_TELEMETRY_DISABLED=1
- KUBECONFIG=/app/config/kubeconfig
labels:
- "traefik.enable=true"
- "traefik.http.services.homarr.loadbalancer.server.port=7575"
# Prod Router
- "traefik.http.routers.homarr.rule=Host(`homarr.forust.xyz`)"
- "traefik.http.routers.homarr.entrypoints=websecure"
- "traefik.http.routers.homarr.tls.certresolver=letsencrypt"
# Local Router
- "traefik.http.routers.homarr-local.rule=Host(`homarr.workstation.internal`)"
- "traefik.http.routers.homarr-local.entrypoints=websecure"
- "traefik.http.routers.homarr-local.tls=true"
# Dev Router
- "traefik.http.routers.homarr-dev.rule=Host(`homarr.gigaforust.internal`)"
- "traefik.http.routers.homarr-dev.entrypoints=websecure"
- "traefik.http.routers.homarr-dev.tls=true"
networks:
- proxy
networks:
proxy:
external: true
+28
View File
@@ -0,0 +1,28 @@
# apiVersion: cert-manager.io/v1
# kind: Certificate
# metadata:
# name: home-prod-tls
# namespace: homarr
# spec:
# secretName: home-prod-tls
# dnsNames:
# - home.forust.xyz
# issuerRef:
# name: letsencrypt-prod
# kind: ClusterIssuer
# ---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: homarr
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+9
View File
@@ -0,0 +1,9 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: homarr-config
namespace: homarr
data:
TZ: "Europe/Bratislava"
TURBO_TELEMETRY_DISABLED: "1"
ENABLE_KUBERNETES: "true"
+83
View File
@@ -0,0 +1,83 @@
apiVersion: v1
kind: Service
metadata:
name: homarr-service
namespace: homarr
spec:
selector:
app: homarr
ports:
- port: 7575
targetPort: 7575
---
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: homarr-deployment
namespace: homarr
spec:
replicas: 1
selector:
matchLabels:
app: homarr
strategy:
type: Recreate
template:
metadata:
labels:
app: homarr
spec:
serviceAccountName: homarr
containers:
- name: homarr
image: ghcr.io/homarr-labs/homarr:v2.2.0
envFrom:
- configMapRef:
name: homarr-config
- secretRef:
name: homarr-secrets
ports:
- containerPort: 7575
readinessProbe:
httpGet:
path: /
port: 7575
initialDelaySeconds: 30
periodSeconds: 10
failureThreshold: 6
livenessProbe:
httpGet:
path: /
port: 7575
initialDelaySeconds: 60
periodSeconds: 30
failureThreshold: 3
volumeMounts:
- name: homarr-data
mountPath: /appdata
resources:
requests:
cpu: "250m"
memory: "350Mi"
limits:
cpu: "500m"
memory: "700Mi"
volumes:
- name: homarr-data
persistentVolumeClaim:
claimName: homarr-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: homarr-pvc
namespace: homarr
spec:
resources:
requests:
storage: 2Gi
volumeMode: Filesystem
accessModes:
- ReadWriteOnce
+33
View File
@@ -0,0 +1,33 @@
# apiVersion: traefik.io/v1alpha1
# kind: IngressRoute
# metadata:
# name: homarr-prod
# namespace: homarr
# spec:
# entryPoints:
# - websecure
# routes:
# - match: Host(`home.forust.xyz`)
# kind: Rule
# services:
# - name: homarr-service
# port: 7575
# tls:
# secretName: home-prod-tls
# ---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: homarr-local
namespace: homarr
spec:
entryPoints:
- websecure
routes:
- match: Host(`home.workstation.internal`) || Host(`home.gigaforust.internal`)
kind: Rule
services:
- name: homarr-service
port: 7575
tls:
secretName: internal-wildcard-tls
@@ -1,4 +1,4 @@
apiVersion: v1 apiVersion: v1
kind: Namespace kind: Namespace
metadata: metadata:
name: edu-master name: homarr
+58
View File
@@ -0,0 +1,58 @@
apiVersion: v1
kind: ServiceAccount
metadata:
name: homarr
namespace: homarr
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
name: homarr-readonly
rules:
- apiGroups: [""]
resources:
- pods
- services
- endpoints
- namespaces
- nodes
- configmaps
- persistentvolumeclaims
- events
verbs: ["get", "list", "watch"]
- apiGroups: ["apps"]
resources:
- deployments
- statefulsets
- daemonsets
- replicasets
verbs: ["get", "list", "watch"]
- apiGroups: ["networking.k8s.io"]
resources:
- ingresses
verbs: ["get", "list", "watch"]
- apiGroups: ["traefik.io"]
resources:
- ingressroutes
- ingressroutetcps
- ingressrouteudps
- middlewares
verbs: ["get", "list", "watch"]
- apiGroups: ["metrics.k8s.io"]
resources:
- pods
- nodes
verbs: ["get", "list"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: homarr-readonly
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: homarr-readonly
subjects:
- kind: ServiceAccount
name: homarr
namespace: homarr
+9
View File
@@ -0,0 +1,9 @@
apiVersion: v1
kind: Secret
metadata:
name: homarr-secrets
namespace: homarr
type: Opaque
stringData:
# openssl rand -hex 32
SECRET_ENCRYPTION_KEY: "REPLACE_ME"
+2 -2
View File
@@ -3,7 +3,7 @@ services:
build: build:
context: . context: .
dockerfile: Dockerfile.forust dockerfile: Dockerfile.forust
image: gcr.forust.xyz/forust/forust-homepage:latest image: gcr.forust.xyz/forust/forust-homepage:prod
pull_policy: build pull_policy: build
# ports: # ports:
# - "8085:80" # - "8085:80"
@@ -35,7 +35,7 @@ services:
build: build:
context: . context: .
dockerfile: Dockerfile.xdfnx dockerfile: Dockerfile.xdfnx
image: gcr.forust.xyz/forust/xdfnx-homepage:latest image: gcr.forust.xyz/forust/xdfnx-homepage:prod
pull_policy: build pull_policy: build
restart: unless-stopped restart: unless-stopped
# ports: # ports:
+1 -1
View File
@@ -174,7 +174,7 @@
<h2>./projects</h2> <h2>./projects</h2>
<ul class="repo-list"> <ul class="repo-list">
<li> <li>
<a href="https://gitea.forust.xyz/forust/gosleep" target="_blank">forust/gosleep</a> <a href="https://git.forust.xyz/forust/gosleep" target="_blank">forust/gosleep</a>
<span class="comment">// linux sleep timer written in rust (originally in go)</span> <span class="comment">// linux sleep timer written in rust (originally in go)</span>
</li> </li>
</ul> </ul>
+31
View File
@@ -0,0 +1,31 @@
# Gateway API PoC for homepages. Lives next to the TLS secrets so
# certificateRefs stay same-namespace and no ReferenceGrant is needed.
# Listener ports must match the Traefik entryPoints (80/443),
# otherwise Traefik marks the listener Invalid.
# Local .internal hosts are deliberately left on IngressRoute,
# only prod is migrated here.
apiVersion: gateway.networking.k8s.io/v1
kind: Gateway
metadata:
name: homepages
namespace: homepages
spec:
gatewayClassName: traefik
listeners:
- name: http
protocol: HTTP
port: 80
allowedRoutes:
namespaces:
from: Same
- name: https
protocol: HTTPS
port: 443
tls:
mode: Terminate
certificateRefs:
- name: forust-homepage-prod-tls
- name: xdfnx-homepage-prod-tls
allowedRoutes:
namespaces:
from: Same
+8 -4
View File
@@ -20,6 +20,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: forust-homepage app: forust-homepage
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -39,10 +41,10 @@ spec:
failureThreshold: 3 failureThreshold: 3
resources: resources:
requests: requests:
memory: "10Mi" memory: "32Mi"
cpu: "20m" cpu: "20m"
limits: limits:
memory: "100Mi" memory: "128Mi"
cpu: "50m" cpu: "50m"
--- ---
apiVersion: v1 apiVersion: v1
@@ -67,6 +69,8 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: xdfnx-homepage app: xdfnx-homepage
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -86,8 +90,8 @@ spec:
failureThreshold: 3 failureThreshold: 3
resources: resources:
requests: requests:
memory: "10Mi" memory: "32Mi"
cpu: "20m" cpu: "20m"
limits: limits:
memory: "100Mi" memory: "128Mi"
cpu: "50m" cpu: "50m"
+69
View File
@@ -0,0 +1,69 @@
# PoC: homepages prod hosts via Gateway API.
# Runs alongside k8s/ingress.yaml - delete the prod IngressRoutes only after verification.
# There are no local .internal hosts here, they stay on the local IngressRoute.
# No per-route security middlewares: L3 enforcement moved to the host
# firewall bouncer, so HTTPRoutes stay clean.
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: homepages-http-redirect
namespace: homepages
spec:
parentRefs:
- name: homepages
kind: Gateway
sectionName: http
hostnames:
- forust.xyz
- www.forust.xyz
- xdfnx.cfd
rules:
- filters:
- type: RequestRedirect
requestRedirect:
scheme: https
statusCode: 301
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: forust-homepage-https
namespace: homepages
spec:
parentRefs:
- name: homepages
kind: Gateway
sectionName: https
hostnames:
- forust.xyz
- www.forust.xyz
rules:
- matches:
- path:
type: PathPrefix
value: /
backendRefs:
- name: forust-homepage-service
port: 80
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: xdfnx-homepage-https
namespace: homepages
spec:
parentRefs:
- name: homepages
kind: Gateway
sectionName: https
hostnames:
- xdfnx.cfd
rules:
- matches:
- path:
type: PathPrefix
value: /
backendRefs:
- name: xdfnx-homepage-service
port: 80
-6
View File
@@ -9,9 +9,6 @@ spec:
routes: routes:
- match: Host(`forust.xyz`) || Host(`www.forust.xyz`) - match: Host(`forust.xyz`) || Host(`www.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
priority: 10 priority: 10
services: services:
- name: forust-homepage-service - name: forust-homepage-service
@@ -49,9 +46,6 @@ spec:
routes: routes:
- match: Host(`xdfnx.cfd`) - match: Host(`xdfnx.cfd`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: xdfnx-homepage-service - name: xdfnx-homepage-service
port: 80 port: 80
Loaded 100 of 378 files, more files were not shown because too many files have changed in this diff. Show more