Compare commits

..
Author SHA1 Message Date
forust 7ee7d0c961 Update README.md
deploy / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
ci / build (push) Skipped
ci / lint-prettier (pull_request) Successful in 3s
ci / lint-ruff (pull_request) Successful in 0s
ci / lint-yaml (pull_request) Successful in 2s
ci / validate (pull_request) Successful in 1s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 8s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (pull_request) Successful in 1s
2026-09-26 00:15:01 +02:00
forust 71e769c002 chore(deploy): disable checkmk, enable netbox and rackpeek
deploy / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-prettier (push) Failing after 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / lint-prettier (pull_request) Failing after 3s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-yaml (pull_request) Successful in 3s
ci / lint-dockerfiles (pull_request) Successful in 1s
ci / validate (pull_request) Successful in 1s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 21s
2026-09-26 00:04:37 +02:00
forust 16dd67c2c0 feat(rackpeek): add internal-only rack visualization service
Compose and k8s manifests behind workstation/gigaforust internal hosts.
2026-09-26 00:00:49 +02:00
forust c536a16a2a feat(netbox): add enterprise-grade server documenting app w/ shared postgres 2026-09-25 23:24:21 +02:00
forust a4a4bb4cc5 feat(netbird): add tailscale-alternative
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
deploy / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-prettier (pull_request) Successful in 3s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-yaml (pull_request) Successful in 5s
ci / lint-dockerfiles (pull_request) Successful in 1s
ci / validate (pull_request) Successful in 1s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 10s
2026-09-25 20:26:11 +02:00
forust 648b354951 refactor(deploy): marker-driven selection (k8s/active, root active); enable headscale/nextcloud hybrid, disable dockmon/kener/downtify/n8n
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / validate (push) Successful in 2s
deploy / preflight (push) Successful in 1s
renovate-ci / validate-renovate (push) Successful in 8s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / build (push) Successful in 1s
deploy / validate (push) Successful in 1m40s
deploy / apply-k8s (push) Successful in 1m41s
deploy / apply-compose (push) Successful in 13s
2026-09-23 18:11:51 +02:00
forust b0a9b3476d fix(deploy): drop broken %q quoting that wrapped remote vars in literal quotes
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 1s
deploy / redeploy (push) Failing after 1m43s
2026-09-23 16:33:51 +02:00
forust ac3bf4a55a fix(deploy): strip quotes and CR from DEPLOY_PATH
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 1s
deploy / redeploy (push) Failing after 1s
2026-09-23 16:31:06 +02:00
forust 34fb6f85ba Merge pull request 'chore(deps): update darthnorse/dockmon docker tag to v2.5.0' (#39) from renovate/darthnorse-dockmon-2.x into main
renovate-ci / validate-renovate (push) Successful in 6s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / build (push) Successful in 1s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
deploy / redeploy (push) Failing after 1s
Reviewed-on: #39
2026-09-23 14:16:21 +00:00
renovate-bot 66eacd86d1 chore(deps): update darthnorse/dockmon docker tag to v2.5.0 2026-09-23 14:16:21 +00:00
forust 54f43fc2c7 Merge pull request 'chore(deps): update ghcr.io/lukegus/termix docker tag to v2.8.0' (#40) from renovate/ghcr.io-lukegus-termix-2.x into main
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 1s
ci / validate (push) Successful in 2s
deploy / redeploy (push) Failing after 0s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Successful in 1s
Reviewed-on: #40
2026-09-23 14:16:06 +00:00
renovate-bot 451d0dc6b7 chore(deps): update ghcr.io/lukegus/termix docker tag to v2.8.0 2026-09-23 14:16:06 +00:00
forust 88ae20a543 Merge pull request 'chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v3' (#42) from renovate/ghcr.io-henriquesebastiao-downtify-3.x into main
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 0s
ci / validate (push) Successful in 1s
deploy / redeploy (push) Failing after 0s
ci / build (push) Canceled after 0s
renovate-ci / validate-renovate (push) Successful in 7s
Reviewed-on: #42
2026-09-23 14:15:51 +00:00
renovate-bot 6bd183ba73 chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v3
renovate-ci / validate-renovate (push) Skipped
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-yaml (pull_request) Successful in 2s
ci / lint-dockerfiles (pull_request) Successful in 1s
ci / validate (pull_request) Successful in 2s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 10s
ci / build (push) Skipped
ci / lint-prettier (pull_request) Successful in 3s
2026-09-23 14:14:58 +00:00
forust aeefdd8560 Merge pull request 'chore(deps): update docker.n8n.io/n8nio/n8n docker tag to v2.41.0' (#43) from renovate/docker.n8n.io-n8nio-n8n-2.x into main
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 0s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
deploy / redeploy (push) Failing after 1s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 1s
Reviewed-on: #43
2026-09-23 14:14:39 +00:00
renovate-bot c10d344fd7 chore(deps): update docker.n8n.io/n8nio/n8n docker tag to v2.41.0 2026-09-23 14:14:39 +00:00
forust 6e5f80611b Merge pull request 'Cicd/deploy rework' (#44) from cicd/deploy-rework into main
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 0s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
deploy / redeploy (push) Failing after 0s
renovate-ci / validate-renovate (push) Successful in 6s
ci / build (push) Successful in 1m47s
Reviewed-on: #44
2026-09-23 14:11:42 +00:00
forust 5cf0c90258 style: prettier formatting for edu_master alerts and postgres readme
renovate-ci / validate-renovate (push) Skipped
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
ci / build (push) Skipped
ci / lint-prettier (pull_request) Successful in 4s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 0s
ci / lint-yaml (pull_request) Successful in 2s
ci / lint-dockerfiles (pull_request) Successful in 0s
ci / validate (pull_request) Successful in 1s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 7s
2026-09-23 16:10:33 +02:00
forust 90a452a253 feat(deploy): auto-rollout on push to main (gated by branch protection)
ci / lint-prettier (push) Failing after 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / lint-yaml (pull_request) Successful in 2s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / lint-prettier (pull_request) Failing after 3s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-dockerfiles (pull_request) Successful in 0s
ci / validate (pull_request) Successful in 2s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 14s
2026-09-23 15:58:15 +02:00
forust a9eabfba02 fix(userbot): include internal-certificate in kustomize base 2026-09-23 15:52:32 +02:00
forust 46c7e99b1d chore(deploy): rework k8s pipeline, monitoring and postgres 17
Deploy workflow uses git-tracked manifests, DISABLED flag and kustomize overlays; add webinar-checker metrics with ServiceMonitor and alerts; upgrade shared postgres to 17 with statuspage DB and probes/resources.
2026-09-23 15:47:26 +02:00
forust 8b2cf29771 feat(tls): internal CA wildcard for *.internal routes
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
renovate-ci / validate-renovate (push) Successful in 10s
ci / build (push) Successful in 4m51s
ci / deploy-userbot-panel (push) Failing after 2s
2026-09-23 14:47:14 +02:00
forust 7b7fc3bcb0 fix(tls): drop internal certs for dormant namespaces (kener, downtify have no ns live)
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 1s
ci / lint-prettier (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
2026-09-23 14:46:09 +02:00
forust bc1e69ebe0 feat(tls): internal CA wildcard for *.internal routes
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 0s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
Selfsigned root (10y) + internal-ca issuer; per-namespace
internal-wildcard-tls certs referenced by all -local routers.
Root public cert committed for client trust stores.
2026-09-23 14:45:05 +02:00
forust f29bb3d580 revert: drop Grafana cert-manager dashboard 2026-09-23 14:42:09 +02:00
forust 1f7166026b feat(observability): cert-manager metrics, dashboard and alerts
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 13s
ci / build (push) Successful in 1s
ci / deploy-userbot-panel (push) Skipped
2026-09-23 14:32:36 +02:00
forust 1cbdfc1d6f feat(observability): cert-manager metrics, dashboard and alerts
- ServiceMonitor for cert-manager (release: prometheus-stack)
- Grafana dashboard Cert-manager-Kubernetes (ID 22908, datasource
  refs fixed) via dashboard sidecar ConfigMap
- PrometheusRule: NotReady (crit), expiry <14d (warn) / <7d
  (crit), ACME 429 rate-limit (warn)
2026-09-23 14:32:36 +02:00
forust 6be3769288 feat(tls): migrate public ingress TLS from Traefik ACME to cert-manager
ci / validate (push) Successful in 2s
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 42s
ci / deploy-userbot-panel (push) Skipped
Big-bang: 24 Certificates (HTTP-01, letsencrypt-prod) replace
Traefik certResolver on all prod IngressRoutes. Traefik
certificatesResolvers removed (its acme-http router hijacked
HTTP-01 for every host). AdGuard DoT shares the cert-manager
adguard-certs secret; sync CronJob retired. Dormant files
converted, n8n untouched (live-only deletion rule).
2026-09-23 14:16:23 +02:00
forust ef325cd3b1 feat(tls): migrate public ingress TLS from Traefik ACME to cert-manager
ci / lint-ruff (push) Successful in 1s
ci / lint-prettier (push) Successful in 3s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
All prod IngressRoutes switch tls.certResolver to tls.secretName
backed by per-router Certificates (HTTP-01, letsencrypt-prod).
adguard-prod reuses the shared adguard-certs secret (also feeds
DoT :853); sync CronJob removed as redundant.

Traefik certificatesResolvers removed: its internal
acme-http@internal router hijacks HTTP-01 for every host while
enabled, blocking external solvers. Dormant files (kener,
downtify) converted for consistency but not applied; n8n
untouched per live-only rule.
2026-09-23 14:12:36 +02:00
forust aa81f1bf8a feat(adguard): Traefik-synced TLS for AdGuard DoT + reloader
ci / validate (push) Successful in 1s
renovate-ci / validate-renovate (push) Successful in 14s
ci / build (push) Successful in 1s
ci / deploy-userbot-panel (push) Skipped
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
- CronJob mirrors Traefik prod cert dns.forust.xyz into
  adguard-certs (cert-manager HTTP-01 is hijacked by Traefik
  acme-http router, see adguardhome/k8s/cert-sync.yaml)
- reloader for auto-restart on secret rotation
- drop retired adguard.forust.xyz from prod route
- traefik: enable kubernetesIngress, drop unused staging resolver
2026-09-23 14:04:14 +02:00
forust 93d768e988 fix(adguard): split deployment RBAC rule for list/watch
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
Collection verbs cannot combine with resourceNames (grant would
be void). Instance verbs stay name-scoped to adguard-deployment;
list/watch is namespace-scoped (single Deployment in ns).
2026-09-23 13:54:06 +02:00
forust a8f7c79934 fix(adguard): sync job RBAC and idempotency
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 0s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
- grant list+watch on adguard-deployment (rollout status hung
  without it, job hit activeDeadline and failed)
- compare content digests only (old hash embedded filenames, so
  every run patched + restarted even when in sync)

Keeps explicit rollout restart alongside reloader annotation:
one extra restart per rotation (~60d) is accepted for
determinism if reloader is down.
2026-09-23 13:52:57 +02:00
forust c42bf14c9a fix(adguard): drop retired adguard.forust.xyz from prod route so dns.forust.xyz gets LE cert
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 1s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 1s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
adguard.forust.xyz is NXDOMAIN (host retired); the combo SAN cert kept
failing and dns.forust.xyz served TRAEFIK DEFAULT CERT. Scope prod route
to dns.forust.xyz only.

Also commit live traefik-values state (remove letsencrypt-staging
resolver, live since helm rev 33).
2026-09-23 13:42:04 +02:00
forust 1e8479b853 feat(adguard): sync Traefik prod cert for dns.forust.xyz into adguard-certs
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
CronJob adguard-cert-sync (daily 03:17) copies the public cert/key for
dns.forust.xyz from Traefik acme.json into Secret adguard-certs, which
AdGuard mounts for DNS-over-TLS on :853.

- least-privilege RBAC: read pods/exec in ns traefik, get/update/patch
  Secret adguard-certs and get/patch adguard-deployment in ns adguard
- script selects the PROD resolver entry only, matches main domain or
  SANs, compares sha256 hashes, patches the secret and restarts the
  deployment ONLY on change; exits non-zero and touches nothing when
  Traefik holds no cert yet (HTTP-01 currently cannot complete)
2026-09-23 13:39:38 +02:00
forust c139d700f1 feat(adguard,traefik,cert-manager,reloader): foundation for adguard cert sync
- traefik: enable kubernetesIngress provider (needed for cert-manager HTTP-01)
- adguard: add reloader auto annotation for secret-driven restarts
- cert-manager: namespace, helm values (crds), staging+prod ClusterIssuers
- reloader: namespace manifest
2026-09-23 13:19:26 +02:00
forust 67b9996911 fix(renovate): default endpoint to gitea API URL
ci / lint-prettier (push) Failing after 34s
ci / lint-ruff (push) Failing after 36s
ci / lint-yaml (push) Failing after 27s
ci / lint-dockerfiles (push) Failing after 34s
ci / validate (push) Failing after 21s
ci / build (push) Skipped
ci / deploy-userbot-panel (push) Skipped
renovate-ci / validate-renovate (push) Failing after 28s
Fall back to https://gitea.forust.xyz/api/v1 when RENOVATE_ENDPOINT is unset, in both local config and in-cluster ConfigMap.
2026-09-23 13:06:19 +02:00
forust 4e9ee567ca Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.106.0' (#41) from renovate/renovate-renovate-44.x into main
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 3s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 1s
ci / validate (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 12s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Skipped
Reviewed-on: #41
2026-09-21 20:03:24 +00:00
renovate-bot 4863e13596 chore(deps): update renovate/renovate docker tag to v44.106.0
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / lint-ruff (pull_request) Successful in 1s
ci / lint-prettier (push) Successful in 2s
ci / lint-ruff (push) Successful in 1s
ci / validate (push) Successful in 2s
ci / lint-prettier (pull_request) Successful in 3s
ci / lint-yaml (pull_request) Successful in 3s
ci / lint-dockerfiles (pull_request) Successful in 2s
ci / validate (pull_request) Successful in 2s
renovate-ci / validate-renovate (pull_request) Successful in 7s
ci / build (push) Has been skipped
ci / build (pull_request) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
2026-09-21 16:18:03 +00:00
forust 9c4580a522 fix(prometheus): exclude xui services from TraefikServiceHighLatency
ci / lint-prettier (push) Successful in 3s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 1s
ci / deploy-userbot-panel (push) Has been skipped
ci / lint-ruff (push) Successful in 2s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 3s
Long-lived VPN WebSocket sessions inflate P95 request duration; keep 5xx/down/cert alerts for xui unchanged.
2026-09-19 19:06:42 +02:00
forust 4e3ad00202 feat(xui): add 3x-ui VPN panel behind cloudflared tunnel
VLESS+WS inbound (port 10000) via Traefik IngressRoute, panel on internal domains only with public route commented out. gitignore now covers nested k8s secrets and local-only grafana values.
2026-09-19 19:06:37 +02:00
forust 86ac43567d chore(renovate): sync pins to 44.103.0
ci / lint-prettier (push) Successful in 3s
ci / lint-ruff (push) Successful in 2s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 33s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 1s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
2026-09-18 22:21:48 +02:00
forust 35bf980bda Merge branch 'main' of https://gitea.forust.xyz/forust/homelab
ci / lint-prettier (push) Successful in 4s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 2s
ci / validate (push) Successful in 2s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
2026-09-18 22:19:57 +02:00
forust 51677ae184 ci: run all lint jobs natively without docker 2026-09-18 22:18:59 +02:00
forust c9e6fc0e2b ci: run ruff and yamllint natively, cache npm
Ruff and yamllint ship in Arch repos, drop their container pulls. Prettier keeps the node container but mounts persistent npm cache. Hadolint and kubeconform stay containerized (AUR-only / hermetic pin).
2026-09-18 19:41:02 +02:00
forust ab386fc436 ci: keep gitea workflows only, harden and speed up pipelines
Drop duplicated .github/workflows (helm blocks ported to .gitea deploy first). Add ci concurrency with cancel on branches, pin checkout to SHA and prettier to 3.9.8, registry layer cache for image builds. renovate-run gains config validation and dry-run input.
2026-09-18 19:38:56 +02:00
forust 7e17ae638a Merge pull request 'chore(deps): update container patch updates' (#35) from renovate/container-patch-updates into main
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #35
2026-09-18 17:20:48 +00:00
renovate-bot 872d9695c3 chore(deps): update container patch updates 2026-09-18 17:20:48 +00:00
forust 69e7df6a63 Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.103.0' (#36) from renovate/renovate-renovate-44.x into main
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 5s
ci / lint-prettier (push) Successful in 8s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 3s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #36
2026-09-18 17:20:39 +00:00
renovate-bot 98aceef192 chore(deps): update renovate/renovate docker tag to v44.103.0 2026-09-18 17:20:39 +00:00
forust dfd9cc6fbf Merge pull request 'chore(deps): update cloudflare/cloudflared docker tag to v2026.9.1' (#37) from renovate/cloudflare-cloudflared-2026.x into main
ci / validate (push) Successful in 5s
ci / lint-prettier (push) Successful in 7s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Successful in 3s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #37
2026-09-18 17:20:31 +00:00
renovate-bot 40e7499e7c chore(deps): update cloudflare/cloudflared docker tag to v2026.9.1
ci / lint-prettier (pull_request) Successful in 9s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Successful in 8s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 6s
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (pull_request) Successful in 8s
ci / build (pull_request) Has been skipped
ci / build (push) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
2026-09-18 17:19:16 +00:00
forust 7f0bd5f609 feat(renovate): add manual run pipeline and sync configs
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
renovate-ci / validate-renovate (push) Successful in 1m27s
ci / build (push) Successful in 3s
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 5s
ci / validate (push) Successful in 5s
ci / deploy-userbot-panel (push) Has been skipped
renovate-run workflow_dispatch runs pinned renovate via docker on self-hosted runner. Sync helm-values manager into config.js/configmap, bump compose and validator pins to 44.97.2.
2026-09-18 19:15:11 +02:00
forust 043923fc64 style(crowdsec): fix prettier formatting in grafana dashboards
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 7s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 8s
ci / build (push) Successful in 3s
ci / deploy-userbot-panel (push) Has been skipped
ci / lint-dockerfiles (push) Successful in 4s
2026-09-18 19:07:46 +02:00
forust f0f8a35b0f Merge branch 'main' of https://gitea.forust.xyz/forust/homelab
ci / lint-prettier (push) Failing after 9s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
# Conflicts:
#	postgres/k8s/postgres.yaml
2026-09-18 19:05:56 +02:00
forust f5f389b440 chore(ci): deploy loki alloy and track helm values in renovate
Deploy hook installs loki 7.3.0 and alloy 1.12.1 when loki/k8s/active exists. Renovate watches k8s values files.
2026-09-18 19:02:46 +02:00
forust 7cca330438 feat(crowdsec): switch agent to loki acquisition with static identity
Read traefik logs from Loki instead of file tail. Static machine identity via pre-created secret stops 403 register races. Add janitor cronjob, dashboards, whitelists and metrics.
2026-09-18 19:02:43 +02:00
forust 5936de3e56 fix(alertmanager): quote null receiver and wire telegram template
Unquoted null parses as YAML null and breaks routing. Add shared telegram message template.
2026-09-18 19:02:41 +02:00
forust c98ea8b957 feat(monitoring): align storage to live pvc, add loki datasource and alerts
Match grafana/alertmanager storageClass to live PVCs to avoid immutable-field failures. Add Loki datasource and LokiDown/AlloyDown alerts.
2026-09-18 19:02:38 +02:00
forust 34c33697bb feat(loki): add single-binary loki and alloy stack
Filesystem storage on local-path-retain, 14d retention. Alloy daemonset ships k8s pod logs to loki-gateway.
2026-09-18 19:02:36 +02:00
forust 92bd920113 Merge pull request 'chore(deps): update postgres docker tag to v17.11' (#34) from renovate/postgres-17.x into main
ci / lint-prettier (push) Successful in 7s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #34
2026-09-17 18:04:58 +00:00
renovate-bot 8007f82d52 chore(deps): update postgres docker tag to v17.11
ci / lint-dockerfiles (pull_request) Successful in 4s
ci / lint-prettier (pull_request) Successful in 6s
ci / lint-ruff (pull_request) Successful in 4s
ci / lint-yaml (pull_request) Successful in 7s
ci / validate (pull_request) Successful in 4s
renovate-ci / validate-renovate (pull_request) Successful in 8s
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 5s
ci / build (pull_request) Has been skipped
ci / build (push) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
2026-09-17 17:59:25 +00:00
forust 60b9766449 Merge pull request 'chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v2.13.0' (#32) from renovate/ghcr.io-henriquesebastiao-downtify-2.x into main
renovate-ci / validate-renovate (push) Successful in 7s
ci / lint-prettier (push) Successful in 7s
ci / lint-ruff (push) Successful in 3s
ci / lint-yaml (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 3s
ci / validate (push) Successful in 5s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #32
2026-09-17 17:58:27 +00:00
renovate-bot 0ebb7264bc chore(deps): update ghcr.io/henriquesebastiao/downtify docker tag to v2.13.0 2026-09-17 17:58:27 +00:00
forust ce21c60eba Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.97.2' (#33) from renovate/renovate-renovate-44.x into main
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 4s
renovate-ci / validate-renovate (push) Successful in 7s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #33
2026-09-17 17:58:15 +00:00
renovate-bot 53638f831d chore(deps): update renovate/renovate docker tag to v44.97.2
ci / lint-prettier (pull_request) Successful in 8s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Successful in 7s
ci / lint-dockerfiles (pull_request) Successful in 4s
ci / validate (pull_request) Successful in 6s
renovate-ci / validate-renovate (pull_request) Successful in 8s
ci / lint-prettier (push) Successful in 7s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
ci / build (pull_request) Has been skipped
ci / build (push) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
2026-09-17 17:57:23 +00:00
forust 4953da2dd7 Merge pull request 'Feat/centralized postgres' (#26) from feat/centralized-postgres into main
ci / lint-prettier (push) Successful in 9s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 8s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 8s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #26
2026-09-17 17:55:58 +00:00
forust 4ac65f743c lint(postgres): 126:77 error no new line character at the end of file (new-line-at-end-of-file)
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 4s
ci / lint-ruff (pull_request) Successful in 5s
ci / validate (push) Successful in 6s
ci / lint-prettier (pull_request) Successful in 7s
ci / lint-yaml (pull_request) Successful in 8s
ci / lint-dockerfiles (pull_request) Successful in 7s
ci / validate (pull_request) Successful in 6s
renovate-ci / validate-renovate (pull_request) Successful in 8s
ci / build (push) Has been skipped
ci / build (pull_request) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
2026-09-17 17:55:50 +00:00
forust 8f2e9d66c8 chore(gite): updated deprecated access log config 2026-09-17 17:55:50 +00:00
forust c7d42fb90a feat(postgres): migrate gitea to shared postgres database 2026-09-17 17:55:50 +00:00
forust 003b1e5dca feat(postgres): upgrade shared database to PostgreSQL 17
Move the shared postgres service from 15.19 to 17.6 as the postgres17 StatefulSet with its own PVC, extend the initdb and ingress policy with the statuspage database, and drop the now-unused per-app postgres manifests for authentik, gitea and netronome.
2026-09-17 17:55:50 +00:00
forustandCopilot b3463705c3 feat(postgres): migrate apps to shared database
Move Authentik and Netronome to the shared PostgreSQL service after logical dump and restore.

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
2026-09-17 17:55:50 +00:00
forust 52e1f50a80 feat(postgres): add shared database deployments 2026-09-17 17:55:50 +00:00
forust cdc2f10fe8 Merge pull request 'chore(deps): update container patch updates' (#30) from renovate/container-patch-updates into main
ci / lint-prettier (push) Successful in 7s
ci / lint-yaml (push) Successful in 7s
ci / lint-ruff (push) Successful in 5s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 3s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #30
2026-09-17 17:55:32 +00:00
renovate-bot 89fbdef10e chore(deps): update container patch updates 2026-09-17 17:55:32 +00:00
forust ac795feeed Merge pull request 'chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.15' (#31) from renovate/ghcr.io-alexta69-metube-2026.x into main
ci / validate (push) Successful in 5s
renovate-ci / validate-renovate (push) Successful in 8s
ci / lint-prettier (push) Successful in 7s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 8s
ci / lint-dockerfiles (push) Successful in 4s
ci / build (push) Successful in 3s
ci / deploy-userbot-panel (push) Has been skipped
Reviewed-on: #31
2026-09-17 17:55:11 +00:00
renovate-bot 3af5ecd07f chore(deps): update ghcr.io/alexta69/metube docker tag to v2026.09.15
ci / lint-prettier (pull_request) Successful in 8s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Successful in 7s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 5s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (pull_request) Successful in 9s
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 6s
ci / build (pull_request) Has been skipped
ci / build (push) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
2026-09-17 17:54:43 +00:00
forust 82949613db fix(cloudflared): drop bogus exec livenessProbe that SIGTERMed the tunnel 2026-09-17 19:34:28 +02:00
forust 47d788ca13 feat(cloudflared): add Cloudflare Tunnel deployment (token-based, cfddns-style) 2026-09-17 19:09:02 +02:00
forust 77be606912 Merge pull request 'chore(deps): update grafana/grafana docker tag to v13.2.2' (#28) from renovate/container-patch-updates into main
ci / build (push) Successful in 2s
ci / deploy-userbot-panel (push) Has been skipped
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 9s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 8s
Reviewed-on: #28
2026-09-15 12:07:56 +00:00
renovate-bot 4eab6a43c8 chore(deps): update grafana/grafana docker tag to v13.2.2 2026-09-15 12:07:56 +00:00
forust 6c24d4fb17 Merge pull request 'chore(deps): update renovate/renovate docker tag to v44.91.0' (#27) from renovate/renovate-renovate-44.x into main
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 7s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
ci / deploy-userbot-panel (push) Has been skipped
renovate-ci / validate-renovate (push) Successful in 9s
ci / build (push) Successful in 2s
Reviewed-on: #27
2026-09-15 12:07:44 +00:00
renovate-bot e36595a045 chore(deps): update renovate/renovate docker tag to v44.91.0 2026-09-15 12:07:44 +00:00
forust e07537ff8b Merge pull request 'chore(deps): update docker.n8n.io/n8nio/n8n docker tag to v2.40.0' (#29) from renovate/docker.n8n.io-n8nio-n8n-2.x into main
ci / lint-prettier (push) Successful in 13s
ci / lint-ruff (push) Successful in 5s
ci / lint-yaml (push) Successful in 8s
ci / lint-dockerfiles (push) Successful in 5s
ci / validate (push) Successful in 6s
renovate-ci / validate-renovate (push) Successful in 11s
ci / deploy-userbot-panel (push) Has been skipped
ci / build (push) Successful in 2s
Reviewed-on: #29
2026-09-15 12:07:27 +00:00
renovate-bot 303eaaa71b chore(deps): update docker.n8n.io/n8nio/n8n docker tag to v2.40.0
ci / lint-prettier (pull_request) Successful in 9s
renovate-ci / validate-renovate (pull_request) Successful in 9s
ci / lint-ruff (pull_request) Successful in 5s
ci / lint-yaml (pull_request) Successful in 7s
ci / lint-dockerfiles (pull_request) Successful in 5s
ci / validate (pull_request) Successful in 6s
ci / lint-prettier (push) Successful in 8s
ci / lint-ruff (push) Successful in 4s
ci / lint-yaml (push) Successful in 6s
ci / lint-dockerfiles (push) Successful in 4s
ci / validate (push) Successful in 5s
ci / build (push) Has been skipped
ci / build (pull_request) Has been skipped
ci / deploy-userbot-panel (pull_request) Has been skipped
ci / deploy-userbot-panel (push) Has been skipped
2026-09-15 10:18:03 +00:00
163 changed files with 4222 additions and 880 deletions

No files matched your search

+35 -56
View File
@@ -7,6 +7,10 @@ on:
pull_request:
workflow_dispatch:
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
env:
REGISTRY: gcr.forust.xyz
@@ -15,7 +19,7 @@ jobs:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Check formatting with Prettier
shell: bash
@@ -31,32 +35,24 @@ jobs:
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
node:22-alpine \
sh -lc 'npx --yes prettier@3 --check --ignore-unknown "$@"' sh "${prettier_files[@]}"
prettier --check --ignore-unknown "${prettier_files[@]}"
lint-ruff:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint Python with Ruff
shell: bash
run: |
docker run --rm \
-v "$PWD:/work" \
-w /work \
ghcr.io/astral-sh/ruff:latest \
check .
ruff check .
lint-yaml:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint YAML syntax
shell: bash
@@ -72,17 +68,13 @@ jobs:
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
cytopia/yamllint:latest \
-c .yamllint "${yaml_files[@]}"
yamllint -c .yamllint "${yaml_files[@]}"
lint-dockerfiles:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint Dockerfiles
shell: bash
@@ -96,18 +88,13 @@ jobs:
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
--entrypoint hadolint \
hadolint/hadolint:latest-debian \
-c .hadolint.yaml "${dockerfiles[@]}"
hadolint -c .hadolint.yaml "${dockerfiles[@]}"
validate:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Validate Kubernetes manifests
shell: bash
@@ -122,10 +109,7 @@ jobs:
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
ghcr.io/yannh/kubeconform:latest \
kubeconform \
-strict \
-ignore-missing-schemas \
-summary \
@@ -139,7 +123,7 @@ jobs:
services: ${{ steps.services.outputs.services }}
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
fetch-depth: 0
@@ -230,7 +214,10 @@ jobs:
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" dtek_notif
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" dtek_notif
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
@@ -250,7 +237,10 @@ jobs:
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" errorpages
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" errorpages
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
@@ -280,7 +270,10 @@ jobs:
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" "$context"
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" "$context"
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
@@ -309,7 +302,10 @@ jobs:
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
@@ -340,7 +336,10 @@ jobs:
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" "$context"
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" "$context"
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
@@ -348,23 +347,3 @@ jobs:
;;
esac
done
deploy-userbot-panel:
needs: build
if: github.ref_name == 'main' && contains(needs.build.outputs.services, 'userbot')
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Apply and roll out userbot panel
shell: bash
run: |
kubectl apply -f userbot/k8s/base/panel.yaml
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
# Keep legacy deployments (forust/anna) in sync with manifests; they have no replicas field, so apply leaves scaling to the user manager only.
kubectl apply -f userbot/k8s/base/userbots.yaml
kubectl rollout restart deployment/userbot-panel -n userbot
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
+241
View File
@@ -0,0 +1,241 @@
#!/usr/bin/env bash
# Shared stages for the deploy workflow. Runs on the workstation, invoked as:
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
# source "$REPO/.gitea/workflows/deploy-lib.sh"
# run_stage "$STAGE"
# EOF
set -euo pipefail
: "${REPO:?REPO must be set}"
APPLY_PRUNE="${APPLY_PRUNE:-false}"
log() {
echo "== $* =="
}
collect_k8s() {
git -C "$REPO" ls-files -- "$1" \
| grep -E '\.ya?ml$' \
| grep -Ev '/overlays/' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
| sort
}
kustomize_overlay() {
if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
echo "$1/overlays/prod"
elif [ -f "$1/base/kustomization.yaml" ]; then
echo "$1/base"
fi
}
select_manifests() {
K8S_MANIFESTS=()
KUSTOMIZE_APPS=()
COMPOSE_STACKS=()
local kd_rel kd overlay cf_rel cf f
while IFS= read -r kd_rel; do
kd="$REPO/$kd_rel"
if [ ! -f "$kd/active" ]; then
echo "skip (no k8s/active): $kd_rel"
continue
fi
overlay="$(kustomize_overlay "$kd" || true)"
if [ -n "${overlay:-}" ]; then
echo "kustomize app: ${overlay#$REPO/}"
KUSTOMIZE_APPS+=("$overlay")
else
while IFS= read -r f; do
[ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f")
done < <(collect_k8s "$kd_rel" || true)
fi
done < <(
git -C "$REPO" ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
while IFS= read -r cf_rel; do
cf="$REPO/$cf_rel"
if [ -f "$(dirname "$cf")/active" ]; then
echo "compose: $cf_rel"
COMPOSE_STACKS+=("$cf")
else
echo "skip (no root active): $cf_rel"
fi
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
}
stage_preflight() {
if [ ! -d "$REPO/.git" ]; then
echo "Repository not found at $REPO"
exit 1
fi
git -C "$REPO" fetch origin main
log "Workstation state"
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
echo " remote: $(git -C "$REPO" rev-parse --short origin/main)"
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
echo "ERROR: workstation has local tracked modifications, refusing reset:"
git -C "$REPO" status --porcelain --untracked-files=no
git -C "$REPO" diff --stat
exit 1
fi
git -C "$REPO" reset --hard origin/main
}
stage_validate() {
cd "$REPO"
select_manifests
local m k cf
log "Validate compose stacks"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " config: $cf"
docker compose -f "$cf" config --quiet
done
log "Validate k8s manifests (kubectl dry-run=client)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
kubectl apply --dry-run=client -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=client >/dev/null
done
log "Validate k8s manifests (kubectl dry-run=server)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
kubectl apply --dry-run=server -f "$m" >/dev/null
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl apply -k "$k" --dry-run=server >/dev/null
done
log "Checking referenced Secrets exist"
echo " (deploy never applies *secret*.yaml; create missing ones from the laptop)"
local ref_secrets=() missing_secrets=() all_secrets s
if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then
while IFS= read -r s; do
[ -n "$s" ] && ref_secrets+=("$s")
done < <(
{
grep -h -A1 -E 'secretRef:|secretKeyRef:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
grep -h -E 'secretName:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
} | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true
)
fi
all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)"
for s in ${ref_secrets[@]+"${ref_secrets[@]}"}; do
if printf '%s\n' "$all_secrets" | grep -qx "$s"; then
echo " ok: $s"
else
echo " MISSING: $s"
missing_secrets+=("$s")
fi
done
if [ "${#missing_secrets[@]}" -gt 0 ]; then
echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:"
printf ' - %s\n' "${missing_secrets[@]}"
echo "Create them manually from the laptop, e.g.:"
echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example"
exit 1
fi
}
stage_apply_k8s() {
cd "$REPO"
select_manifests >/dev/null
local ns_files=() other_files=() m k prune_opts=()
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
case "$m" in
*/namespace.y?ml) ns_files+=("$m") ;;
*) other_files+=("$m") ;;
esac
done
if [ "$APPLY_PRUNE" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
if [ "${#ns_files[@]}" -gt 0 ]; then
log "Applying namespaces (${#ns_files[@]} files)"
for m in "${ns_files[@]}"; do
kubectl apply -f "$m"
done
fi
if [ -f "$REPO/prometheus-stack/k8s/active" ]; then
if [ ! -f "$REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
exit 1
fi
log "Upgrading kube-prometheus-stack"
helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \
--namespace prometheus \
--version 86.2.3 \
--values "$REPO/prometheus-stack/k8s/grafana-values.yaml" \
--wait --timeout 10m
fi
if [ -f "$REPO/loki/k8s/active" ]; then
log "Upgrading loki/alloy"
helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true
helm repo update grafana >/dev/null 2>&1 || true
helm upgrade --install loki grafana/loki \
--version 7.3.0 \
--namespace prometheus \
--values "$REPO/loki/k8s/loki-values.yaml" \
--wait --timeout 10m
helm upgrade --install alloy grafana/alloy \
--version 1.12.1 \
--namespace prometheus \
--values "$REPO/loki/k8s/alloy-values.yaml" \
--wait --timeout 10m
fi
if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files)"
for m in "${other_files[@]}"; do
kubectl apply "${prune_opts[@]}" -f "$m"
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#$REPO/}"
kubectl apply -k "$k"
done
if [ -f "$REPO/userbot/k8s/active" ]; then
log "userbot panel hook"
if kubectl get secret userbot-common-secrets -n userbot >/dev/null 2>&1; then
echo " userbot-common-secrets already present in userbot ns, not touching"
elif kubectl get secret userbot-common-secrets -n default >/dev/null 2>&1; then
echo " bootstrapping userbot-common-secrets into userbot ns"
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
else
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
fi
kubectl rollout restart deployment/userbot-panel -n userbot
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
fi
}
stage_apply_compose() {
cd "$REPO"
select_manifests >/dev/null
local cf
log "Redeploying docker compose stacks (${#COMPOSE_STACKS[@]} stacks)"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " compose: $cf"
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
docker compose -f "$cf" build
docker compose -f "$cf" push
fi
docker compose -f "$cf" up -d --pull always --remove-orphans
done
}
run_stage() {
case "${1:?stage required}" in
preflight) stage_preflight ;;
validate) stage_validate ;;
apply-k8s) stage_apply_k8s ;;
apply-compose) stage_apply_compose ;;
*)
echo "ERROR: unknown stage: $1"
exit 1
;;
esac
}
+46 -127
View File
@@ -1,152 +1,71 @@
name: deploy
on:
push:
branches:
- main
workflow_dispatch:
concurrency:
group: deploy-main
cancel-in-progress: false
jobs:
redeploy:
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Redeploy workstation
shell: bash
env:
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
# Set APPLY_PRUNE=true to enable kubectl apply --prune. Requires every
# manifest to carry label app.kubernetes.io/managed-by=homelab-deploy,
# otherwise previously applied resources get deleted on the next run.
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
jobs:
preflight:
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Fetch and reset workstation
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh preflight
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
: "${DEPLOY_USER:?missing DEPLOY_USER}"
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
validate:
needs: [preflight]
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
deploy_port="${DEPLOY_PORT:-22}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
ssh_key="$RUNNER_TEMP/deploy_key"
mkdir -p "$RUNNER_TEMP"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key"
ssh_opts=(
-i "$ssh_key"
-p "$deploy_port"
-o BatchMode=yes
-o StrictHostKeyChecking=accept-new
)
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
"DEPLOY_PATH=$(printf '%q' \"$deploy_path\") APPLY_PRUNE=$(printf '%q' \"${APPLY_PRUNE:-false}\") bash -se" <<'EOF'
- name: Dry-run manifests and check Secrets
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh validate
repo="${DEPLOY_PATH:-/srv/homelab}"
apply-k8s:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
if [ ! -d "$repo/.git" ]; then
echo "Repository not found at $repo"
exit 1
fi
- name: Apply Kubernetes manifests
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh apply-k8s
git -C "$repo" fetch origin main
git -C "$repo" reset --hard origin/main
apply-compose:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# Runtime selection: a service is k8s-managed when $SERVICE/k8s/active
# exists. Otherwise it is compose-managed, and only k8s/routing/*
# manifests (external Services / EndpointSlices / ServersTransport /
# Ingresses that route to docker backends) are applied.
# migrate: touch SERVICE/k8s/active (+ move routing files up)
# rollback: rm SERVICE/k8s/active
collect_k8s() {
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
! -path '*/routing/*' ! -path '*/overlays/*' \
! -name 'kustomization.y*ml' ! -name '*.example.y*ml' \
! -name '*values.y*ml' ! -name 'patch-*.y*ml' \
| sort
}
collect_k8s_inactive() {
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
\( -name 'namespace.y*ml' -o -path '*/routing/*' \) \
! -path '*/overlays/*' ! -name '*.example.y*ml' \
| sort
}
mapfile -t compose_stacks < <(
find "$repo" -type f \( -name 'compose.yaml' -o -name 'compose.yml' \) | sort
)
mapfile -t k8s_manifests < <(
for kd in $(find "$repo" -type d -name k8s ! -path '*/.git/*' | sort); do
if [ -f "$kd/active" ]; then
collect_k8s "$kd"
else
collect_k8s_inactive "$kd"
fi
done
)
echo "== Validate compose stacks =="
for cf in "${compose_stacks[@]}"; do
dir=$(dirname "$cf")
if [ -f "$dir/k8s/active" ]; then
echo " skip (k8s-managed): $dir"
continue
fi
echo " config: $cf"
docker compose -f "$cf" config --quiet
done
echo "== Validate k8s manifests (kubectl dry-run) =="
for m in "${k8s_manifests[@]}"; do
echo " apply --dry-run=client $m"
kubectl apply --dry-run=client -f "$m" >/dev/null
done
echo "== Applying Kubernetes manifests =="
ns_files=()
other_files=()
for m in "${k8s_manifests[@]}"; do
case "$m" in
*/namespace.y?ml) ns_files+=("$m") ;;
*) other_files+=("$m") ;;
esac
done
prune_opts=()
if [ "${APPLY_PRUNE:-false}" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
if [ "${#ns_files[@]}" -gt 0 ]; then
echo " namespaces first: ${ns_files[*]}"
kubectl apply -f "${ns_files[@]}"
fi
if [ "${#other_files[@]}" -gt 0 ]; then
echo " resources: ${other_files[*]}"
kubectl apply "${prune_opts[@]}" -f "${other_files[@]}"
fi
echo "== Redeploying docker compose stacks =="
for cf in "${compose_stacks[@]}"; do
dir=$(dirname "$cf")
if [ -f "$dir/k8s/active" ]; then
echo " skip (k8s-managed): $dir"
continue
fi
echo " compose: $dir"
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
docker compose -f "$cf" build
docker compose -f "$cf" push
fi
docker compose -f "$cf" up -d --pull always --remove-orphans
done
EOF
- name: Redeploy docker compose stacks
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh apply-compose
+2 -2
View File
@@ -12,7 +12,7 @@ jobs:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Validate Renovate Compose draft
shell: bash
@@ -48,5 +48,5 @@ jobs:
docker run --rm \
-v "$PWD:/work" \
-w /work \
renovate/renovate:44.83.2 \
renovate/renovate:44.103.0 \
renovate-config-validator renovate.json
+69
View File
@@ -0,0 +1,69 @@
name: renovate-run
on:
workflow_dispatch:
inputs:
repositories:
description: "Repositories to scan (comma-separated)"
required: false
default: "forust/homelab"
log_level:
description: "Renovate log level"
required: false
default: "info"
type: choice
options:
- info
- debug
dry_run:
description: "Plan only, do not open or update PRs"
required: false
default: false
type: boolean
concurrency:
group: renovate-run
cancel-in-progress: false
jobs:
run-renovate:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Validate Renovate config
shell: bash
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
renovate/renovate:44.103.0 \
renovate-config-validator
- name: Run Renovate
shell: bash
env:
RENOVATE_TOKEN: ${{ secrets.RENOVATE_TOKEN }}
RENOVATE_GITHUB_COM_TOKEN: ${{ secrets.RENOVATE_GITHUB_COM_TOKEN }}
RENOVATE_REPOSITORIES: ${{ inputs.repositories }}
RENOVATE_DRY_RUN: ${{ inputs.dry_run && 'full' || '' }}
LOG_LEVEL: ${{ inputs.log_level }}
run: |
set -euo pipefail
: "${RENOVATE_TOKEN:?missing RENOVATE_TOKEN secret — add a renovate-bot PAT in repo/org Actions secrets}"
docker run --rm \
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
-e RENOVATE_PLATFORM=gitea \
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
-e RENOVATE_DRY_RUN="${RENOVATE_DRY_RUN:-}" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
-e RENOVATE_BASE_DIR=/tmp/renovate \
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
renovate/renovate:44.103.0
+25
View File
@@ -0,0 +1,25 @@
#!/usr/bin/env bash
# usage: ssh-run.sh <stage>
# Runs one deploy-lib.sh stage on the workstation over SSH.
set -euo pipefail
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
: "${DEPLOY_USER:?missing DEPLOY_USER}"
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
deploy_port="${DEPLOY_PORT:-22}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
ssh_key="$RUNNER_TEMP/deploy_key"
mkdir -p "$RUNNER_TEMP"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key"
ssh -i "$ssh_key" -p "$deploy_port" \
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
"${DEPLOY_USER}@${DEPLOY_HOST}" \
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} STAGE=$1 bash -se" <<'EOF'
source "$REPO/.gitea/workflows/deploy-lib.sh"
run_stage "$STAGE"
EOF
-370
View File
@@ -1,370 +0,0 @@
name: ci
on:
push:
branches:
- "**"
pull_request:
workflow_dispatch:
env:
REGISTRY: gcr.forust.xyz
jobs:
lint-prettier:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Check formatting with Prettier
shell: bash
run: |
mapfile -t prettier_files < <(
git ls-files \
| grep -E '\.(md|json|ya?ml|html|css)$' \
| grep -Ev '^(\.docs/|\.zed/|errorpages/html/|homepages/(forust_files|xdfnx_files)/)'
)
if [ "${#prettier_files[@]}" -eq 0 ]; then
echo "No Prettier-managed files found."
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
node:22-alpine \
sh -lc 'npx --yes prettier@3 --check --ignore-unknown "$@"' sh "${prettier_files[@]}"
lint-ruff:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Lint Python with Ruff
shell: bash
run: |
docker run --rm \
-v "$PWD:/work" \
-w /work \
ghcr.io/astral-sh/ruff:latest \
check .
lint-yaml:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Lint YAML syntax
shell: bash
run: |
mapfile -t yaml_files < <(
git ls-files '*.yaml' '*.yml' \
':!node_modules/**' \
':!**/.venv/**'
)
if [ "${#yaml_files[@]}" -eq 0 ]; then
echo "No YAML files found."
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
cytopia/yamllint:latest \
-c .yamllint "${yaml_files[@]}"
lint-dockerfiles:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Lint Dockerfiles
shell: bash
run: |
mapfile -t dockerfiles < <(
git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*'
)
if [ "${#dockerfiles[@]}" -eq 0 ]; then
echo "No Dockerfiles found."
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
--entrypoint hadolint \
hadolint/hadolint:latest-debian \
-c .hadolint.yaml "${dockerfiles[@]}"
validate:
runs-on: [self-hosted, linux, arch, homelab]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Validate Kubernetes manifests
shell: bash
run: |
mapfile -t manifests < <(
git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
)
if [ "${#manifests[@]}" -eq 0 ]; then
echo "No Kubernetes manifests found."
exit 0
fi
docker run --rm \
-v "$PWD:/work" \
-w /work \
ghcr.io/yannh/kubeconform:latest \
-strict \
-ignore-missing-schemas \
-summary \
"${manifests[@]}"
build:
needs: [lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
runs-on: [self-hosted, linux, arch, homelab]
outputs:
services: ${{ steps.services.outputs.services }}
steps:
- name: Checkout repository
uses: actions/checkout@v4
with:
fetch-depth: 0
- name: Detect changed docker-built services
id: services
shell: bash
run: |
base="${{ github.event.before }}"
if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then
base="$(git rev-list --max-parents=0 HEAD)"
fi
mapfile -t changed_files < <(git diff --name-only "$base" "${GITHUB_SHA}")
services=()
add_service() {
local name="$1"
local seen=0
for existing in "${services[@]}"; do
if [ "$existing" = "$name" ]; then
seen=1
break
fi
done
if [ "$seen" -eq 0 ]; then
services+=("$name")
fi
}
for file in "${changed_files[@]}"; do
case "$file" in
dtek_notif/*)
add_service dtek_notif
;;
errorpages/*)
add_service errorpages
;;
userbot/*)
add_service userbot
;;
homepages/*)
add_service homepages
;;
edu_master/phpsessid-bot/*|edu_master/webinar-checker/*|edu_master/compose.yaml)
add_service edu_master
;;
esac
done
if [ "${#services[@]}" -eq 0 ]; then
echo "No docker-built services changed."
echo "services=" >> "$GITHUB_OUTPUT"
exit 0
fi
printf '%s\n' "${services[@]}" | tee /tmp/services.txt
echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT"
- name: Log in to registry
if: steps.services.outputs.services != ''
shell: bash
run: |
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login "${REGISTRY}" \
-u "${{ secrets.REGISTRY_USERNAME }}" \
--password-stdin
- name: Build and push changed images
if: steps.services.outputs.services != ''
shell: bash
run: |
IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}"
for service in "${services[@]}"; do
case "$service" in
dtek_notif)
image="${REGISTRY}/forust/dtek-notif"
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" dtek_notif
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
;;
errorpages)
image="${REGISTRY}/forust/error-pages"
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" errorpages
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
;;
userbot)
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
for target in runtime panel; do
case "$target" in
runtime)
context="userbot"
image="${REGISTRY}/forust/userbot"
;;
panel)
context="userbot/panel"
image="${REGISTRY}/forust/userbot-panel"
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" "$context"
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
done
;;
homepages)
for service in forust xdfnx; do
case "$service" in
forust)
image="${REGISTRY}/forust/forust-homepage"
;;
xdfnx)
image="${REGISTRY}/forust/xdfnx-homepage"
;;
esac
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
done
;;
edu_master)
for service in session-keeper webinar-checker; do
case "$service" in
session-keeper)
context="edu_master/phpsessid-bot"
image="${REGISTRY}/forust/session-keeper"
;;
webinar-checker)
context="edu_master/webinar-checker"
image="${REGISTRY}/forust/webinar-checker"
;;
esac
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build "${build_args[@]}" "$context"
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
done
;;
esac
done
deploy-userbot-panel:
needs: build
if: github.ref_name == 'main' && contains(needs.build.outputs.services, 'userbot')
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Apply and roll out userbot panel
shell: bash
run: |
kubectl apply -f userbot/k8s/base/panel.yaml
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
# Keep legacy deployments (forust/anna) in sync with manifests; they have no replicas field, so apply leaves scaling to the user manager only.
kubectl apply -f userbot/k8s/base/userbots.yaml
kubectl rollout restart deployment/userbot-panel -n userbot
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
-161
View File
@@ -1,161 +0,0 @@
name: deploy
on:
workflow_dispatch:
concurrency:
group: deploy-main
cancel-in-progress: false
jobs:
redeploy:
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Redeploy workstation
shell: bash
env:
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
# Set APPLY_PRUNE=true to enable kubectl apply --prune. Requires every
# manifest to carry label app.kubernetes.io/managed-by=homelab-deploy,
# otherwise previously applied resources get deleted on the next run.
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
run: |
set -euo pipefail
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
: "${DEPLOY_USER:?missing DEPLOY_USER}"
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
deploy_port="${DEPLOY_PORT:-22}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
ssh_key="$RUNNER_TEMP/deploy_key"
mkdir -p "$RUNNER_TEMP"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key"
ssh_opts=(
-i "$ssh_key"
-p "$deploy_port"
-o BatchMode=yes
-o StrictHostKeyChecking=accept-new
)
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
"DEPLOY_PATH=$(printf '%q' \"$deploy_path\") APPLY_PRUNE=$(printf '%q' \"${APPLY_PRUNE:-false}\") bash -se" <<'EOF'
set -euo pipefail
repo="${DEPLOY_PATH:-/srv/homelab}"
if [ ! -d "$repo/.git" ]; then
echo "Repository not found at $repo"
exit 1
fi
git -C "$repo" fetch origin main
git -C "$repo" reset --hard origin/main
# Runtime selection: a service is k8s-managed when $SERVICE/k8s/active
# exists. Otherwise it is compose-managed, and only k8s/routing/*
# manifests (external Services / EndpointSlices / ServersTransport /
# Ingresses that route to docker backends) are applied.
# migrate: touch SERVICE/k8s/active (+ move routing files up)
# rollback: rm SERVICE/k8s/active
collect_k8s() {
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
! -path '*/routing/*' ! -path '*/overlays/*' \
! -name 'kustomization.y*ml' ! -name '*.example.y*ml' \
! -name '*values.y*ml' ! -name 'patch-*.y*ml' \
| sort
}
collect_k8s_inactive() {
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
\( -name 'namespace.y*ml' -o -path '*/routing/*' \) \
! -path '*/overlays/*' ! -name '*.example.y*ml' \
| sort
}
mapfile -t compose_stacks < <(
find "$repo" -type f \( -name 'compose.yaml' -o -name 'compose.yml' \) | sort
)
mapfile -t k8s_manifests < <(
for kd in $(find "$repo" -type d -name k8s ! -path '*/.git/*' | sort); do
if [ -f "$kd/active" ]; then
collect_k8s "$kd"
else
collect_k8s_inactive "$kd"
fi
done
)
echo "== Validate compose stacks =="
for cf in "${compose_stacks[@]}"; do
dir=$(dirname "$cf")
if [ -f "$dir/k8s/active" ]; then
echo " skip (k8s-managed): $dir"
continue
fi
echo " config: $cf"
docker compose -f "$cf" config --quiet
done
echo "== Validate k8s manifests (kubectl dry-run) =="
for m in "${k8s_manifests[@]}"; do
echo " apply --dry-run=client $m"
kubectl apply --dry-run=client -f "$m" >/dev/null
done
echo "== Applying Kubernetes manifests =="
ns_files=()
other_files=()
for m in "${k8s_manifests[@]}"; do
case "$m" in
*/namespace.y?ml) ns_files+=("$m") ;;
*) other_files+=("$m") ;;
esac
done
prune_opts=()
if [ "${APPLY_PRUNE:-false}" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
if [ "${#ns_files[@]}" -gt 0 ]; then
echo " namespaces first: ${ns_files[*]}"
kubectl apply -f "${ns_files[@]}"
fi
if [ -f "$repo/prometheus-stack/k8s/active" ]; then
echo "== Upgrading kube-prometheus-stack =="
helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \
--namespace prometheus \
--version 86.2.3 \
--values "$repo/prometheus-stack/k8s/grafana-values.yaml" \
--wait
fi
if [ "${#other_files[@]}" -gt 0 ]; then
echo " resources: ${other_files[*]}"
kubectl apply "${prune_opts[@]}" -f "${other_files[@]}"
fi
echo "== Redeploying docker compose stacks =="
for cf in "${compose_stacks[@]}"; do
dir=$(dirname "$cf")
if [ -f "$dir/k8s/active" ]; then
echo " skip (k8s-managed): $dir"
continue
fi
echo " compose: $dir"
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
docker compose -f "$cf" build
docker compose -f "$cf" push
fi
docker compose -f "$cf" up -d --pull always --remove-orphans
done
EOF
+7
View File
@@ -21,6 +21,9 @@ checkmk/checkmk/*
downtify/Downtify_downloads
headscale/config/*
headscale/data/*
# NetBird local hostnames and generated secrets
netbird/.env
netbird/secrets/
searxng/core-config/*
# Steaming services files
@@ -104,6 +107,10 @@ temp/*
# kubernetes
*/k8s/*secret*
!*/k8s/*secret*.example
**/k8s/*secret*
!**/k8s/*secret*.example
# Local-only tweaks, not for upstream
prometheus-stack/k8s/grafana-values.yaml
traefik/k8s/local-tls.yaml
converters/k8s/config.yaml
convertx/k8s/config.yaml
+2
View File
@@ -62,6 +62,8 @@ spec:
metadata:
labels:
app: adguard
annotations:
reloader.stakater.com/auto: "true"
spec:
containers:
- name: adguard
+12
View File
@@ -0,0 +1,12 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: adguard-certs
namespace: adguard
spec:
secretName: adguard-certs
dnsNames:
- dns.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
+5 -3
View File
@@ -7,7 +7,7 @@ spec:
entryPoints:
- websecure
routes:
- match: Host(`adguard.forust.xyz`) || Host(`dns.forust.xyz`)
- match: Host(`dns.forust.xyz`)
kind: Rule
middlewares:
- name: crowdsec-bouncer
@@ -15,13 +15,13 @@ spec:
services:
- name: adguard-service
port: 3000
- match: (Host(`adguard.forust.xyz`) || Host(`dns.forust.xyz`)) && PathPrefix(`/dns-query`)
- match: (Host(`dns.forust.xyz`)) && PathPrefix(`/dns-query`)
kind: Rule
services:
- name: adguard-service
port: 3000
tls:
certResolver: letsencrypt
secretName: adguard-certs
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -42,3 +42,5 @@ spec:
services:
- name: adguard-service
port: 3000
tls:
secretName: internal-wildcard-tls
+15
View File
@@ -0,0 +1,15 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: adguard
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+2 -2
View File
@@ -41,7 +41,7 @@ spec:
spec:
containers:
- name: authentik-server
image: ghcr.io/goauthentik/server:2026.8.2
image: ghcr.io/goauthentik/server:2026.8.3
args: ["server"]
envFrom:
- configMapRef:
@@ -75,7 +75,7 @@ spec:
spec:
containers:
- name: authentik-worker
image: ghcr.io/goauthentik/server:2026.8.2
image: ghcr.io/goauthentik/server:2026.8.3
args: ["worker"]
securityContext:
runAsUser: 0
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: authentik-prod-tls
namespace: authentik
spec:
secretName: authentik-prod-tls
dnsNames:
- auth.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: authentik
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+3 -1
View File
@@ -16,7 +16,7 @@ spec:
- name: authentik-server-service
port: 9000
tls:
certResolver: letsencrypt
secretName: authentik-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -32,3 +32,5 @@ spec:
services:
- name: authentik-server-service
port: 9000
tls:
secretName: internal-wildcard-tls
@@ -0,0 +1,9 @@
crds:
enabled: true
prometheus:
servicemonitor:
enabled: true
interval: 60s
scrapeTimeout: 30s
labels:
release: prometheus-stack
+29
View File
@@ -0,0 +1,29 @@
apiVersion: cert-manager.io/v1
kind: ClusterIssuer
metadata:
name: letsencrypt-staging
spec:
acme:
email: bobrovod@national.shitposting.agency
server: https://acme-staging-v02.api.letsencrypt.org/directory
privateKeySecretRef:
name: letsencrypt-staging-account-key
solvers:
- http01:
ingress:
class: traefik
---
apiVersion: cert-manager.io/v1
kind: ClusterIssuer
metadata:
name: letsencrypt-prod
spec:
acme:
email: bobrovod@national.shitposting.agency
server: https://acme-v02.api.letsencrypt.org/directory
privateKeySecretRef:
name: letsencrypt-prod-account-key
solvers:
- http01:
ingress:
class: traefik
@@ -0,0 +1,30 @@
-----BEGIN CERTIFICATE-----
MIIFFjCCAv6gAwIBAgIUetKpTfEDOn2985FFMu6G26itT+wwDQYJKoZIhvcNAQEN
BQAwIzEhMB8GA1UEAxMYaG9tZWxhYiBpbnRlcm5hbCByb290IENBMB4XDTI2MDky
MzEyNDA0N1oXDTM2MDkyMDEyNDA0N1owIzEhMB8GA1UEAxMYaG9tZWxhYiBpbnRl
cm5hbCByb290IENBMIICIjANBgkqhkiG9w0BAQEFAAOCAg8AMIICCgKCAgEAvmNP
ZCOoD8NtNuYJKVXBlTPjX7D7sJCSK5neH7ZbYV5+lmUlEErY8Mik7j37V5k5NfpF
Ig85pOjP7RckTPz5V6ek3yaN40s4AL053sN5ZPauDVYjalaEHTgj5sEMqlLACQWI
yZmJOZspZykae8dIpQnqCoFpRT4FurJ78v4a0ylnFVLMQn/lyCHedwTjkEdtYWYr
ccJy8vQwqkzs/rWvEH1lDqZhennLOrmcCjfonG7D/pruMn4z+6E28p4+ejkRrI6x
luak3KnpT1XMeHtgU21hiRGaMDBchHMFgAhnY1qosymKenXvfTZItwgjZbwa1hJI
GAiDm+jQDKMjzRZ3rH6Xfc0auUcykNz73PpNu1NGm78nndXwCXcXn1LFKNQJ+r1U
sJiyAmUZmXVn4aM4OMf2F38k7wTYIKg7nRGaUkNeKDlNkjA4HvgWw+jwO1KmdHQ/
mOem1rosDWHRK01wg+Gga9mQCnhNhxglg3t/UeSic6uOaRsvaz4qkzHq8MbCujVz
DpKQjqdikYOAXZOs4KlBLWrS7NaK4NzfSD02pBUErh54ruJfY/bWz9KyXzBD/lQZ
VUTKyvUVB0bkVHEdf1jJmX3H4IZRQSF5JPqOBotW6bJI5fEGNBvj9Zxy4nm2WWGz
yyP3uWsQz8U/Wdx9nXZLHInTZBsvgLYtKUAWA30CAwEAAaNCMEAwDgYDVR0PAQH/
BAQDAgKkMA8GA1UdEwEB/wQFMAMBAf8wHQYDVR0OBBYEFEkKm2rxPaK6+O9WD80z
BLC6F9QsMA0GCSqGSIb3DQEBDQUAA4ICAQAnFyHz97Umf5VIu+dKTJid7C73VugJ
TIar/xJBs/4CxP+znBxhJjXygRoyIfzoVGWcB2ZSL//vL78Qlts79K/Imc9a4RFF
wMvCxsRXAEQ4TpeWi3ophPNcs4rhsP+gQKQFtnyKP9519bqpfxp0bTqwOV2o18fn
za7rlQViiEnNV58j7CVoM9+mJvVVfBEX1Km+GyJL9GadzbIQ7FxClVJZefCbft93
zHVk9gDOw8ys1XGSR2OUCyCLinXO6mqS16CmBb2MAKXq/YyH7E0N8iotAPGtfA8V
M/0ddy947rY0xCrtECfWwvGQpJS7NRv/Z9b2jCfXrI5LXmL2nfQRg0y9GE4Vjwr+
WxtGU5jOeFt0jQ+xRzcgG0Op+qK3x55l5LSo2hOcOVYbiHxcHEJFgwNi1ADeBFwb
q/HdysfURSOghqjIpMMAUabBp+DBUg2EUF7pIaUqbdqExFYcr9EYisEMiNsmKmN+
8ZbcOeerFKDQj+t/R0bFXa7UBn2UWsjI8zlR74aa2kLDXwtyz/XlO/FlYm66eBFo
2/eYUSeU+S4ej+wUAs/dvjF7f190/DUQGuwOTlLTahqWDztmhCk7qzbECu56CwKT
E5Ect2P72UleYwdblkVOVd352AmiwEzdOaziIRrPh8uenEknH6JBYPO3mjk7cCg+
GPGcNIBctjdXhg==
-----END CERTIFICATE-----
+33
View File
@@ -0,0 +1,33 @@
apiVersion: cert-manager.io/v1
kind: ClusterIssuer
metadata:
name: selfsigned
spec:
selfSigned: {}
---
# Homelab internal root CA (10y). Install the .crt on clients (see below).
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-ca-root
namespace: cert-manager
spec:
isCA: true
commonName: homelab internal root CA
duration: 87600h
renewBefore: 7200h
secretName: internal-ca-root
privateKey:
algorithm: RSA
size: 4096
issuerRef:
name: selfsigned
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: ClusterIssuer
metadata:
name: internal-ca
spec:
ca:
secretName: internal-ca-root
+4
View File
@@ -0,0 +1,4 @@
apiVersion: v1
kind: Namespace
metadata:
name: cert-manager
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: checkmk-prod-tls
namespace: checkmk
spec:
secretName: checkmk-prod-tls
dnsNames:
- cmk.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: checkmk
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+3 -1
View File
@@ -16,7 +16,7 @@ spec:
- name: checkmk-service
port: 5000
tls:
certResolver: letsencrypt
secretName: checkmk-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRouteTCP
@@ -48,3 +48,5 @@ spec:
services:
- name: checkmk-service
port: 5000
tls:
secretName: internal-wildcard-tls
+1
View File
@@ -0,0 +1 @@
secret.yaml
File renamed without changes.
+37
View File
@@ -0,0 +1,37 @@
apiVersion: apps/v1
kind: Deployment
metadata:
name: cloudflared
labels:
app: cloudflared
spec:
replicas: 1
selector:
matchLabels:
app: cloudflared
template:
metadata:
labels:
app: cloudflared
spec:
containers:
- name: cloudflared
image: cloudflare/cloudflared:2026.9.1
imagePullPolicy: IfNotPresent
args:
- tunnel
- --no-autoupdate
- run
env:
- name: TUNNEL_TOKEN
valueFrom:
secretKeyRef:
name: cloudflared-secrets
key: TUNNEL_TOKEN
resources:
requests:
memory: "32Mi"
cpu: "30m"
limits:
memory: "128Mi"
cpu: "200m"
+7
View File
@@ -0,0 +1,7 @@
apiVersion: v1
kind: Secret
metadata:
name: cloudflared-secrets
type: Opaque
stringData:
TUNNEL_TOKEN: your_tunnel_token_here
+42
View File
@@ -0,0 +1,42 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: convertx-prod-tls
namespace: converters
spec:
secretName: convertx-prod-tls
dnsNames:
- forust.xyz
- www.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: bentopdf-prod-tls
namespace: converters
spec:
secretName: bentopdf-prod-tls
dnsNames:
- pdf.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: converters
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+7 -2
View File
@@ -14,7 +14,7 @@ spec:
- name: convertx-service
port: 3000
tls:
certResolver: letsencrypt
secretName: convertx-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -31,6 +31,9 @@ spec:
services:
- name: convertx-service
port: 3000
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -47,7 +50,7 @@ spec:
- name: bentopdf-service
port: 8080
tls:
certResolver: letsencrypt
secretName: bentopdf-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -63,3 +66,5 @@ spec:
services:
- name: bentopdf-service
port: 8080
tls:
secretName: internal-wildcard-tls
+91 -46
View File
@@ -1,57 +1,102 @@
container_runtime: containerd
agent:
acquisition: []
additionalAcquisition:
- labels:
type: traefik
limit: 1000
query: |
{namespace="traefik"}
source: loki
url: http://loki.prometheus.svc.cluster.local:3100/
wait_for_ready: 30s
env:
- name: COLLECTIONS
value: "crowdsecurity/traefik crowdsecurity/base-http-scenarios"
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
- name: DISABLE_COLLECTIONS
value: "crowdsecurity/linux crowdsecurity/sshd"
acquisition:
- namespace: traefik
podName: "*traefik*"
program: traefik
poll_without_inotify: true
resources:
requests:
cpu: 50m
memory: 100Mi
limits:
cpu: 200m
memory: 500Mi
lapi:
env:
- name: COLLECTIONS
value: "crowdsecurity/traefik crowdsecurity/base-http-scenarios"
- name: DISABLE_COLLECTIONS
value: "crowdsecurity/linux crowdsecurity/sshd"
service:
type: ClusterIP
persistentVolume:
data:
enabled: true
storageClassName: local-path-retain
size: 1Gi
config:
enabled: true
storageClassName: local-path-retain
size: 100Mi
storeLAPICscliCredentialsInSecret: true
resources:
requests:
cpu: 50m
memory: 150Mi
limits:
cpu: 400m
memory: 500Mi
value: crowdsecurity/sshd
metrics:
enabled: true
serviceMonitor:
additionalLabels:
release: prometheus-stack
enabled: true
interval: 30s
scrapeTimeout: 10s
namespace: prometheus
# Static machine identity: agent pods mount pre-created LAPI credentials
# (Secret crowdsec-agent-credentials, key local_api_credentials.yaml)
# at the exact path the agent entrypoint expects. Together with the
# patched register-init (enforced by janitor-cronjob.yaml) the agent
# never calls `cscli lapi register` in steady state, so pod names,
# restarts and reboots can no longer break it.
extraVolumes:
- name: static-creds
secret:
secretName: crowdsec-agent-credentials
items:
- key: local_api_credentials.yaml
path: local_api_credentials.yaml
extraVolumeMounts:
- name: static-creds
mountPath: /tmp_config/local_api_credentials.yaml
subPath: local_api_credentials.yaml
readOnly: true
resources:
limits:
cpu: 200m
memory: 500Mi
requests:
cpu: 50m
memory: 100Mi
config:
parsers:
s02-enrich:
mobile-whitelist.yaml: |
name: forust/mobile-whitelist
description: "Whitelist SWAN/4ka mobile network"
whitelist:
reason: "Mobile IP whitelist"
cidr:
- "84.245.64.0/18"
postoverflows:
s01-whitelist:
home-dynamic-ip.yaml: |
name: forust/home-dynamic-ip
description: "Whitelist home dynamic IP"
whitelist:
reason: "Home dynamic IP"
expression:
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
lapi:
env:
- name: COLLECTIONS
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
- name: DISABLE_COLLECTIONS
value: crowdsecurity/linux crowdsecurity/sshd
metrics:
enabled: true
serviceMonitor:
additionalLabels:
release: prometheus-stack
enabled: true
persistentVolume:
config:
enabled: true
size: 100Mi
storageClassName: local-path-retain
data:
enabled: true
size: 1Gi
storageClassName: local-path-retain
resources:
limits:
cpu: 400m
memory: 500Mi
requests:
cpu: 50m
memory: 150Mi
service:
type: ClusterIP
storeLAPICscliCredentialsInSecret: true
+32
View File
@@ -0,0 +1,32 @@
apiVersion: v1
data:
crowdsec-overview.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"graph\",\n \"name\": \"Graph (old)\",\n \"version\": \"\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"stat\",\n \"name\": \"Stat\",\n \"version\": \"\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"timeseries\",\n \"name\": \"Time series\",\n \"version\": \"\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 24,\n \"panels\": [],\n \"title\": \"Summary\",\n \"type\": \"row\"\n },\n {\n \"cacheTimeout\": null,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [\n {\n \"options\": {\n \"match\": \"null\",\n \"result\": {\n \"text\": \"N/A\"\n }\n },\n \"type\": \"special\"\n }\n ],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"#E02F44\",\n \"value\": null\n },\n {\n \"color\": \"#E02F44\",\n \"value\": 10\n },\n {\n \"color\": \"#299c46\",\n \"value\": 10\n }\n ]\n },\n \"unit\": \"none\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 6,\n \"x\": 0,\n \"y\": 1\n },\n \"id\": 2,\n \"interval\": null,\n \"links\": [],\n \"maxDataPoints\": 100,\n \"options\": {\n \"colorMode\": \"background\",\n \"graphMode\": \"none\",\n \"justifyMode\": \"auto\",\n \"orientation\": \"horizontal\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"text\": {},\n \"textMode\": \"auto\"\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"count(cs_info)\",\n \"interval\": \"\",\n \"legendFormat\": \"\",\n \"refId\": \"A\"\n }\n ],\n \"timeFrom\": null,\n \"timeShift\": null,\n \"title\": \"Running Crowdsec\",\n \"transparent\": true,\n \"type\": \"stat\"\n },\n {\n \"aliasColors\": {},\n \"bars\": false,\n \"dashLength\": 10,\n \"dashes\": false,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"decimals\": 1,\n \"fieldConfig\": {\n \"defaults\": {\n \"links\": []\n },\n \"overrides\": []\n },\n \"fill\": 1,\n \"fillGradient\": 0,\n \"gridPos\": {\n \"h\": 8,\n \"w\": 18,\n \"x\": 6,\n \"y\": 1\n },\n \"hiddenSeries\": false,\n \"id\": 8,\n \"legend\": {\n \"alignAsTable\": true,\n \"avg\": false,\n \"current\": false,\n \"max\": false,\n \"min\": false,\n \"rightSide\": true,\n \"show\": true,\n \"sort\": \"total\",\n \"sortDesc\": true,\n \"total\": true,\n \"values\": true\n },\n \"lines\": true,\n \"linewidth\": 1,\n \"nullPointMode\": \"null\",\n \"options\": {\n \"alertThreshold\": true\n },\n \"percentage\": false,\n \"pluginVersion\": \"8.1.2\",\n \"pointradius\": 2,\n \"points\": false,\n \"renLine truncated
kind: ConfigMap
metadata:
labels:
app.kubernetes.io/managed-by: manual
grafana_dashboard: "1"
name: crowdsec-crowdsec-overview
namespace: prometheus
---
apiVersion: v1
data:
crowdsec-lapi-metrics.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"panel\",\n \"id\": \"bargauge\",\n \"name\": \"Bar gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"iteration\": 1655915193937,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 10,\n \"panels\": [],\n \"title\": \"Agents\",\n \"type\": \"row\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n },\n {\n \"color\": \"red\",\n \"value\": 80\n }\n ]\n }\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 12,\n \"x\": 0,\n \"y\": 1\n },\n \"id\": 2,\n \"options\": {\n \"displayMode\": \"gradient\",\n \"orientation\": \"vertical\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"showUnfilled\": false,\n \"text\": {}\n },\n \"pluginVersion\": \"8.1.2\",\n \"repeat\": \"query0\",\n \"repeatDirection\": \"h\",\n \"targets\": [\n {\n \"exemplar\": false,\n \"expr\": \"sum(rate(cs_lapi_request_duration_seconds_bucket{endpoint=\\\"/v1/watchers/login\\\", instance=\\\"$lapi\\\"}[$__rate_interval])) by (le)\",\n \"format\": \"heatmap\",\n \"interval\": \"\",\n \"legendFormat\": \"{{le}}\",\n \"refId\": \"A\"\n }\n ],\n \"title\": \"Agents Login\",\n \"type\": \"heatmap\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n }\n ]\n },\n \"unit\": \"none\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 12,\n \"x\": 12,\n \"y\": 1\n },\n \"id\": 6,\n \"options\": {\n \"displayMode\": \"gradient\",\n \"orientation\": \"auto\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"showUnfilled\": false,\n \"text\": {}\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"sum(rate(cs_lapi_request_duration_seconds_bucket{endpoint=\\\"/v1/watchers/login\\\"}[$__rate_interval])) by (le)\",\n \"format\": \"heatmap\",\n \"interval\": \"\",\n \"legendFormat\": \"{{le}}\",\n \"refId\": \"A\"\n }\n ],\n \"title\": \"Heartbeat\",\n \"type\": \"heatmap\"\n },\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 9\n },\n \"id\": 12,\n \"panels\": [],\n \"title\": \"Decisions\",\n \"type\": \"row\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n Line truncated
kind: ConfigMap
metadata:
labels:
app.kubernetes.io/managed-by: manual
grafana_dashboard: "1"
name: crowdsec-crowdsec-lapi-metrics
namespace: prometheus
---
apiVersion: v1
data:
crowdsec-insight.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"panel\",\n \"id\": \"bargauge\",\n \"name\": \"Bar gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"gauge\",\n \"name\": \"Gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"stat\",\n \"name\": \"Stat\",\n \"version\": \"\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"iteration\": 1655915159751,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": true,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 22,\n \"panels\": [\n {\n \"cacheTimeout\": null,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [\n {\n \"options\": {\n \"match\": \"null\",\n \"result\": {\n \"text\": \"N/A\"\n }\n },\n \"type\": \"special\"\n }\n ],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n },\n {\n \"color\": \"red\",\n \"value\": 80\n }\n ]\n },\n \"unit\": \"dateTimeAsIso\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 9,\n \"w\": 5,\n \"x\": 2,\n \"y\": 1\n },\n \"id\": 2,\n \"interval\": null,\n \"links\": [],\n \"maxDataPoints\": 100,\n \"options\": {\n \"colorMode\": \"none\",\n \"graphMode\": \"none\",\n \"justifyMode\": \"auto\",\n \"orientation\": \"horizontal\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"text\": {},\n \"textMode\": \"auto\"\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"(process_start_time_seconds{instance=\\\"$instance\\\"})*1000\",\n \"interval\": \"\",\n \"legendFormat\": \"{{instance}}\",\n \"refId\": \"A\"\n }\n ],\n \"timeFrom\": null,\n \"timeShift\": null,\n \"title\": \"Up since\",\n \"type\": \"stat\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"displayName\": \"\",\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n }\n ]\n },\n \"unit\": \"decbytes\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 9,\n \"w\": 5,\n \"x\": 7,\n \"y\": 1\n },\n \"id\": 4,\n \"options\": {\n \"orientation\": \"auto\",\n \"reduceOptions\": {\n \"calcs\": [\n \"mean\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\Line truncated
kind: ConfigMap
metadata:
labels:
app.kubernetes.io/managed-by: manual
grafana_dashboard: "1"
name: crowdsec-crowdsec-insight
namespace: prometheus
+195
View File
@@ -0,0 +1,195 @@
# CrowdSec self-healing: static machine identity + enforcement loops.
#
# Problem it fixes: the chart's agent init container runs
# `cscli lapi register --machine "$POD_NAME" ...`
# unconditionally. Credentials live in an emptyDir, the machine row lives
# in LAPI's persistent DB. Any init re-run for an already-known pod name
# (kubelet restart, node reboot) dies with
# 403 Forbidden: user '<pod>' already exist
# and the DaemonSet pod sticks in Init forever. Every DS restart also
# leaves an orphan machine row that is never cleaned.
#
# Design (name-independent):
# * Agent identity is a STATIC machine `crowdsec-agent-workstation`
# whose password lives in Secret `crowdsec-agent-credentials`
# (created once, manually - like all other secrets in this repo).
# The secret is mounted into agent pods at
# /tmp_config/local_api_credentials.yaml (see extraVolumeMounts in
# crowdsec-values.yaml), which is exactly the path the agent's main
# container copies into place at startup.
# * The DS init command is patched (strategic merge, by container name)
# to SKIP registration when that file exists, keeping the legacy
# register path only as fallback. Detection marker in the patched
# command: `[ -s /tmp_config`.
# * This CronJob enforces the desired state hourly, so recovery is
# automatic even after `helm upgrade` reverts the DS patch or the
# LAPI database is wiped:
# 1. patch DS init if it still has the unconditional register
# (no-op otherwise - no restart churn);
# 2. prune machines with no heartbeat for 2h (orphan hygiene);
# 3. ensure the static machine exists, recreating it with the
# Secret password if missing (agent retry loops reconnect
# on their own - same name + same password);
# 4. prune bouncer entries idle for 30d.
#
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
# Force a run:
# kubectl create job -n crowdsec --from=cronjob/crowdsec-janitor janitor-now
#
# Helm upgrades: the janitor's strategic patch puts the DS field under
# the `kubectl-patch` field manager, so a plain `helm upgrade` FAILS
# with an SSA conflict on initContainers[].command. Procedure:
# 1. revert init to chart state (kills the conflict):
# helm template crowdsec crowdsec/crowdsec --version <ver> \
# -n crowdsec -f crowdsec/k8s/crowdsec-values.yaml > /tmp/r.yaml
# python3 -c "import yaml,json; ..." # build revert patch from
# the rendered DaemonSet init command, then
# kubectl patch ds crowdsec-agent -n crowdsec \
# --type strategic -p "\$(cat /tmp/revert_patch.json)"
# 2. helm upgrade --install crowdsec ... (no --force needed)
# 3. janitor-now right away (upgrade reverts init; new pods would
# sit in Init until the next hourly run otherwise).
#
# One-time bootstrap (order matters):
# 1. Create Secret + static machine (see commands in chat).
# 2. Apply this file, trigger janitor-now, wait for agent 1/1.
# 3. One-time orphan cleanup:
# kubectl exec -n crowdsec deploy/crowdsec-lapi -- \
# cscli machines prune --duration 1h --force
# 4. Only then `helm upgrade` crowdsec with the extraVolumes values.
# Upgrade reverts the DS patch; trigger janitor-now right after it
# (otherwise new pods sit in Init until the next hourly run, then
# self-heal anyway).
#
# Password rotation: update the Secret, delete the machine
# (`cscli machines delete crowdsec-agent-workstation`), trigger
# janitor-now (recreates it), then `kubectl rollout restart
# ds/crowdsec-agent -n crowdsec` (agent reads the file at startup only).
apiVersion: v1
kind: ServiceAccount
metadata:
name: crowdsec-janitor
namespace: crowdsec
labels:
app.kubernetes.io/part-of: crowdsec
---
apiVersion: rbac.authorization.k8s.io/v1
kind: Role
metadata:
name: crowdsec-janitor
namespace: crowdsec
labels:
app.kubernetes.io/part-of: crowdsec
rules:
- apiGroups: [""]
resources: ["pods"]
verbs: ["get", "list"]
- apiGroups: [""]
resources: ["pods/exec"]
verbs: ["create"]
- apiGroups: ["apps"]
resources: ["daemonsets"]
verbs: ["get", "patch"]
# `kubectl exec deploy/<name>` resolves deploy -> replicaset -> pod,
# which needs read access to these (exec itself is pods/exec above).
- apiGroups: ["apps"]
resources: ["deployments", "replicasets"]
verbs: ["get", "list"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: RoleBinding
metadata:
name: crowdsec-janitor
namespace: crowdsec
labels:
app.kubernetes.io/part-of: crowdsec
subjects:
- kind: ServiceAccount
name: crowdsec-janitor
namespace: crowdsec
roleRef:
kind: Role
name: crowdsec-janitor
apiGroup: rbac.authorization.k8s.io
---
apiVersion: batch/v1
kind: CronJob
metadata:
name: crowdsec-janitor
namespace: crowdsec
labels:
app.kubernetes.io/part-of: crowdsec
spec:
schedule: "17 * * * *"
concurrencyPolicy: Forbid
successfulJobsHistoryLimit: 3
failedJobsHistoryLimit: 3
jobTemplate:
spec:
activeDeadlineSeconds: 300
template:
metadata:
labels:
app.kubernetes.io/part-of: crowdsec
spec:
serviceAccountName: crowdsec-janitor
restartPolicy: OnFailure
containers:
- name: janitor
# Same image the chart itself uses for registration jobs;
# IfNotPresent so it works while the node is offline
# (layer cached from the chart install).
image: alpine/kubectl:latest
imagePullPolicy: IfNotPresent
env:
- name: AGENT_PASSWORD
valueFrom:
secretKeyRef:
name: crowdsec-agent-credentials
key: password
command:
- /bin/sh
- -c
- |
set -eu
LAPI_EXEC="kubectl exec -n crowdsec deploy/crowdsec-lapi --"
echo "== 1. enforce patched agent init =="
CUR=$(kubectl get ds crowdsec-agent -n crowdsec \
-o jsonpath='{.spec.template.spec.initContainers[0].command[2]}')
case "$CUR" in
*'-s /tmp_config'*)
echo "init already patched"
;;
*)
echo "patching init"
WAIT='until nc "$LAPI_HOST" "$LAPI_PORT" -z'
WAIT="$WAIT; do echo waiting for lapi to start; sleep 5; done"
LINK='ln -s /staging/etc/crowdsec /etc/crowdsec'
REG='cscli lapi register --machine "$USERNAME"'
REG="$REG -u \"\$LAPI_URL\" --token \"\$REGISTRATION_TOKEN\""
CREDS=/tmp_config/local_api_credentials.yaml
CMD="$WAIT; $LINK; [ -s $CREDS ] || {"
CMD="$CMD $REG && cp"
CMD="$CMD /etc/crowdsec/local_api_credentials.yaml $CREDS; }"
ESC=$(printf '%s' "$CMD" | sed 's/"/\\"/g')
PATCH='{"spec":{"template":{"spec":{"initContainers":'
PATCH=$PATCH'[{"name":"wait-for-lapi-and-register",'
PATCH=$PATCH'"command":["sh","-c","'$ESC'"]}]}}}}'
kubectl patch ds crowdsec-agent -n crowdsec \
--type strategic -p "$PATCH"
;;
esac
echo "== 2. prune orphan machines (no heartbeat for 2h) =="
$LAPI_EXEC cscli machines prune --duration 2h --force
echo "== 3. ensure static machine exists =="
if $LAPI_EXEC cscli machines inspect \
crowdsec-agent-workstation >/dev/null 2>&1; then
echo "static machine present"
else
echo "recreating static machine"
$LAPI_EXEC cscli machines add crowdsec-agent-workstation \
--password "$AGENT_PASSWORD" --force
fi
echo "== 4. prune stale bouncers (no pull for 30d) =="
$LAPI_EXEC cscli bouncers prune -d 720h --force
+7
View File
@@ -25,3 +25,10 @@ spec:
ports:
- protocol: TCP
port: 8080
- from:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: prometheus
ports:
- protocol: TCP
port: 6060
+1 -1
View File
@@ -1,6 +1,6 @@
services:
dockmon:
image: darthnorse/dockmon:2.4.5
image: darthnorse/dockmon:2.5.0
container_name: dockmon
restart: unless-stopped
# ports:
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: dockmon-prod-tls
namespace: dockmon
spec:
secretName: dockmon-prod-tls
dnsNames:
- dockmon.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: dockmon
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+1 -1
View File
@@ -29,7 +29,7 @@ spec:
spec:
containers:
- name: dockmon
image: darthnorse/dockmon:2.4.5
image: darthnorse/dockmon:2.5.0
ports:
- containerPort: 443
volumeMounts:
+3 -1
View File
@@ -26,7 +26,7 @@ spec:
port: 443
serversTransport: dockmon-transport
tls:
certResolver: letsencrypt
secretName: dockmon-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -43,3 +43,5 @@ spec:
- name: dockmon-service
port: 443
serversTransport: dockmon-transport
tls:
secretName: internal-wildcard-tls
+1 -1
View File
@@ -1,7 +1,7 @@
services:
downtify:
container_name: downtify
image: ghcr.io/henriquesebastiao/downtify:2.12.0
image: ghcr.io/henriquesebastiao/downtify:3.1.0
restart: unless-stopped
# ports:
# - '7077:8000'
+1 -1
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: downtify
image: ghcr.io/henriquesebastiao/downtify:2.12.0
image: ghcr.io/henriquesebastiao/downtify:3.1.0
ports:
- containerPort: 8000
volumeMounts:
+3 -1
View File
@@ -17,7 +17,7 @@ spec:
- name: downtify-service
port: 8000
tls:
certResolver: letsencrypt
secretName: downtify-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -33,3 +33,5 @@ spec:
services:
- name: downtify-service
port: 8000
tls:
secretName: internal-wildcard-tls
+1
View File
@@ -0,0 +1 @@
1.56.0
+1 -1
View File
@@ -11,7 +11,7 @@ services:
retries: 5
playwright-service:
image: mcr.microsoft.com/playwright:v1.63.0-jammy
image: mcr.microsoft.com/playwright:v1.56.0-jammy
restart: unless-stopped
command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws
+77
View File
@@ -0,0 +1,77 @@
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: edu-master-webinar
namespace: edu-master
labels:
release: prometheus-stack
spec:
groups:
- name: edu_master.webinar
rules:
# No successful webinar check for 5m (~2-3 missed 2-min checks).
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
- alert: WebinarCheckerNoSuccessfulCheck
expr: |
(time() - webinar_check_last_success_timestamp_seconds > 300)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 2m
labels:
severity: critical
annotations:
summary: "Webinar checker has no successful check for 5m"
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
- alert: WebinarCheckerConsecutiveFailures
expr: |
webinar_check_consecutive_failures >= 3
for: 5m
labels:
severity: critical
annotations:
summary: "Webinar checker failing consecutively"
description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
- alert: WebinarCheckerScrapeDown
expr: |
absent(webinar_check_last_run_timestamp_seconds) == 1
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker metrics missing"
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
- alert: EduPhpsessidMissing
expr: |
edu_phpsessid_present == 0
for: 10m
labels:
severity: critical
annotations:
summary: "EDU_PHPSESSID missing"
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
# Hard deps: checker and playwright deployments unavailable.
- alert: WebinarCheckerDeploymentDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker deployment unavailable"
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
- alert: PlaywrightServiceDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Playwright service unavailable"
description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it."
+2 -1
View File
@@ -17,7 +17,8 @@ spec:
spec:
containers:
- name: playwright
image: mcr.microsoft.com/playwright:v1.63.0-jammy
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
image: mcr.microsoft.com/playwright:v1.56.0-jammy
imagePullPolicy: IfNotPresent
command:
- npx
+2
View File
@@ -21,6 +21,8 @@ stringData:
WEBINAR_TELEGRAM_TOKEN: ""
WEBINAR_ADMIN_ID: ""
WEBINAR_CHECK_INTERVAL: "60"
# Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml)
METRICS_PORT: "8000"
# Database
REDIS_HOST: "redis"
REDIS_PORT: "6379"
+15
View File
@@ -0,0 +1,15 @@
apiVersion: v1
kind: Service
metadata:
name: webinar-checker
namespace: edu-master
labels:
app: edu-master-webinar-checker
spec:
selector:
app: edu-master-webinar-checker
ports:
- name: metrics
port: 8000
targetPort: metrics
protocol: TCP
+16
View File
@@ -0,0 +1,16 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: webinar-checker
namespace: edu-master
labels:
release: prometheus-stack
spec:
selector:
matchLabels:
app: edu-master-webinar-checker
endpoints:
- port: metrics
path: /metrics
interval: 30s
scrapeTimeout: 10s
+4
View File
@@ -47,6 +47,10 @@ spec:
- name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:latest
imagePullPolicy: Always
ports:
- name: metrics
containerPort: 8000
protocol: TCP
envFrom:
- secretRef:
name: edu-master-secrets
+5 -2
View File
@@ -2,8 +2,11 @@ FROM python:3.11-slim
WORKDIR /app
# Install dependencies
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==1.56.0 redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
# renovate: datasource=pypi depName=playwright versioning=pep440
ARG PLAYWRIGHT_VERSION=1.56.0
# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
COPY checker.py .
+148 -15
View File
@@ -1,12 +1,15 @@
import asyncio
import contextlib
import json
import logging
import os
import re
import tempfile
import threading
import time
from datetime import datetime, timedelta
from html import escape
from http.server import BaseHTTPRequestHandler, HTTPServer
import redis
from playwright.async_api import async_playwright
@@ -48,6 +51,106 @@ USER_AGENT = _env(
)
WEBINAR_TELEGRAM_TOKEN = _env('WEBINAR_TELEGRAM_TOKEN')
ADMIN_ID = int(_env('WEBINAR_ADMIN_ID', '0'))
METRICS_PORT = int(_env('METRICS_PORT', '8000'))
# --- Prometheus metrics (stdlib only, no extra deps) ---
# Scraped by prometheus-stack via ServiceMonitor (edu_master/k8s/servicemonitor.yaml).
# Critical alerts in edu_master/k8s/alerts.yaml fire to Telegram via Alertmanager.
_METRICS_LOCK = threading.Lock()
_METRICS = {
'last_run': 0.0, # Unix ts of last check start
'last_success': 0.0, # Unix ts of last successful check
'last_duration': 0.0, # Duration of last check in seconds
'success_total': 0,
'failure_total': 0,
'consecutive_failures': 0,
'phpsessid_present': 1, # 1 if EDU_PHPSESSID found in redis, else 0
}
def _metric_check_start():
with _METRICS_LOCK:
_METRICS['last_run'] = time.time()
def _metric_check_ok(duration: float):
now = time.time()
with _METRICS_LOCK:
_METRICS['last_success'] = now
_METRICS['last_duration'] = duration
_METRICS['success_total'] += 1
_METRICS['consecutive_failures'] = 0
_METRICS['phpsessid_present'] = 1
def _metric_check_fail(duration: float, phpsessid_missing: bool = False):
with _METRICS_LOCK:
_METRICS['last_duration'] = duration
_METRICS['failure_total'] += 1
_METRICS['consecutive_failures'] += 1
_METRICS['phpsessid_present'] = 0 if phpsessid_missing else 1
def _metrics_render() -> bytes:
with _METRICS_LOCK:
m = dict(_METRICS)
lines = [
'# HELP webinar_check_last_run_timestamp_seconds Unix timestamp of last webinar check start.',
'# TYPE webinar_check_last_run_timestamp_seconds gauge',
f'webinar_check_last_run_timestamp_seconds {m["last_run"]}',
'# HELP webinar_check_last_success_timestamp_seconds Unix timestamp of last successful webinar check.',
'# TYPE webinar_check_last_success_timestamp_seconds gauge',
f'webinar_check_last_success_timestamp_seconds {m["last_success"]}',
'# HELP webinar_check_last_duration_seconds Duration of last webinar check in seconds.',
'# TYPE webinar_check_last_duration_seconds gauge',
f'webinar_check_last_duration_seconds {m["last_duration"]}',
'# HELP webinar_check_success_total Total successful webinar checks.',
'# TYPE webinar_check_success_total counter',
f'webinar_check_success_total {m["success_total"]}',
'# HELP webinar_check_failure_total Total failed webinar checks (timeout, playwright error, page error).',
'# TYPE webinar_check_failure_total counter',
f'webinar_check_failure_total {m["failure_total"]}',
'# HELP webinar_check_consecutive_failures Consecutive failed webinar checks (reset on success).',
'# TYPE webinar_check_consecutive_failures gauge',
f'webinar_check_consecutive_failures {m["consecutive_failures"]}',
'# HELP edu_phpsessid_present 1 if EDU_PHPSESSID exists in redis, 0 otherwise.',
'# TYPE edu_phpsessid_present gauge',
f'edu_phpsessid_present {m["phpsessid_present"]}',
]
return ('\n'.join(lines) + '\n').encode()
class _MetricsHandler(BaseHTTPRequestHandler):
def do_GET(self):
if self.path == '/metrics':
body = _metrics_render()
self.send_response(200)
self.send_header('Content-Type', 'text/plain; version=0.0.4')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
elif self.path in ('/healthz', '/health'):
body = b'ok\n'
self.send_response(200)
self.send_header('Content-Type', 'text/plain')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
else:
self.send_response(404)
self.end_headers()
def log_message(self, *args):
pass # keep bot logs clean
def start_metrics_server(port: int = METRICS_PORT):
server = HTTPServer(('0.0.0.0', port), _MetricsHandler) # noqa: S104 - k8s ServiceMonitor scrapes pod IP
thread = threading.Thread(target=server.serve_forever, name='metrics-server', daemon=True)
thread.start()
logger.info(f'Metrics server listening on :{port}/metrics')
return server
# Redis Keys
KEY_WHITELIST = 'bot:whitelist'
@@ -597,8 +700,9 @@ async def _collect_event_times(page) -> dict:
async def fetch_diary_data(phpsessid: str) -> dict | None:
logger.info('Fetching diary data via Playwright...')
try:
async with asyncio.timeout(60):
async with async_playwright() as p:
browser = await p.chromium.connect(PLAYWRIGHT_WS)
browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
try:
context_browser = await browser.new_context(user_agent=USER_AGENT)
await context_browser.add_cookies(
@@ -607,7 +711,7 @@ async def fetch_diary_data(phpsessid: str) -> dict | None:
page = await context_browser.new_page()
try:
await page.goto(DIARY_URL, wait_until='domcontentloaded')
await asyncio.wait_for(page.goto(DIARY_URL, wait_until='domcontentloaded'), timeout=30)
await page.wait_for_selector('table.calendar', timeout=10000)
await page.wait_for_timeout(1500)
@@ -646,10 +750,16 @@ async def fetch_diary_data(phpsessid: str) -> dict | None:
logger.error(f'Error parsing diary: {e}')
return None
finally:
await page.close()
await context_browser.close()
with contextlib.suppress(Exception):
await asyncio.wait_for(page.close(), timeout=5)
with contextlib.suppress(Exception):
await asyncio.wait_for(context_browser.close(), timeout=5)
finally:
await browser.close()
with contextlib.suppress(Exception):
await asyncio.wait_for(browser.close(), timeout=5)
except TimeoutError:
logger.error('Diary fetch timed out (60s)')
return None
except Exception as e:
logger.error(f'Playwright error in diary fetch: {e}')
return None
@@ -931,8 +1041,9 @@ def _parse_schedule_html(table_html: str) -> dict:
async def fetch_schedule_data(phpsessid: str) -> dict | None:
logger.info('Fetching schedule data via Playwright...')
try:
async with asyncio.timeout(60):
async with async_playwright() as p:
browser = await p.chromium.connect(PLAYWRIGHT_WS)
browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
try:
context_browser = await browser.new_context(user_agent=USER_AGENT)
await context_browser.add_cookies(
@@ -941,7 +1052,7 @@ async def fetch_schedule_data(phpsessid: str) -> dict | None:
page = await context_browser.new_page()
try:
await page.goto(SCHEDULE_URL, wait_until='domcontentloaded')
await asyncio.wait_for(page.goto(SCHEDULE_URL, wait_until='domcontentloaded'), timeout=30)
await page.wait_for_selector('table.schedule-table', timeout=10000)
await page.wait_for_timeout(1500)
@@ -969,10 +1080,16 @@ async def fetch_schedule_data(phpsessid: str) -> dict | None:
logger.error(f'Error parsing schedule: {e}')
return None
finally:
await page.close()
await context_browser.close()
with contextlib.suppress(Exception):
await asyncio.wait_for(page.close(), timeout=5)
with contextlib.suppress(Exception):
await asyncio.wait_for(context_browser.close(), timeout=5)
finally:
await browser.close()
with contextlib.suppress(Exception):
await asyncio.wait_for(browser.close(), timeout=5)
except TimeoutError:
logger.error('Schedule fetch timed out (60s)')
return None
except Exception as e:
logger.error(f'Playwright error in schedule fetch: {e}')
return None
@@ -1485,10 +1602,13 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
int: Number of webinars found, or None if check failed
"""
logger.info('Running webinar check...')
_t0 = time.time()
_metric_check_start()
phpsessid = redis_client.get(KEY_PHPSESSID)
if not phpsessid:
logger.warning('PHPSESSID missing. Skipping check.')
_metric_check_fail(time.time() - _t0, phpsessid_missing=True)
# --- DEBUG LOGGING ---
try:
with open('phpsessid_missing.log', 'a') as f:
@@ -1502,9 +1622,10 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
content = ''
try:
async with asyncio.timeout(90):
async with async_playwright() as p:
# Connect to remote Playwright service
browser = await p.chromium.connect(PLAYWRIGHT_WS)
browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
try:
# Create browser context with user agent
@@ -1520,7 +1641,7 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
try:
# Navigate to webinar page
await page.goto(WEBINAR_URL, wait_until='domcontentloaded')
await asyncio.wait_for(page.goto(WEBINAR_URL, wait_until='domcontentloaded'), timeout=30)
# Wait for the table to load
await page.wait_for_selector('#meetings table', timeout=10000)
@@ -1564,16 +1685,25 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
if page and not content:
content = await page.content()
_metric_check_fail(time.time() - _t0)
return None
finally:
await page.close()
await context_browser.close()
with contextlib.suppress(Exception):
await asyncio.wait_for(page.close(), timeout=5)
with contextlib.suppress(Exception):
await asyncio.wait_for(context_browser.close(), timeout=5)
finally:
await browser.close()
with contextlib.suppress(Exception):
await asyncio.wait_for(browser.close(), timeout=5)
except TimeoutError:
logger.error('Webinar check timed out after 90s (playwright hang)')
_metric_check_fail(time.time() - _t0)
return None
except Exception as e:
logger.error(f'Playwright error: {e}')
_metric_check_fail(time.time() - _t0)
return None
# --- DEBUG LOGGING (Saving last response content) ---
@@ -1637,6 +1767,7 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
else:
logger.info(f'Found {len(current_webinars)} webinar(s), but all are already known')
_metric_check_ok(time.time() - _t0)
return len(current_webinars)
@@ -1680,6 +1811,8 @@ def main():
job_queue = app.job_queue
job_queue.run_repeating(check_webinars_job, interval=WEBINAR_CHECK_INTERVAL, first=10)
start_metrics_server()
logger.info('Bot started polling...')
app.run_polling()
File renamed without changes.
+29
View File
@@ -0,0 +1,29 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: gitea-prod-tls
namespace: gitea
spec:
secretName: gitea-prod-tls
dnsNames:
- gcr.forust.xyz
- gitea.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: gitea
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+1 -1
View File
@@ -31,7 +31,7 @@ spec:
spec:
containers:
- name: gitea
image: docker.gitea.com/gitea:1.27.3
image: gitea/gitea:1.27.3
envFrom:
- configMapRef:
name: gitea-config
+4 -1
View File
@@ -24,7 +24,7 @@ spec:
- name: gitea-service
port: 3000
tls:
certResolver: letsencrypt
secretName: gitea-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -45,6 +45,9 @@ spec:
services:
- name: gitea-service
port: 3000
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRouteTCP
+29
View File
@@ -0,0 +1,29 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: glance-prod-tls
namespace: glance
spec:
secretName: glance-prod-tls
dnsNames:
- forust.xyz
- www.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: glance
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+4 -1
View File
@@ -16,7 +16,7 @@ spec:
- name: glance-service
port: 8080
tls:
certResolver: letsencrypt
secretName: glance-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -35,6 +35,9 @@ spec:
services:
- name: glance-service
port: 8080
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: Middleware
View File
File renamed without changes.
View File
Whitespace-only changes.
+41
View File
@@ -0,0 +1,41 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: headscale-prod-tls
namespace: headscale
spec:
secretName: headscale-prod-tls
dnsNames:
- hs.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: headplane-prod-tls
namespace: headscale
spec:
secretName: headplane-prod-tls
dnsNames:
- hp.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: headscale
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+7 -2
View File
@@ -38,7 +38,7 @@ spec:
- name: headscale-server-external
port: 9090
tls:
certResolver: letsencrypt
secretName: headscale-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -64,7 +64,7 @@ spec:
- name: headplane-external
port: 3000
tls:
certResolver: letsencrypt
secretName: headplane-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -90,6 +90,9 @@ spec:
services:
- name: headscale-server-external
port: 9090
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -110,3 +113,5 @@ spec:
services:
- name: headplane-external
port: 3000
tls:
secretName: internal-wildcard-tls
+42
View File
@@ -0,0 +1,42 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: forust-homepage-prod-tls
namespace: homepages
spec:
secretName: forust-homepage-prod-tls
dnsNames:
- forust.xyz
- www.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: xdfnx-homepage-prod-tls
namespace: homepages
spec:
secretName: xdfnx-homepage-prod-tls
dnsNames:
- xdfnx.cfd
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: homepages
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+7 -2
View File
@@ -17,7 +17,7 @@ spec:
- name: forust-homepage-service
port: 80
tls:
certResolver: letsencrypt
secretName: forust-homepage-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -34,6 +34,9 @@ spec:
services:
- name: forust-homepage-service
port: 80
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -53,7 +56,7 @@ spec:
- name: xdfnx-homepage-service
port: 80
tls:
certResolver: letsencrypt
secretName: xdfnx-homepage-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -69,3 +72,5 @@ spec:
services:
- name: xdfnx-homepage-service
port: 80
tls:
secretName: internal-wildcard-tls
+3 -1
View File
@@ -16,7 +16,7 @@ spec:
- name: kener-service
port: 3000
tls:
certResolver: letsencrypt
secretName: kener-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -32,3 +32,5 @@ spec:
services:
- name: kener-service
port: 3000
tls:
secretName: internal-wildcard-tls
View File
Whitespace-only changes.
+57
View File
@@ -0,0 +1,57 @@
# Pinned chart: grafana/alloy 1.12.1 (app v1.19.2).
# Install (deferred to deploy task, namespace prometheus):
# helm upgrade --install alloy grafana/alloy --version 1.12.1 \
# --namespace prometheus --values loki/k8s/alloy-values.yaml --wait
# DaemonSet ships k8s pod logs (API-tailed, no hostPath mounts) to Loki.
# Scope phase 1: k8s only, compose leftovers out.
controller:
type: daemonset
resources:
requests:
memory: "128Mi"
cpu: "50m"
limits:
memory: "512Mi"
cpu: "500m"
image:
tag: "v1.19.2"
alloy:
configMap:
create: true
content: |
discovery.kubernetes "pods" {
role = "pod"
}
discovery.relabel "pods" {
targets = discovery.kubernetes.pods.targets
rule {
source_labels = ["__meta_kubernetes_namespace"]
target_label = "namespace"
}
rule {
source_labels = ["__meta_kubernetes_pod_name"]
target_label = "pod"
}
rule {
source_labels = ["__meta_kubernetes_pod_container_name"]
target_label = "container"
}
}
loki.source.kubernetes "pods" {
targets = discovery.relabel.pods.output
forward_to = [loki.write.default.receiver]
}
loki.write "default" {
endpoint {
url = "http://loki-gateway.prometheus.svc.cluster.local/loki/api/v1/push"
}
}
+97
View File
@@ -0,0 +1,97 @@
# Pinned chart: grafana/loki 7.3.0 (app 3.6.12).
# Install (deferred to deploy task, namespace prometheus):
# helm upgrade --install loki grafana/loki --version 7.3.0 \
# --namespace prometheus --values loki/k8s/loki-values.yaml --wait
# SingleBinary, filesystem storage on local-path-retain, 14d retention.
# No IngressRoute: Loki is cluster-internal, queried via Grafana datasource.
deploymentMode: SingleBinary
loki:
# Multitenancy off: single-node homelab, gateway + Alloy + Grafana talk to one tenant.
auth_enabled: false
image:
tag: "3.6.12"
commonConfig:
# Single replica: default RF=3 would require 3 ingesters and fail all writes.
replication_factor: 1
storage:
type: filesystem
schemaConfig:
configs:
- from: "2024-04-01"
store: tsdb
object_store: filesystem
schema: v13
index:
prefix: index_
period: 24h
compactor:
retention_enabled: true
delete_request_store: filesystem
limits_config:
retention_period: 336h
rulerConfig:
wal:
dir: /var/loki/ruler-wal
storage:
type: local
local:
directory: /var/loki/rules
singleBinary:
replicas: 1
persistence:
enabled: true
size: 20Gi
storageClass: local-path-retain
resources:
requests:
memory: "512Mi"
cpu: "200m"
limits:
memory: "2Gi"
cpu: "1000m"
# Zeroed: unused in SingleBinary mode (chart validation requires it).
write:
replicas: 0
read:
replicas: 0
backend:
replicas: 0
gateway:
replicas: 1
resources:
requests:
memory: "64Mi"
cpu: "50m"
limits:
memory: "256Mi"
cpu: "300m"
monitoring:
serviceMonitor:
enabled: true
labels:
release: prometheus-stack
interval: 15s
rules:
enabled: true
namespace: prometheus
labels:
release: prometheus-stack
# Disabled: memcached caches don't fit a memory-tight single node.
# SingleBinary works without them (slower repeated queries, fine at homelab scale).
resultsCache:
enabled: false
chunksCache:
enabled: false
# Disabled: synthetic canary traffic + helm test pod, noise on a single node.
lokiCanary:
enabled: false
test:
enabled: false
+1 -1
View File
@@ -1,6 +1,6 @@
services:
metube:
image: ghcr.io/alexta69/metube:2026.08.28
image: ghcr.io/alexta69/metube:2026.09.15
container_name: metube
restart: unless-stopped
# ports:
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: metube-prod-tls
namespace: metube
spec:
secretName: metube-prod-tls
dnsNames:
- metube.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: metube
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+3 -1
View File
@@ -15,7 +15,7 @@ spec:
- name: metube-service
port: 8081
tls:
certResolver: letsencrypt
secretName: metube-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -31,3 +31,5 @@ spec:
services:
- name: metube-service
port: 8081
tls:
secretName: internal-wildcard-tls
+1 -1
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: metube
image: ghcr.io/alexta69/metube:2026.08.28
image: ghcr.io/alexta69/metube:2026.09.15
envFrom:
- configMapRef:
name: metube-config
+1 -1
View File
@@ -1,6 +1,6 @@
services:
n8n:
image: docker.n8n.io/n8nio/n8n:2.39.5
image: docker.n8n.io/n8nio/n8n:2.41.0
container_name: n8n
restart: unless-stopped
environment:
+2
View File
@@ -32,3 +32,5 @@ spec:
services:
- name: n8n-service
port: 5678
tls:
secretName: internal-wildcard-tls
+15
View File
@@ -0,0 +1,15 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: n8n
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+1 -1
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: n8n
image: docker.n8n.io/n8nio/n8n:2.39.5
image: docker.n8n.io/n8nio/n8n:2.41.0
envFrom:
- configMapRef:
name: n8n-config
+13
View File
@@ -0,0 +1,13 @@
# Public hostname advertised to NetBird clients and used for TLS/OAuth.
NETBIRD_DOMAIN=nb.forust.xyz
# Internal-only aliases routed by the existing Traefik instance.
NETBIRD_LOCAL_DOMAIN=netbird.workstation.internal
NETBIRD_DEV_DOMAIN=netbird.gigaforust.internal
NETBIRD_PROXY_SUBNET=auto
NETBIRD_CLIENT_HOSTNAME=hostname
# Add a dashboard-generated setup key before starting client.compose.yaml.
# NB_SETUP_KEY=
+123
View File
@@ -0,0 +1,123 @@
# NetBird
Self-hosted NetBird with the combined management, signal, relay, and STUN server. The dashboard and server run behind the repository's existing external Traefik instance on the Docker `proxy` network. Only STUN UDP `3478` is published directly.
The deployment uses SQLite for a single-instance homelab server. The persistent `netbird_data` volume and the datastore encryption key are both required to recover the installation.
## Files
- `compose.yaml`: dashboard and combined server; selected by the marker-driven deploy workflow through `active`.
- `config.template.yaml`: non-secret server configuration rendered at startup.
- `entrypoint.sh`: injects Docker secrets into an in-memory runtime configuration.
- `client.compose.yaml`: optional host-network peer using a dashboard-generated setup key.
- `.env`: ignored local hostnames, the detected Traefik Docker-network subnet, and optional client setup key.
- `secrets/`: ignored relay secret and datastore encryption key.
## First deployment
Run these commands on the Docker host before merging the activating branch. The deploy preflight resets tracked files but preserves ignored local state.
```bash
cd /srv/homelab/netbird
./setup.sh
$EDITOR .env
docker compose config --quiet
docker compose up -d
```
Review the values in `.env` before starting. The example public hostname is `netbird.forust.xyz`; change it if a different public domain was selected. `setup.sh` replaces `NETBIRD_PROXY_SUBNET=auto` with the first IPv4 subnet of the external Docker `proxy` network. Keep that value synchronized with the network; set an explicit CIDR instead if the network is managed elsewhere.
`setup.sh` is idempotent and never replaces existing secrets. Do not delete or regenerate `secrets/datastore-encryption-key` after the first successful start unless all encrypted setup keys and API tokens are intentionally being invalidated.
## Network prerequisites
- Point the public hostname directly to the Docker host. Do not proxy UDP `3478` through Cloudflare or another CDN.
- Allow inbound TCP `80`, TCP `443`, and UDP `3478` through the host firewall and upstream router.
- Ensure the external `proxy` Docker network exists and Traefik uses its `websecure` entrypoint and `letsencrypt` resolver. `NETBIRD_PROXY_SUBNET` must describe that network; it is used to trust only forwarded client addresses from Traefik.
- Ensure the internal names in `.env` resolve where the local and development aliases are needed.
- Keep Traefik's `websecure` read timeout disabled for long-lived gRPC and WebSocket sessions. This repository configures `--entrypoints.websecure.transport.respondingTimeouts.readTimeout=0` in `traefik/compose.yaml`.
After startup, verify OIDC discovery through the public TLS endpoint:
```bash
curl -fsS "https://${NETBIRD_DOMAIN}/oauth2/.well-known/openid-configuration"
```
Open `https://${NETBIRD_DOMAIN}` immediately and complete the initial owner setup. Treat the initial setup flow as public until the owner exists.
## Optional host client
The client intentionally lives in a separate Compose project. Normal server deploys use `--remove-orphans`, so keeping the client in the server project would cause it to be removed.
1. Create a reusable or ephemeral setup key in the NetBird dashboard.
2. Put `NB_SETUP_KEY=<key>` in the ignored `netbird/.env` file.
3. Set `NETBIRD_CLIENT_HOSTNAME` to this machine's desired peer name.
4. Start and inspect the client:
```bash
cd /srv/homelab/netbird
docker compose -f client.compose.yaml config --quiet
docker compose -f client.compose.yaml up -d
docker compose -f client.compose.yaml exec netbird-client netbird status
```
The client uses host networking and requires `NET_ADMIN`, `SYS_ADMIN`, `SYS_RESOURCE`, and `/dev/net/tun`. Remove it without affecting the server stack:
```bash
docker compose -f client.compose.yaml down
```
## Operations
Inspect status and logs:
```bash
docker compose ps
docker compose logs --tail=200 netbird-server dashboard
```
Stop or remove containers without deleting data:
```bash
docker compose down
```
Do not add `-v` to `docker compose down`; it would delete the NetBird datastore.
## Backup and restore
Back up both the persistent volume and the ignored secret files. For a consistent SQLite backup, briefly stop the server first and store the resulting archive and `datastore-encryption-key` in an encrypted backup:
```bash
cd /srv/homelab/netbird
mkdir -p backups
docker compose stop netbird-server
docker run --rm \
-v netbird_data:/data:ro \
-v "$PWD/backups:/backup" \
busybox:1.37.0 \
tar -C /data -czf "/backup/netbird-data-$(date -u +%Y%m%dT%H%M%SZ).tar.gz" .
docker compose start netbird-server
```
Also securely back up:
- `secrets/datastore-encryption-key` — required to decrypt stored secrets.
- `secrets/relay-auth-secret` — keeps issued relay credentials valid across restoration.
- `netbird/.env` — optional, but it records the public and internal hostnames.
Test a restore in an isolated Docker host before relying on a backup.
## Upgrade
1. Take and verify a backup.
2. Review NetBird release notes for server, client, and dashboard compatibility.
3. Update the pinned tags in `compose.yaml`; update `client.compose.yaml` separately when deploying the client.
4. Pull and recreate the selected services:
```bash
docker compose pull
docker compose up -d
```
The image tags are intentionally pinned instead of using `latest`, matching this repository's pull-on-deploy policy.
+31
View File
@@ -0,0 +1,31 @@
name: netbird-client
services:
netbird-client:
image: netbirdio/netbird:0.79.0
container_name: netbird-client
hostname: "${NETBIRD_CLIENT_HOSTNAME:?Set NETBIRD_CLIENT_HOSTNAME in netbird/.env}"
restart: unless-stopped
cap_add:
- NET_ADMIN
- SYS_ADMIN
- SYS_RESOURCE
devices:
- /dev/net/tun
network_mode: host
environment:
NB_SETUP_KEY: "${NB_SETUP_KEY:?Set NB_SETUP_KEY in netbird/.env after creating a peer setup key}"
NB_MANAGEMENT_URL: "https://${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}"
volumes:
- netbird-client:/var/lib/netbird
healthcheck:
test: ["CMD", "/usr/local/bin/netbird", "status", "--check", "live"]
interval: 30s
timeout: 5s
retries: 5
start_period: 30s
stop_grace_period: 30s
volumes:
netbird-client:
name: netbird-client
+152
View File
@@ -0,0 +1,152 @@
name: netbird
services:
netbird-server:
image: netbirdio/netbird-server:0.79.0
container_name: netbird-server
restart: unless-stopped
environment:
NETBIRD_DOMAIN: "${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}"
NETBIRD_PROXY_SUBNET: "${NETBIRD_PROXY_SUBNET:?Set NETBIRD_PROXY_SUBNET in netbird/.env (run setup.sh)}"
entrypoint:
- /bin/sh
- /opt/netbird/entrypoint.sh
command:
- --config
- /run/netbird/config.yaml
ports:
- "3478:3478/udp"
volumes:
- netbird_data:/var/lib/netbird
- ./config.template.yaml:/opt/netbird/config.template.yaml:ro
- ./entrypoint.sh:/opt/netbird/entrypoint.sh:ro
secrets:
- relay_auth_secret
- datastore_encryption_key
tmpfs:
- /run/netbird:mode=0700
healthcheck:
test:
- CMD
- bash
- -ec
- exec 3<>/dev/tcp/127.0.0.1/80
interval: 30s
timeout: 5s
retries: 5
start_period: 30s
stop_grace_period: 30s
labels:
- "traefik.enable=true"
- "traefik.http.services.netbird-server.loadbalancer.server.port=80"
- "traefik.http.services.netbird-server-h2c.loadbalancer.server.port=80"
- "traefik.http.services.netbird-server-h2c.loadbalancer.server.scheme=h2c"
# gRPC routers
# Prod Router
- "traefik.http.routers.netbird-grpc.rule=Host(`${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))"
- "traefik.http.routers.netbird-grpc.entrypoints=websecure"
- "traefik.http.routers.netbird-grpc.service=netbird-server-h2c"
- "traefik.http.routers.netbird-grpc.priority=100"
- "traefik.http.routers.netbird-grpc.tls=true"
- "traefik.http.routers.netbird-grpc.tls.certresolver=letsencrypt"
# Local Router
- "traefik.http.routers.netbird-grpc-local.rule=Host(`${NETBIRD_LOCAL_DOMAIN:?Set NETBIRD_LOCAL_DOMAIN in netbird/.env}`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))"
- "traefik.http.routers.netbird-grpc-local.entrypoints=websecure"
- "traefik.http.routers.netbird-grpc-local.service=netbird-server-h2c"
- "traefik.http.routers.netbird-grpc-local.priority=100"
- "traefik.http.routers.netbird-grpc-local.tls=true"
# Dev Router
- "traefik.http.routers.netbird-grpc-dev.rule=Host(`${NETBIRD_DEV_DOMAIN:?Set NETBIRD_DEV_DOMAIN in netbird/.env}`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))"
- "traefik.http.routers.netbird-grpc-dev.entrypoints=websecure"
- "traefik.http.routers.netbird-grpc-dev.service=netbird-server-h2c"
- "traefik.http.routers.netbird-grpc-dev.priority=100"
- "traefik.http.routers.netbird-grpc-dev.tls=true"
# Backend routers
# Prod Router
- "traefik.http.routers.netbird-backend.rule=Host(`${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))"
- "traefik.http.routers.netbird-backend.entrypoints=websecure"
- "traefik.http.routers.netbird-backend.service=netbird-server"
- "traefik.http.routers.netbird-backend.priority=100"
- "traefik.http.routers.netbird-backend.tls=true"
- "traefik.http.routers.netbird-backend.tls.certresolver=letsencrypt"
# Local Router
- "traefik.http.routers.netbird-backend-local.rule=Host(`${NETBIRD_LOCAL_DOMAIN:?Set NETBIRD_LOCAL_DOMAIN in netbird/.env}`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))"
- "traefik.http.routers.netbird-backend-local.entrypoints=websecure"
- "traefik.http.routers.netbird-backend-local.service=netbird-server"
- "traefik.http.routers.netbird-backend-local.priority=100"
- "traefik.http.routers.netbird-backend-local.tls=true"
# Dev Router
- "traefik.http.routers.netbird-backend-dev.rule=Host(`${NETBIRD_DEV_DOMAIN:?Set NETBIRD_DEV_DOMAIN in netbird/.env}`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))"
- "traefik.http.routers.netbird-backend-dev.entrypoints=websecure"
- "traefik.http.routers.netbird-backend-dev.service=netbird-server"
- "traefik.http.routers.netbird-backend-dev.priority=100"
- "traefik.http.routers.netbird-backend-dev.tls=true"
networks:
- proxy
dashboard:
image: netbirdio/dashboard:v2.90.10
container_name: netbird-dashboard
restart: unless-stopped
environment:
NETBIRD_MGMT_API_ENDPOINT: "https://${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}"
NETBIRD_MGMT_GRPC_API_ENDPOINT: "https://${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}"
AUTH_AUDIENCE: netbird-dashboard
AUTH_CLIENT_ID: netbird-dashboard
AUTH_CLIENT_SECRET: ""
AUTH_AUTHORITY: "https://${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}/oauth2"
AUTH_SUPPORTED_SCOPES: openid profile email groups
AUTH_REDIRECT_URI: /nb-auth
AUTH_SILENT_REDIRECT_URI: /nb-silent-auth
USE_AUTH0: "false"
LETSENCRYPT_DOMAIN: none
depends_on:
netbird-server:
condition: service_healthy
healthcheck:
test: ["CMD", "curl", "--fail", "--silent", "--show-error", "http://127.0.0.1/"]
interval: 30s
timeout: 5s
retries: 5
start_period: 15s
labels:
- "traefik.enable=true"
- "traefik.http.services.netbird-dashboard.loadbalancer.server.port=80"
# Dashboard catch-all routers
# Prod Router
- "traefik.http.routers.netbird-dashboard.rule=Host(`${NETBIRD_DOMAIN:?Set NETBIRD_DOMAIN in netbird/.env}`)"
- "traefik.http.routers.netbird-dashboard.entrypoints=websecure"
- "traefik.http.routers.netbird-dashboard.service=netbird-dashboard"
- "traefik.http.routers.netbird-dashboard.priority=1"
- "traefik.http.routers.netbird-dashboard.tls=true"
- "traefik.http.routers.netbird-dashboard.tls.certresolver=letsencrypt"
# Local Router
- "traefik.http.routers.netbird-dashboard-local.rule=Host(`${NETBIRD_LOCAL_DOMAIN:?Set NETBIRD_LOCAL_DOMAIN in netbird/.env}`)"
- "traefik.http.routers.netbird-dashboard-local.entrypoints=websecure"
- "traefik.http.routers.netbird-dashboard-local.service=netbird-dashboard"
- "traefik.http.routers.netbird-dashboard-local.priority=1"
- "traefik.http.routers.netbird-dashboard-local.tls=true"
# Dev Router
- "traefik.http.routers.netbird-dashboard-dev.rule=Host(`${NETBIRD_DEV_DOMAIN:?Set NETBIRD_DEV_DOMAIN in netbird/.env}`)"
- "traefik.http.routers.netbird-dashboard-dev.entrypoints=websecure"
- "traefik.http.routers.netbird-dashboard-dev.service=netbird-dashboard"
- "traefik.http.routers.netbird-dashboard-dev.priority=1"
- "traefik.http.routers.netbird-dashboard-dev.tls=true"
networks:
- proxy
networks:
proxy:
external: true
volumes:
netbird_data:
name: netbird_data
secrets:
relay_auth_secret:
file: ./secrets/relay-auth-secret
datastore_encryption_key:
file: ./secrets/datastore-encryption-key
+26
View File
@@ -0,0 +1,26 @@
server:
listenAddress: ":80"
exposedAddress: "https://__NETBIRD_DOMAIN__:443"
stunPorts:
- 3478
metricsPort: 9090
healthcheckAddress: ":9000"
logLevel: info
logFile: console
authSecret: "__NETBIRD_AUTH_SECRET__"
dataDir: "/var/lib/netbird"
disableAnonymousMetrics: true
auth:
issuer: "https://__NETBIRD_DOMAIN__/oauth2"
signKeyRefreshEnabled: true
dashboardRedirectURIs:
- "https://__NETBIRD_DOMAIN__/nb-auth"
- "https://__NETBIRD_DOMAIN__/nb-silent-auth"
reverseProxy:
trustedHTTPProxies:
- "__NETBIRD_PROXY_SUBNET__"
trustedPeers:
- "__NETBIRD_PROXY_SUBNET__"
store:
engine: sqlite
encryptionKey: "__NETBIRD_ENCRYPTION_KEY__"
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: netbird-prod-tls
namespace: netbird
spec:
secretName: netbird-prod-tls
dnsNames:
- nb.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: netbird
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+160
View File
@@ -0,0 +1,160 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: netbird-config
namespace: netbird
data:
# Public hostname, rendered into the server config by entrypoint.sh.
NETBIRD_DOMAIN: "nb.forust.xyz"
NETBIRD_PROXY_SUBNET: "10.244.0.0/16"
NETBIRD_MGMT_API_ENDPOINT: "https://nb.forust.xyz"
NETBIRD_MGMT_GRPC_API_ENDPOINT: "https://nb.forust.xyz"
AUTH_AUDIENCE: "netbird-dashboard"
AUTH_CLIENT_ID: "netbird-dashboard"
AUTH_CLIENT_SECRET: ""
AUTH_AUTHORITY: "https://nb.forust.xyz/oauth2"
AUTH_SUPPORTED_SCOPES: "openid profile email groups"
AUTH_REDIRECT_URI: "/nb-auth"
AUTH_SILENT_REDIRECT_URI: "/nb-silent-auth"
USE_AUTH0: "false"
LETSENCRYPT_DOMAIN: "none"
config.template.yaml: |
server:
listenAddress: ":80"
exposedAddress: "https://__NETBIRD_DOMAIN__:443"
stunPorts:
- 3478
metricsPort: 9090
healthcheckAddress: ":9000"
logLevel: info
logFile: console
authSecret: "__NETBIRD_AUTH_SECRET__"
dataDir: "/var/lib/netbird"
disableAnonymousMetrics: true
auth:
issuer: "https://__NETBIRD_DOMAIN__/oauth2"
signKeyRefreshEnabled: true
dashboardRedirectURIs:
- "https://__NETBIRD_DOMAIN__/nb-auth"
- "https://__NETBIRD_DOMAIN__/nb-silent-auth"
reverseProxy:
trustedHTTPProxies:
- "__NETBIRD_PROXY_SUBNET__"
trustedPeers:
- "__NETBIRD_PROXY_SUBNET__"
store:
engine: sqlite
encryptionKey: "__NETBIRD_ENCRYPTION_KEY__"
entrypoint.sh: |
#!/bin/sh
set -eu
umask 077
TEMPLATE_PATH=/opt/netbird/config.template.yaml
RENDERED_PATH=/run/netbird/config.yaml
RELAY_SECRET_PATH=/run/secrets/relay_auth_secret
ENCRYPTION_KEY_PATH=/run/secrets/datastore_encryption_key
is_valid_proxy_subnet() {
candidate="$1"
case "$candidate" in
0.0.0.0/0)
return 1
;;
*/*)
address="${candidate%%/*}"
prefix="${candidate#*/}"
;;
*)
return 1
;;
esac
case "$prefix" in
0|[1-9]|[1-2][0-9]|3[0-2]) ;;
*)
return 1
;;
esac
old_ifs="$IFS"
IFS=.
# shellcheck disable=SC2086
set -- $address
IFS="$old_ifs"
[ "$#" -eq 4 ] || return 1
for octet do
case "$octet" in
0|[1-9]|[1-9][0-9]|1[0-9][0-9]|2[0-4][0-9]|25[0-5]) ;;
*)
return 1
;;
esac
done
}
read_secret() {
secret_path="$1"
if [ ! -r "$secret_path" ]; then
echo "Required secret is not readable: $secret_path" >&2
exit 1
fi
secret_value="$(cat "$secret_path")"
if [ -z "$secret_value" ]; then
echo "Required secret is empty: $secret_path" >&2
exit 1
fi
printf '%s' "$secret_value"
}
if [ -z "${NETBIRD_DOMAIN:-}" ]; then
echo "NETBIRD_DOMAIN must be set" >&2
exit 1
fi
case "$NETBIRD_DOMAIN" in
*[!A-Za-z0-9.-]*)
echo "NETBIRD_DOMAIN contains unsupported characters" >&2
exit 1
;;
esac
if [ -z "${NETBIRD_PROXY_SUBNET:-}" ] || [ "$NETBIRD_PROXY_SUBNET" = "auto" ]; then
echo "NETBIRD_PROXY_SUBNET must be an explicit IPv4 CIDR; run netbird/setup.sh first" >&2
exit 1
fi
if ! is_valid_proxy_subnet "$NETBIRD_PROXY_SUBNET"; then
echo "NETBIRD_PROXY_SUBNET must be a non-default IPv4 CIDR, for example 172.20.0.0/16" >&2
exit 1
fi
if [ "$#" -ne 2 ] || [ "$1" != "--config" ] || [ "$2" != "$RENDERED_PATH" ]; then
echo "Expected: --config $RENDERED_PATH" >&2
exit 1
fi
relay_secret="$(read_secret "$RELAY_SECRET_PATH")"
encryption_key="$(read_secret "$ENCRYPTION_KEY_PATH")"
mkdir -p "$(dirname "$RENDERED_PATH")"
sed \
-e "s|__NETBIRD_DOMAIN__|${NETBIRD_DOMAIN}|g" \
-e "s|__NETBIRD_AUTH_SECRET__|${relay_secret}|g" \
-e "s|__NETBIRD_ENCRYPTION_KEY__|${encryption_key}|g" \
-e "s|__NETBIRD_PROXY_SUBNET__|${NETBIRD_PROXY_SUBNET}|g" \
"$TEMPLATE_PATH" >"$RENDERED_PATH"
if grep -q '__NETBIRD_' "$RENDERED_PATH"; then
echo "Rendered NetBird configuration still contains unresolved placeholders" >&2
exit 1
fi
exec /go/bin/netbird-server "$@"
+83
View File
@@ -0,0 +1,83 @@
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: netbird-prod
namespace: netbird
spec:
entryPoints:
- websecure
routes:
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))
kind: Rule
priority: 100
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: netbird-server-service
port: 80
scheme: h2c
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))
kind: Rule
priority: 100
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: netbird-server-service
port: 80
- match: Host(`nb.forust.xyz`)
kind: Rule
priority: 1
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: netbird-dashboard-service
port: 80
tls:
secretName: netbird-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: netbird-local
namespace: netbird
spec:
entryPoints:
- websecure
routes:
- match: (Host(`netbird.workstation.internal`) || Host(`netbird.gigaforust.internal`)) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))
kind: Rule
priority: 100
services:
- name: netbird-server-service
port: 80
scheme: h2c
- match: (Host(`netbird.workstation.internal`) || Host(`netbird.gigaforust.internal`)) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))
kind: Rule
priority: 100
services:
- name: netbird-server-service
port: 80
- match: Host(`netbird.workstation.internal`) || Host(`netbird.gigaforust.internal`)
kind: Rule
priority: 1
services:
- name: netbird-dashboard-service
port: 80
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRouteUDP
metadata:
name: netbird-stun
namespace: netbird
spec:
entryPoints:
- netbird-stun
routes:
- services:
- name: netbird-server-service
port: 3478
+4
View File
@@ -0,0 +1,4 @@
apiVersion: v1
kind: Namespace
metadata:
name: netbird
+181
View File
@@ -0,0 +1,181 @@
apiVersion: v1
kind: Service
metadata:
name: netbird-server-service
namespace: netbird
spec:
selector:
app: netbird-server
ports:
- port: 80
name: http
targetPort: 80
protocol: TCP
- port: 3478
name: stun
targetPort: 3478
protocol: UDP
---
apiVersion: v1
kind: Service
metadata:
name: netbird-dashboard-service
namespace: netbird
spec:
selector:
app: netbird-dashboard
ports:
- port: 80
name: http
targetPort: 80
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: netbird-server-deployment
namespace: netbird
spec:
replicas: 1
selector:
matchLabels:
app: netbird-server
template:
metadata:
labels:
app: netbird-server
spec:
containers:
- name: netbird-server
image: netbirdio/netbird-server:0.79.0
command: ["/bin/sh", "/opt/netbird/entrypoint.sh", "--config", "/run/netbird/config.yaml"]
envFrom:
- configMapRef:
name: netbird-config
ports:
- containerPort: 80
name: http
protocol: TCP
- containerPort: 3478
name: stun
protocol: UDP
volumeMounts:
- name: netbird-data
mountPath: /var/lib/netbird
- name: netbird-files
mountPath: /opt/netbird
readOnly: true
- name: netbird-secrets
mountPath: /run/secrets/relay_auth_secret
subPath: relay_auth_secret
readOnly: true
- name: netbird-secrets
mountPath: /run/secrets/datastore_encryption_key
subPath: datastore_encryption_key
readOnly: true
- name: netbird-run
mountPath: /run/netbird
readinessProbe:
tcpSocket:
port: 80
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 5
livenessProbe:
tcpSocket:
port: 80
initialDelaySeconds: 60
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 5
resources:
requests:
memory: "256Mi"
cpu: "250m"
limits:
memory: "1Gi"
cpu: "1000m"
volumes:
- name: netbird-data
persistentVolumeClaim:
claimName: netbird-pvc
- name: netbird-files
configMap:
name: netbird-config
defaultMode: 0755
items:
- key: config.template.yaml
path: config.template.yaml
- key: entrypoint.sh
path: entrypoint.sh
- name: netbird-secrets
secret:
secretName: netbird-secrets
items:
- key: relay_auth_secret
path: relay_auth_secret
- key: datastore_encryption_key
path: datastore_encryption_key
- name: netbird-run
emptyDir:
medium: Memory
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: netbird-dashboard-deployment
namespace: netbird
spec:
replicas: 1
selector:
matchLabels:
app: netbird-dashboard
template:
metadata:
labels:
app: netbird-dashboard
spec:
containers:
- name: dashboard
image: netbirdio/dashboard:v2.90.10
envFrom:
- configMapRef:
name: netbird-config
ports:
- containerPort: 80
name: http
readinessProbe:
httpGet:
path: /
port: 80
initialDelaySeconds: 15
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 5
livenessProbe:
httpGet:
path: /
port: 80
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 5
resources:
requests:
memory: "64Mi"
cpu: "50m"
limits:
memory: "256Mi"
cpu: "300m"
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: netbird-pvc
namespace: netbird
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 2Gi
+11
View File
@@ -0,0 +1,11 @@
apiVersion: v1
kind: Secret
metadata:
name: netbird-secrets
namespace: netbird
type: Opaque
stringData:
# hex, 64 chars: openssl rand -hex 32
relay_auth_secret: "REPLACE_ME"
# base64, 44 chars: openssl rand -base64 32
datastore_encryption_key: "REPLACE_ME"
+32
View File
@@ -0,0 +1,32 @@
POSTGRES_DB=netbox
POSTGRES_USER=netbox
POSTGRES_PASSWORD=CHANGE_ME_POSTGRES_PASSWORD
DB_NAME=netbox
DB_USER=netbox
DB_PASSWORD=CHANGE_ME_POSTGRES_PASSWORD
DB_HOST=postgres
DB_PORT=5432
DB_SSLMODE=disable
REDIS_HOST=redis
REDIS_PORT=6379
REDIS_PASSWORD=CHANGE_ME_REDIS_PASSWORD
REDIS_DATABASE=0
REDIS_CACHE_HOST=redis-cache
REDIS_CACHE_PORT=6379
REDIS_CACHE_PASSWORD=CHANGE_ME_REDIS_CACHE_PASSWORD
REDIS_CACHE_DATABASE=1
ALLOWED_HOSTS=localhost,127.0.0.1,[::1],netbox.forust.xyz,netbox.workstation.internal
CSRF_TRUSTED_ORIGINS=https://netbox.forust.xyz,https://netbox.workstation.internal
SECRET_KEY=CHANGE_ME_DJANGO_SECRET_KEY
API_TOKEN_PEPPER_1=CHANGE_ME_API_TOKEN_PEPPER
TIME_ZONE=Europe/Bratislava
TZ=Europe/Bratislava
SKIP_SUPERUSER=false
SUPERUSER_NAME=admin
SUPERUSER_EMAIL=admin@example.com
SUPERUSER_PASSWORD=CHANGE_ME_SUPERUSER_PASSWORD
+96
View File
@@ -0,0 +1,96 @@
# NetBox
NetBox for homelab documentation and visualization. Two runtimes are available:
| Runtime | Manifest | Purpose |
| ------- | -------------- | -------------------------------------------------------------- |
| Docker | `compose.yaml` | Local stand on `127.0.0.1:8000` (no public exposure) |
| k8s | `k8s/` | Homelab service on `netbox.forust.xyz` (and the internal name) |
Both use the same image (`netboxcommunity/netbox:v4.7-5.1.1`) and Valkey for tasks
plus a second logical database for caching. The Docker stand keeps its own
PostgreSQL container, while the k8s deployment uses the shared `database` cluster
(`postgres.database.svc.cluster.local:5432`, role/database `netbox`); only Valkey
stays a per-service StatefulSet.
## Docker Compose
```bash
cp .env.example .env
# replace CHANGE_ME
docker compose up -d
```
The UI is available at <http://localhost:8000>. The port is bound to `127.0.0.1`
intentionally, so this stand is not exposed on the LAN or public interfaces.
The `netbox` service is also attached to the external `proxy` network and carries
Traefik labels for `netbox.forust.xyz` and `netbox.workstation.internal`. Those
labels only take effect while the Docker Traefik stack is running; it is currently
stopped, and the live ingress path in this homelab is the k8s Traefik.
Inspect startup and health with:
```bash
docker compose ps
docker compose logs -f netbox
```
Stop it with `docker compose down`; data is kept in the named volumes
`netbox-postgres`, `netbox-media-files`, `netbox-reports-files`,
`netbox-scripts-files` and `netbox-redis-data`.
## Kubernetes
`k8s/` is deployed in the homelab cluster and serves `netbox.forust.xyz` publicly
plus `netbox.workstation.internal` / `netbox.gigaforust.internal` internally. To
rebuild it from scratch:
```bash
# 1. shared PostgreSQL: the password lives in the shared secret, NetBox keeps a copy
kubectl -n database patch secret postgres-shared-secrets \
--type merge -p '{"stringData":{"NETBOX_DB_PASSWORD":"<same value>"}}'
kubectl -n database exec postgres17-0 -- psql -U postgres -d postgres \
-c 'CREATE ROLE netbox LOGIN PASSWORD ...' -c 'CREATE DATABASE netbox OWNER netbox'
# 2. secrets first: the deploy workflow never applies *secret*.yaml
cp k8s/secrets.yaml.example k8s/secrets.yaml # replace CHANGE_ME
kubectl apply -f k8s/secrets.yaml
# 3. manifests
kubectl apply -f k8s/
```
The shared cluster is reached at `postgres.database.svc.cluster.local:5432`. Its
NetworkPolicy (`postgres/k8s/network-policy.yaml`) must list the `netbox` namespace
or connections are dropped, and `postgres/initdb/01-create-databases.sh` already
creates the role and database on a fresh data directory. NetBox has no PostgreSQL
StatefulSet of its own — only `netbox-valkey`.
`netbox.forust.xyz` resolves to this host (`78.98.72.122`) through the `DOMAINS`
list in the `default/cfddns` secret. cert-manager issues `netbox-prod-tls` with the
`letsencrypt-prod` issuer, the internal route uses `internal-wildcard-tls`.
Resources are permanent again now that the first-boot migrations are complete:
the web container reserves `100m`/`512Mi` and is capped at `2` CPU/`2Gi`, the
worker reserves `50m`/`256Mi` and is capped at `1` CPU/`1Gi`, and Valkey reserves
`25m`/`64Mi` and is capped at `250m`/`256Mi`. The deliberately generous CPU caps
leave enough headroom for future schema migrations without letting one process
consume the whole node.
The first start applies ~810 migrations, each in its own transaction with DDL and
a commit; every later start is a no-op. The startup probe allows 15 minutes and
`progressDeadlineSeconds` is 1800 for the same reason. Probes run inside the pod
and explicitly set `Host: netbox.forust.xyz`; a kubelet `httpGet.host` field would
replace the probe destination with that public hostname and bypass the pod.
## Secrets
- `netbox/.env` (compose) and `netbox/k8s/secrets.yaml` (k8s) are gitignored. Only
`.env.example` and `k8s/secrets.yaml.example` are committed.
- `netbox/configuration/configuration.py` is env-driven: hosts, database, Redis and
the Django keys all come from the environment, so the same settings file works in
both runtimes. The k8s copy lives in the `netbox-settings` ConfigMap
(`k8s/settings.yaml`) and must be kept in sync with the file.
- Rotating `SECRET_KEY` invalidates all sessions; rotating `API_TOKEN_PEPPER_1`
invalidates every API token.
View File
Whitespace-only changes.
+137
View File
@@ -0,0 +1,137 @@
services:
netbox:
image: docker.io/netboxcommunity/netbox:v4.7-5.1.1
container_name: netbox
restart: unless-stopped
user: "netbox:root"
ports:
- "127.0.0.1:8000:8080"
env_file:
- .env
environment:
GRANIAN_WORKERS: "2"
depends_on:
postgres:
condition: service_healthy
redis:
condition: service_healthy
redis-cache:
condition: service_healthy
volumes:
- ./configuration:/etc/netbox/config:z,ro
- netbox-media-files:/opt/netbox/netbox/media
- netbox-reports-files:/opt/netbox/netbox/reports
- netbox-scripts-files:/opt/netbox/netbox/scripts
networks:
- default
- proxy
labels:
- "traefik.enable=true"
- "traefik.http.services.netbox.loadbalancer.server.port=8080"
# Prod Router
- "traefik.http.routers.netbox.rule=Host(`netbox.forust.xyz`)"
- "traefik.http.routers.netbox.entrypoints=websecure"
- "traefik.http.routers.netbox.middlewares=security-headers@file"
- "traefik.http.routers.netbox.tls.certresolver=letsencrypt"
# Local Router
- "traefik.http.routers.netbox-local.rule=Host(`netbox.workstation.internal`)"
- "traefik.http.routers.netbox-local.entrypoints=websecure"
- "traefik.http.routers.netbox-local.tls=true"
healthcheck:
test: ["CMD", "/opt/netbox/health.sh"]
start_period: 600s
timeout: 5s
interval: 15s
retries: 10
netbox-worker:
image: docker.io/netboxcommunity/netbox:v4.7-5.1.1
container_name: netbox-worker
restart: unless-stopped
user: "netbox:root"
command:
- /opt/netbox/venv/bin/python
- /opt/netbox/netbox/manage.py
- rqworker
env_file:
- .env
depends_on:
netbox:
condition: service_healthy
volumes:
- ./configuration:/etc/netbox/config:z,ro
- netbox-media-files:/opt/netbox/netbox/media
- netbox-reports-files:/opt/netbox/netbox/reports
- netbox-scripts-files:/opt/netbox/netbox/scripts
healthcheck:
test: ["CMD-SHELL", "ps -ef | grep -q '[r]qworker'"]
start_period: 30s
timeout: 5s
interval: 15s
retries: 10
postgres:
image: docker.io/postgres:18.6-alpine
container_name: netbox-postgres
restart: unless-stopped
environment:
POSTGRES_DB: "${POSTGRES_DB:?POSTGRES_DB must be set}"
POSTGRES_USER: "${POSTGRES_USER:?POSTGRES_USER must be set}"
POSTGRES_PASSWORD: "${POSTGRES_PASSWORD:?POSTGRES_PASSWORD must be set}"
volumes:
- netbox-postgres:/var/lib/postgresql
healthcheck:
test: ["CMD-SHELL", 'pg_isready -q -t 2 -d "$${POSTGRES_DB}" -U "$${POSTGRES_USER}"']
start_period: 20s
timeout: 5s
interval: 10s
retries: 10
redis:
image: docker.io/valkey/valkey:9.1.2-alpine
container_name: netbox-redis
restart: unless-stopped
command:
- sh
- -c
- valkey-server --appendonly yes --requirepass "$$REDIS_PASSWORD"
environment:
REDIS_PASSWORD: "${REDIS_PASSWORD:?REDIS_PASSWORD must be set}"
volumes:
- netbox-redis-data:/data
healthcheck:
test: ["CMD-SHELL", 'valkey-cli --pass "$${REDIS_PASSWORD}" ping | grep -q PONG']
start_period: 5s
timeout: 5s
interval: 5s
retries: 10
redis-cache:
image: docker.io/valkey/valkey:9.1.2-alpine
container_name: netbox-redis-cache
restart: unless-stopped
command:
- sh
- -c
- valkey-server --requirepass "$$REDIS_CACHE_PASSWORD"
environment:
REDIS_CACHE_PASSWORD: "${REDIS_CACHE_PASSWORD:?REDIS_CACHE_PASSWORD must be set}"
healthcheck:
test: ["CMD-SHELL", 'valkey-cli --pass "$${REDIS_CACHE_PASSWORD}" ping | grep -q PONG']
start_period: 5s
timeout: 5s
interval: 5s
retries: 10
volumes:
netbox-media-files:
netbox-reports-files:
netbox-scripts-files:
netbox-postgres:
netbox-redis-data:
networks:
default:
proxy:
external: true
+48
View File
@@ -0,0 +1,48 @@
import os
def _csv(name, default=""):
return [item.strip() for item in os.environ.get(name, default).split(",") if item.strip()]
ALLOWED_HOSTS = _csv("ALLOWED_HOSTS", "localhost,127.0.0.1,[::1]")
CSRF_TRUSTED_ORIGINS = _csv("CSRF_TRUSTED_ORIGINS")
USE_X_FORWARDED_HOST = True
SECURE_PROXY_SSL_HEADER = ("HTTP_X_FORWARDED_PROTO", "https")
DATABASES = {
"default": {
"NAME": os.environ["DB_NAME"],
"USER": os.environ["DB_USER"],
"PASSWORD": os.environ["DB_PASSWORD"],
"HOST": os.environ["DB_HOST"],
"PORT": os.environ.get("DB_PORT", "5432"),
"OPTIONS": {"sslmode": os.environ.get("DB_SSLMODE", "disable")},
"CONN_MAX_AGE": int(os.environ.get("DB_CONN_MAX_AGE", "300")),
}
}
REDIS = {
"tasks": {
"HOST": os.environ["REDIS_HOST"],
"PORT": int(os.environ.get("REDIS_PORT", "6379")),
"PASSWORD": os.environ["REDIS_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_DATABASE", "0")),
"SSL": False,
},
"caching": {
"HOST": os.environ["REDIS_CACHE_HOST"],
"PORT": int(os.environ.get("REDIS_CACHE_PORT", "6379")),
"PASSWORD": os.environ["REDIS_CACHE_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_CACHE_DATABASE", "1")),
"SSL": False,
},
}
SECRET_KEY = os.environ["SECRET_KEY"]
API_TOKEN_PEPPERS = {1: os.environ["API_TOKEN_PEPPER_1"]}
TIME_ZONE = os.environ.get("TIME_ZONE", "UTC")
MEDIA_ROOT = "/opt/netbox/netbox/media"
REPORTS_ROOT = "/opt/netbox/netbox/reports"
SCRIPTS_ROOT = "/opt/netbox/netbox/scripts"
CENSUS_REPORTING_ENABLED = False
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: netbox-prod-tls
namespace: netbox
spec:
secretName: netbox-prod-tls
dnsNames:
- netbox.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: netbox
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+21
View File
@@ -0,0 +1,21 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: netbox-config
namespace: netbox
data:
DB_HOST: "postgres.database.svc.cluster.local"
DB_PORT: "5432"
DB_SSLMODE: "disable"
REDIS_HOST: "netbox-valkey"
REDIS_PORT: "6379"
REDIS_DATABASE: "0"
REDIS_CACHE_HOST: "netbox-valkey"
REDIS_CACHE_PORT: "6379"
REDIS_CACHE_DATABASE: "1"
TIME_ZONE: "Europe/Bratislava"
TZ: "Europe/Bratislava"
GRANIAN_WORKERS: "2"
ALLOWED_HOSTS: "netbox.forust.xyz,netbox.workstation.internal,netbox.gigaforust.internal"
CSRF_TRUSTED_ORIGINS: "https://netbox.forust.xyz,https://netbox.workstation.internal,https://netbox.gigaforust.internal"
SKIP_SUPERUSER: "false"
+36
View File
@@ -0,0 +1,36 @@
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: netbox-prod
namespace: netbox
spec:
entryPoints:
- websecure
routes:
- match: Host(`netbox.forust.xyz`)
kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: netbox-service
port: 8080
tls:
secretName: netbox-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: netbox-local
namespace: netbox
spec:
entryPoints:
- websecure
routes:
- match: Host(`netbox.workstation.internal`) || Host(`netbox.gigaforust.internal`)
kind: Rule
services:
- name: netbox-service
port: 8080
tls:
secretName: internal-wildcard-tls
+4
View File
@@ -0,0 +1,4 @@
apiVersion: v1
kind: Namespace
metadata:
name: netbox
+204
View File
@@ -0,0 +1,204 @@
apiVersion: v1
kind: Service
metadata:
name: netbox-service
namespace: netbox
spec:
selector:
app: netbox
ports:
- name: http
port: 8080
targetPort: http
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: netbox-deployment
namespace: netbox
labels:
app: netbox
spec:
replicas: 1
progressDeadlineSeconds: 300
selector:
matchLabels:
app: netbox
strategy:
# ReadWriteOnce PVC
type: Recreate
template:
metadata:
labels:
app: netbox
spec:
containers:
- name: netbox
image: docker.io/netboxcommunity/netbox:v4.7-5.1.1
ports:
- name: http
containerPort: 8080
envFrom:
- configMapRef:
name: netbox-config
- secretRef:
name: netbox-secrets
volumeMounts:
- name: netbox-config
mountPath: /etc/netbox/config
readOnly: true
- name: netbox-media
mountPath: /opt/netbox/netbox/media
- name: netbox-reports
mountPath: /opt/netbox/netbox/reports
- name: netbox-scripts
mountPath: /opt/netbox/netbox/scripts
startupProbe:
exec:
command:
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
failureThreshold: 90
periodSeconds: 10
readinessProbe:
exec:
command:
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
periodSeconds: 10
livenessProbe:
exec:
command:
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
initialDelaySeconds: 30
periodSeconds: 30
resources:
requests:
cpu: "100m"
memory: "512Mi"
limits:
cpu: "2"
memory: "2Gi"
volumes:
- name: netbox-config
configMap:
name: netbox-settings
- name: netbox-media
persistentVolumeClaim:
claimName: netbox-media-pvc
- name: netbox-reports
persistentVolumeClaim:
claimName: netbox-reports-pvc
- name: netbox-scripts
persistentVolumeClaim:
claimName: netbox-scripts-pvc
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: netbox-worker-deployment
namespace: netbox
labels:
app: netbox-worker
spec:
replicas: 1
progressDeadlineSeconds: 300
selector:
matchLabels:
app: netbox-worker
strategy:
type: Recreate
template:
metadata:
labels:
app: netbox-worker
spec:
containers:
- name: netbox-worker
image: docker.io/netboxcommunity/netbox:v4.7-5.1.1
command:
- /opt/netbox/venv/bin/python
- netbox/manage.py
- rqworker
workingDir: /opt/netbox
envFrom:
- configMapRef:
name: netbox-config
- secretRef:
name: netbox-secrets
volumeMounts:
- name: netbox-config
mountPath: /etc/netbox/config
readOnly: true
- name: netbox-media
mountPath: /opt/netbox/netbox/media
- name: netbox-reports
mountPath: /opt/netbox/netbox/reports
- name: netbox-scripts
mountPath: /opt/netbox/netbox/scripts
resources:
requests:
cpu: "50m"
memory: "256Mi"
limits:
cpu: "1"
memory: "1Gi"
volumes:
- name: netbox-config
configMap:
name: netbox-settings
- name: netbox-media
persistentVolumeClaim:
claimName: netbox-media-pvc
- name: netbox-reports
persistentVolumeClaim:
claimName: netbox-reports-pvc
- name: netbox-scripts
persistentVolumeClaim:
claimName: netbox-scripts-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: netbox-media-pvc
namespace: netbox
spec:
accessModes: ["ReadWriteOnce"]
resources:
requests:
storage: 2Gi
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: netbox-reports-pvc
namespace: netbox
spec:
accessModes: ["ReadWriteOnce"]
resources:
requests:
storage: 1Gi
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: netbox-scripts-pvc
namespace: netbox
spec:
accessModes: ["ReadWriteOnce"]
resources:
requests:
storage: 1Gi
+18
View File
@@ -0,0 +1,18 @@
apiVersion: v1
kind: Secret
metadata:
name: netbox-secrets
namespace: netbox
type: Opaque
stringData:
DB_NAME: "netbox"
DB_USER: "netbox"
DB_PASSWORD: "CHANGE_ME_POSTGRES_PASSWORD"
REDIS_PASSWORD: "CHANGE_ME_VALKEY_PASSWORD"
REDIS_CACHE_PASSWORD: "CHANGE_ME_VALKEY_PASSWORD"
VALKEY_PASSWORD: "CHANGE_ME_VALKEY_PASSWORD"
SECRET_KEY: "CHANGE_ME_DJANGO_SECRET_KEY"
API_TOKEN_PEPPER_1: "CHANGE_ME_API_TOKEN_PEPPER"
SUPERUSER_NAME: "admin"
SUPERUSER_EMAIL: "admin@example.com"
SUPERUSER_PASSWORD: "CHANGE_ME_SUPERUSER_PASSWORD"
+56
View File
@@ -0,0 +1,56 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: netbox-settings
namespace: netbox
data:
# Sync wit netbox/configuration/configuration.py (the Docker mounts that file).
configuration.py: |
import os
def _csv(name, default=""):
return [item.strip() for item in os.environ.get(name, default).split(",") if item.strip()]
ALLOWED_HOSTS = _csv("ALLOWED_HOSTS", "localhost,127.0.0.1,[::1]")
CSRF_TRUSTED_ORIGINS = _csv("CSRF_TRUSTED_ORIGINS")
USE_X_FORWARDED_HOST = True
SECURE_PROXY_SSL_HEADER = ("HTTP_X_FORWARDED_PROTO", "https")
DATABASES = {
"default": {
"NAME": os.environ["DB_NAME"],
"USER": os.environ["DB_USER"],
"PASSWORD": os.environ["DB_PASSWORD"],
"HOST": os.environ["DB_HOST"],
"PORT": os.environ.get("DB_PORT", "5432"),
"OPTIONS": {"sslmode": os.environ.get("DB_SSLMODE", "disable")},
"CONN_MAX_AGE": int(os.environ.get("DB_CONN_MAX_AGE", "300")),
}
}
REDIS = {
"tasks": {
"HOST": os.environ["REDIS_HOST"],
"PORT": int(os.environ.get("REDIS_PORT", "6379")),
"PASSWORD": os.environ["REDIS_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_DATABASE", "0")),
"SSL": False,
},
"caching": {
"HOST": os.environ["REDIS_CACHE_HOST"],
"PORT": int(os.environ.get("REDIS_CACHE_PORT", "6379")),
"PASSWORD": os.environ["REDIS_CACHE_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_CACHE_DATABASE", "1")),
"SSL": False,
},
}
SECRET_KEY = os.environ["SECRET_KEY"]
API_TOKEN_PEPPERS = {1: os.environ["API_TOKEN_PEPPER_1"]}
TIME_ZONE = os.environ.get("TIME_ZONE", "UTC")
MEDIA_ROOT = "/opt/netbox/netbox/media"
REPORTS_ROOT = "/opt/netbox/netbox/reports"
SCRIPTS_ROOT = "/opt/netbox/netbox/scripts"
CENSUS_REPORTING_ENABLED = False
+82
View File
@@ -0,0 +1,82 @@
apiVersion: v1
kind: Service
metadata:
name: netbox-valkey
namespace: netbox
labels:
app: netbox-valkey
spec:
clusterIP: None
selector:
app: netbox-valkey
ports:
- name: valkey
port: 6379
targetPort: valkey
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: netbox-valkey
namespace: netbox
labels:
app: netbox-valkey
spec:
serviceName: netbox-valkey
replicas: 1
selector:
matchLabels:
app: netbox-valkey
template:
metadata:
labels:
app: netbox-valkey
spec:
containers:
- name: valkey
image: docker.io/valkey/valkey:9.1.2-alpine
command:
- sh
- -c
- valkey-server --appendonly yes --save 30 1 --loglevel warning --requirepass "$VALKEY_PASSWORD"
env:
- name: VALKEY_PASSWORD
valueFrom:
secretKeyRef:
name: netbox-secrets
key: VALKEY_PASSWORD
ports:
- name: valkey
containerPort: 6379
volumeMounts:
- name: valkey-data
mountPath: /data
startupProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$VALKEY_PASSWORD" ping | grep -q PONG']
failureThreshold: 20
periodSeconds: 5
readinessProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$VALKEY_PASSWORD" ping | grep -q PONG']
periodSeconds: 10
livenessProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$VALKEY_PASSWORD" ping | grep -q PONG']
initialDelaySeconds: 20
periodSeconds: 20
resources:
requests:
cpu: "25m"
memory: "64Mi"
limits:
cpu: "250m"
memory: "256Mi"
volumeClaimTemplates:
- metadata:
name: valkey-data
spec:
accessModes: ["ReadWriteOnce"]
resources:
requests:
storage: 1Gi
+28
View File
@@ -0,0 +1,28 @@
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: netronome-prod-tls
namespace: netronome
spec:
secretName: netronome-prod-tls
dnsNames:
- nm.forust.xyz
issuerRef:
name: letsencrypt-prod
kind: ClusterIssuer
---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: netronome
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
+3 -1
View File
@@ -16,7 +16,7 @@ spec:
- name: netronome-service
port: 7575
tls:
certResolver: letsencrypt
secretName: netronome-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
@@ -32,3 +32,5 @@ spec:
services:
- name: netronome-service
port: 7575
tls:
secretName: internal-wildcard-tls
Loaded 100 of 163 files, more files were not shown because too many files have changed in this diff. Show more