Compare commits
176
Commits
sidetree
...
f0e0fba125
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f0e0fba125 | ||
|
|
f7cd75d65e | ||
|
|
6a9a460769 | ||
|
|
ac0f845da6 | ||
|
|
f54589a05c | ||
|
|
f49d91b63d | ||
|
|
aafa74b70a | ||
|
|
4f74fe1778 | ||
|
|
a5d384a4d8 | ||
|
|
af9a22fea9 | ||
|
|
baedea504d | ||
|
|
30995ee009 | ||
|
|
3a05d86e3e | ||
|
|
c00a4724f5 | ||
|
|
284e19ef88 | ||
|
|
0ae0df7473 | ||
|
|
fddd82704f | ||
|
|
0691536f28 | ||
|
|
2b9e34ba4a | ||
|
|
30d2b83efe | ||
|
|
db7bccfd89 | ||
|
|
1505b638ce | ||
|
|
7ce727bc8a | ||
|
|
f22793e32e | ||
|
|
4a8d4feea0 | ||
|
|
1d9a85bef9 | ||
|
|
b625d30568 | ||
|
|
d018a441af | ||
|
|
ecb254017d | ||
|
|
41f18ea993 | ||
|
|
91c344fe2c | ||
|
|
cab6ef2102 | ||
|
|
a2ff9515a3 | ||
|
|
62d39ee4f1 | ||
|
|
f1f7dd4a0a | ||
|
|
f4df6d4437 | ||
|
|
987a89f022 | ||
|
|
b423119632 | ||
|
|
ddbc0cc7e6 | ||
|
|
6ad33d75e6 | ||
|
|
11bfb426c0 | ||
|
|
b7a1835adb | ||
|
|
ad4bb8750d | ||
|
|
f1e00b946f | ||
|
|
7ee7d0c961 | ||
|
|
27ab6b859e | ||
|
|
71e769c002 | ||
|
|
16dd67c2c0 | ||
|
|
c536a16a2a | ||
|
|
a4a4bb4cc5 | ||
|
|
648b354951 | ||
|
|
b0a9b3476d | ||
|
|
ac3bf4a55a | ||
|
|
34fb6f85ba | ||
|
|
66eacd86d1 | ||
|
|
54f43fc2c7 | ||
|
|
451d0dc6b7 | ||
|
|
88ae20a543 | ||
|
|
6bd183ba73 | ||
|
|
aeefdd8560 | ||
|
|
c10d344fd7 | ||
|
|
6e5f80611b | ||
|
|
5cf0c90258 | ||
|
|
90a452a253 | ||
|
|
a9eabfba02 | ||
|
|
46c7e99b1d | ||
|
|
8b2cf29771 | ||
|
|
7b7fc3bcb0 | ||
|
|
bc1e69ebe0 | ||
|
|
f29bb3d580 | ||
|
|
1f7166026b | ||
|
|
1cbdfc1d6f | ||
|
|
6be3769288 | ||
|
|
ef325cd3b1 | ||
|
|
aa81f1bf8a | ||
|
|
93d768e988 | ||
|
|
a8f7c79934 | ||
|
|
c42bf14c9a | ||
|
|
1e8479b853 | ||
|
|
c139d700f1 | ||
|
|
67b9996911 | ||
|
|
4e9ee567ca | ||
|
|
4863e13596 | ||
|
|
9c4580a522 | ||
|
|
4e3ad00202 | ||
|
|
86ac43567d | ||
|
|
35bf980bda | ||
|
|
51677ae184 | ||
|
|
c9e6fc0e2b | ||
|
|
ab386fc436 | ||
|
|
7e17ae638a | ||
|
|
872d9695c3 | ||
|
|
69e7df6a63 | ||
|
|
98aceef192 | ||
|
|
dfd9cc6fbf | ||
|
|
40e7499e7c | ||
|
|
7f0bd5f609 | ||
|
|
043923fc64 | ||
|
|
f0f8a35b0f | ||
|
|
f5f389b440 | ||
|
|
7cca330438 | ||
|
|
5936de3e56 | ||
|
|
c98ea8b957 | ||
|
|
34c33697bb | ||
|
|
92bd920113 | ||
|
|
8007f82d52 | ||
|
|
60b9766449 | ||
|
|
0ebb7264bc | ||
|
|
ce21c60eba | ||
|
|
53638f831d | ||
|
|
4953da2dd7 | ||
|
|
4ac65f743c | ||
|
|
8f2e9d66c8 | ||
|
|
c7d42fb90a | ||
|
|
003b1e5dca | ||
|
|
b3463705c3 | ||
|
|
52e1f50a80 | ||
|
|
cdc2f10fe8 | ||
|
|
89fbdef10e | ||
|
|
ac795feeed | ||
|
|
3af5ecd07f | ||
|
|
82949613db | ||
|
|
47d788ca13 | ||
|
|
77be606912 | ||
|
|
4eab6a43c8 | ||
|
|
6c24d4fb17 | ||
|
|
e36595a045 | ||
|
|
e07537ff8b | ||
|
|
303eaaa71b | ||
|
|
1e66b5f344 | ||
|
|
d638a2c1f9 | ||
|
|
b8512c6033 | ||
|
|
3e057ea18d | ||
|
|
77113fb629 | ||
|
|
a3a0ab92b7 | ||
|
|
68fb5eb45e | ||
|
|
1c58c78892 | ||
|
|
82124017c0 | ||
|
|
e502f46f43 | ||
|
|
a4f8218b5e | ||
|
|
d162a50bba | ||
|
|
9ca8514a0a | ||
|
|
6fa809a8d2 | ||
|
|
c6d8df2317 | ||
|
|
c28d7323b3 | ||
|
|
25f547ff78 | ||
|
|
50911b4ec1 | ||
|
|
663675f9a0 | ||
|
|
8b28bd24ff | ||
|
|
b5d6f75330 | ||
|
|
f5b5ecaafa | ||
|
|
7799b676ca | ||
|
|
24d3686f60 | ||
|
|
b8b3bba264 | ||
|
|
abfbc04067 | ||
|
|
23ed72826a | ||
|
|
7288058df6 | ||
|
|
6715f9e9af | ||
|
|
ccec1102ef | ||
|
|
360a6fc5dc | ||
|
|
9a806724af | ||
|
|
ed1ddaad5d | ||
|
|
726b3ee544 | ||
|
|
13309b26e0 | ||
|
|
eddc256bed | ||
|
|
52f821cbba | ||
|
|
0dbc2fff13 | ||
|
|
91ec83bc1c | ||
|
|
9fec1dae39 | ||
|
|
8fb12a2176 | ||
|
|
fe0c7067b6 | ||
|
|
d0f0843774 | ||
|
|
81a5b207ac | ||
|
|
e3d5970ae3 | ||
|
|
85f05c26cb | ||
|
|
fcc7b0d611 |
No files matched your search
@@ -0,0 +1,10 @@
|
||||
# actionlint configuration. Passed explicitly from the ci workflow:
|
||||
# actionlint -config-file .gitea/actionlint.yaml .gitea/workflows/*.yaml
|
||||
#
|
||||
# The self-hosted act_runner registers custom labels that actionlint cannot know
|
||||
# about, so declare them here instead of silencing the whole runner-label check.
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
- arch
|
||||
- homelab
|
||||
- prod
|
||||
+408
-69
@@ -7,19 +7,115 @@ on:
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
# Every job here is checkout plus local tools. The token needs to read the tree
|
||||
# and nothing else, and saying so keeps a future step that reaches for the API
|
||||
# from quietly holding a token that can write to the repository.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
|
||||
|
||||
env:
|
||||
REGISTRY: gcr.forust.xyz
|
||||
|
||||
jobs:
|
||||
lint-prettier:
|
||||
lint-compose:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# Structure check for every committed Compose file, active or not.
|
||||
# Interpolation, env-file and bind-mount resolution are all switched off,
|
||||
# because inactive stacks have no .env here and would only fail on their
|
||||
# ${VAR:?} guards. Active stacks get the full check with interpolation in
|
||||
# the deploy workflow, where the real .env files live.
|
||||
- name: Validate Compose files
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
source .gitea/workflows/compose-lint.sh
|
||||
|
||||
mapfile -t safe_flags < <(compose_safe_flags)
|
||||
echo "docker compose config ${safe_flags[*]-}"
|
||||
|
||||
mapfile -t files < <(compose_files)
|
||||
if [ "${#files[@]}" -eq 0 ]; then
|
||||
echo "No Compose files found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
failed=0
|
||||
for f in "${files[@]}"; do
|
||||
if ! out="$(validate_compose_file "$f" ${safe_flags[@]+"${safe_flags[@]}"} 2>&1)"; then
|
||||
failed=1
|
||||
echo "::error file=${f}::$(printf '%s' "$out" | head -1)"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$failed" -ne 0 ]; then
|
||||
echo "Compose validation failed."
|
||||
exit 1
|
||||
fi
|
||||
echo "checked ${#files[@]} Compose file(s)"
|
||||
|
||||
lint-actionlint:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint Gitea Actions workflows with actionlint
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml
|
||||
|
||||
lint-shellcheck:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint shell scripts with ShellCheck
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
# userbot/ is a git subtree synced from forust/userbot, so its shell
|
||||
# scripts are upstream's to maintain, not ours. Linting them would let a
|
||||
# routine subtree pull turn the deploy gate red on code we do not own.
|
||||
mapfile -t scripts < <(
|
||||
git ls-files '*.sh' ':(glob)**/*.bash' ':!userbot/**'
|
||||
)
|
||||
if [ "${#scripts[@]}" -eq 0 ]; then
|
||||
echo "No shell scripts found."
|
||||
exit 0
|
||||
fi
|
||||
shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}"
|
||||
|
||||
lint-prettier:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Check formatting with Prettier
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t prettier_files < <(
|
||||
git ls-files \
|
||||
| grep -E '\.(md|json|ya?ml|html|css)$' \
|
||||
@@ -31,51 +127,65 @@ jobs:
|
||||
exit 0
|
||||
fi
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
node:22-alpine \
|
||||
sh -lc 'npx --yes prettier@3 --check --ignore-unknown "$@"' sh "${prettier_files[@]}"
|
||||
prettier --check --ignore-unknown "${prettier_files[@]}"
|
||||
|
||||
lint-ruff:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint Python with Ruff
|
||||
- name: Lint and format-check Python with Ruff
|
||||
shell: bash
|
||||
run: |
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
ghcr.io/astral-sh/ruff:latest \
|
||||
check .
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
ruff check .
|
||||
ruff format --check .
|
||||
|
||||
lint-yaml:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint YAML syntax
|
||||
shell: bash
|
||||
run: |
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
cytopia/yamllint:latest \
|
||||
-c .yamllint .
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t yaml_files < <(
|
||||
git ls-files '*.yaml' '*.yml' \
|
||||
':!node_modules/**' \
|
||||
':!**/.venv/**'
|
||||
)
|
||||
|
||||
if [ "${#yaml_files[@]}" -eq 0 ]; then
|
||||
echo "No YAML files found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
yamllint -c .yamllint "${yaml_files[@]}"
|
||||
|
||||
lint-dockerfiles:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Lint Dockerfiles
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t dockerfiles < <(
|
||||
git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*'
|
||||
)
|
||||
@@ -85,22 +195,171 @@ jobs:
|
||||
exit 0
|
||||
fi
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
--entrypoint hadolint \
|
||||
hadolint/hadolint:latest-debian \
|
||||
-c .hadolint.yaml "${dockerfiles[@]}"
|
||||
hadolint -c .hadolint.yaml "${dockerfiles[@]}"
|
||||
|
||||
# Known, accepted, and recorded. Each line is a real advisory against a
|
||||
# package we build into the panel image, kept in this workflow rather than in
|
||||
# the package manifest so that a subtree sync from forust/userbot cannot
|
||||
# silently widen the exemption.
|
||||
#
|
||||
# starlette is the reason this job is not simply "fail on everything":
|
||||
# fastapi 0.115.12 pins `starlette<0.47.0`, and the fixes for the last four
|
||||
# below need 0.49.1 through 1.3.1, so clearing them means a jump from fastapi
|
||||
# 0.115.12 to 0.141.x. That is upstream's call, not a drive-by in a lint
|
||||
# commit. Of the seven, four are reachable here in principle: 1942 is a
|
||||
# crafted Range header hitting FileResponse, and the panel serves its built
|
||||
# SPA through exactly that; 249 is request.form() ignoring max_fields for
|
||||
# x-www-form-urlencoded, which is the login form; 1941 is a large multipart
|
||||
# body blocking the event loop; 161 and 248 are unvalidated Host and request
|
||||
# path reaching request.url. 2280 needs HTTPEndpoint, which the panel does
|
||||
# not use, and 2281 is Windows-only, and this deploys on Linux.
|
||||
#
|
||||
# The panel answers on userbot.workstation.internal and has no public
|
||||
# forust.xyz route, which is what keeps the four reachable ones from being
|
||||
# an internet-facing DoS. It still manages Telegram credentials.
|
||||
#
|
||||
# Deleting an entry here is how you accept a new advisory, so the diff says
|
||||
# so out loud.
|
||||
scan-deps:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Audit the Python dependencies that ship in the image
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh pip-audit)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
# requirements.txt, not requirements-dev.txt: this is what the image
|
||||
# installs, and the test tooling is not a shipped attack surface.
|
||||
pip-audit -r userbot/panel/backend/requirements.txt --strict \
|
||||
--ignore-vuln CVE-2025-67720 \
|
||||
--ignore-vuln PYSEC-2026-161 \
|
||||
--ignore-vuln PYSEC-2026-1941 \
|
||||
--ignore-vuln PYSEC-2026-1942 \
|
||||
--ignore-vuln PYSEC-2026-2280 \
|
||||
--ignore-vuln PYSEC-2026-2281 \
|
||||
--ignore-vuln PYSEC-2026-248 \
|
||||
--ignore-vuln PYSEC-2026-249
|
||||
|
||||
# devDependencies are excluded on purpose. `npm audit` on the full tree
|
||||
# reports 7 findings, and every one of them is a build- or test-time
|
||||
# package: the esbuild CORS advisory needs a vite dev server serving to
|
||||
# the internet, and nanoid's infinite loop needs a custom generator
|
||||
# called with size 0, which postcss does not do. None of them are in the
|
||||
# 91 kB bundle the panel serves. The one production finding, devalue
|
||||
# via svelte, is moderate, which is where --audit-level draws the line;
|
||||
# this fails on the next high or critical one.
|
||||
- name: Audit the production npm dependencies
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# The pinned node, not whatever the runner has. Its system node is a
|
||||
# rolling Arch package: during this very push its npm was missing
|
||||
# entirely, and an hour later it was npm 12 on node 26. Both are the
|
||||
# wrong major anyway — the panel image is node:22-alpine.
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
cd userbot/panel/frontend
|
||||
npm ci
|
||||
npm audit --omit=dev --audit-level=high
|
||||
|
||||
test-backend:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# 25 tests over the panel's pydantic models, its auth flow, the SPA
|
||||
# fallback and the Kubernetes client it shells out with. They existed and
|
||||
# had never been executed by anything.
|
||||
#
|
||||
# Note that userbot/ is a subtree synced from forust/userbot, so a routine
|
||||
# sync can turn this red on upstream's code. Unlike the shellcheck job,
|
||||
# which skips that tree because style disagreements there are ours to
|
||||
# lose, a failing test here is a real defect in a service we deploy.
|
||||
- name: Run the panel backend test suite
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh uv)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
# A venv in a temp dir rather than a checked-out one: the runner is
|
||||
# shared, and a leftover .venv would let a dependency the
|
||||
# requirements no longer pin still satisfy an import.
|
||||
#
|
||||
# --python is not optional. uv otherwise takes whatever interpreter it
|
||||
# finds first, and which one that is depends on the machine: this
|
||||
# runner runs jobs on the host, where the only interpreter is 3.14,
|
||||
# and pyrogram's sync.py calls the bare asyncio.get_event_loop() that
|
||||
# 3.14 no longer auto-creates, so three tests fail at collection. The
|
||||
# image is python:3.13-slim, so 3.13 is also the version worth
|
||||
# testing: uv fetches a managed build of it when the host has none,
|
||||
# which is what makes this job independent of the runner.
|
||||
venv="$(mktemp -d)/venv"
|
||||
uv venv --python 3.13 --quiet "$venv"
|
||||
uv pip install --quiet --python "$venv/bin/python" \
|
||||
-r userbot/panel/backend/requirements-dev.txt
|
||||
|
||||
# `python -m`, not bare `pytest`: the tests import `app.*` relative to
|
||||
# the backend directory, which only works if the cwd is on sys.path,
|
||||
# and only `python -m` puts it there.
|
||||
cd userbot/panel/backend
|
||||
"$venv/bin/python" -m pytest tests/ -q
|
||||
|
||||
test-frontend:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# One `npm ci` for both checks below: it is by far the slowest part of
|
||||
# this job, and a second one would learn nothing the first did not.
|
||||
#
|
||||
# `npm ci`, not `npm install`, for the same reason the Dockerfile uses it:
|
||||
# the lockfile is what makes the tree that gets checked the tree that
|
||||
# gets shipped.
|
||||
- name: Type-check and test the panel frontend
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# The pinned node, not whatever the runner has. Its system node is a
|
||||
# rolling Arch package: during this very push its npm was missing
|
||||
# entirely, and an hour later it was npm 12 on node 26. Both are the
|
||||
# wrong major anyway — the panel image is node:22-alpine.
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
cd userbot/panel/frontend
|
||||
npm ci
|
||||
|
||||
# svelte-check has been a devDependency all along with no script
|
||||
# pointing at it, so the type errors it reports had nowhere to
|
||||
# surface. It is clean today, which is the only reason it can be a
|
||||
# gate: it stops at whatever upstream introduces rather than
|
||||
# reporting a backlog we inherited.
|
||||
npm run check
|
||||
npm test
|
||||
|
||||
validate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Validate Kubernetes manifests
|
||||
- name: Validate Kubernetes manifests against JSON schemas
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
|
||||
mapfile -t manifests < <(
|
||||
git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \
|
||||
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
|
||||
@@ -111,24 +370,109 @@ jobs:
|
||||
exit 0
|
||||
fi
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
ghcr.io/yannh/kubeconform:latest \
|
||||
kubeconform \
|
||||
-strict \
|
||||
-ignore-missing-schemas \
|
||||
-summary \
|
||||
"${manifests[@]}"
|
||||
|
||||
# kubeconform has no schemas for CRDs, so every IngressRoute, Certificate,
|
||||
# PrometheusRule, Middleware, ServersTransport and ServiceMonitor is silently
|
||||
# skipped above. The live API server knows the real CRD schemas (and runs the
|
||||
# cert-manager / Traefik admission webhooks), so validate there too.
|
||||
#
|
||||
# Only services marked with a k8s/active marker are checked: server-side
|
||||
# dry-run needs the target namespace to exist, and inactive services are not
|
||||
# deployed. Services being enabled for the first time are still covered by
|
||||
# the JSON-schema pass above.
|
||||
#
|
||||
# Main pushes only. `--dry-run=server` persists nothing, but it does execute
|
||||
# the admission webhooks of the production API server, so anyone able to open
|
||||
# a pull request would be able to run arbitrary manifest content through
|
||||
# cert-manager and Traefik. A pull request has nothing to gain from it either:
|
||||
# only main is ever deployed, and this job runs to completion before the
|
||||
# deploy workflow is allowed to start, so a bad CRD is still caught before
|
||||
# anything reaches the cluster -- just on the push rather than on the PR.
|
||||
- name: Note the server-side check is not running here
|
||||
if: github.event_name == 'pull_request' || github.ref != 'refs/heads/main'
|
||||
shell: bash
|
||||
run: |
|
||||
echo "::notice::Skipping the server-side dry-run. It executes the cert-manager and" \
|
||||
"Traefik admission webhooks against the production API server, so it is limited" \
|
||||
"to pushes to main. CRDs are still schema-checked by kubeconform above, and the" \
|
||||
"server-side pass still runs on main before the deploy."
|
||||
|
||||
- name: Validate active manifests against the live API server
|
||||
if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main'
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
if ! kubectl get --raw='/readyz' --request-timeout=10s >/dev/null 2>&1; then
|
||||
echo "::warning::Cluster unreachable — skipped server-side validation of CRDs (IngressRoute, Certificate, PrometheusRule). Review manifest changes manually."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
mapfile -t k8s_dirs < <(
|
||||
git ls-files '*.yaml' '*.yml' \
|
||||
| grep -E '(^|/)k8s/' \
|
||||
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
|
||||
| sort -u
|
||||
)
|
||||
|
||||
manifests=()
|
||||
kustomize_apps=()
|
||||
for dir in "${k8s_dirs[@]}"; do
|
||||
if [ ! -f "${dir}/active" ]; then
|
||||
echo "skip (no k8s/active): ${dir}"
|
||||
continue
|
||||
fi
|
||||
if [ -f "${dir}/overlays/prod/kustomization.yaml" ]; then
|
||||
kustomize_apps+=("${dir}/overlays/prod")
|
||||
elif [ -f "${dir}/base/kustomization.yaml" ]; then
|
||||
kustomize_apps+=("${dir}/base")
|
||||
else
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] && manifests+=("$f")
|
||||
done < <(
|
||||
git ls-files "${dir}/*.yaml" "${dir}/*.yml" \
|
||||
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
|
||||
)
|
||||
fi
|
||||
done
|
||||
|
||||
echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps"
|
||||
failed=0
|
||||
for m in ${manifests[@]+"${manifests[@]}"}; do
|
||||
if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then
|
||||
failed=1
|
||||
echo "::error file=${m}::$(printf '%s' "$out" | head -1)"
|
||||
fi
|
||||
done
|
||||
for k in ${kustomize_apps[@]+"${kustomize_apps[@]}"}; do
|
||||
if ! out="$(kubectl apply -k "$k" --dry-run=server 2>&1)"; then
|
||||
failed=1
|
||||
echo "::error file=${k}::$(printf '%s' "$out" | head -1)"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$failed" -ne 0 ]; then
|
||||
echo "Server-side validation failed. The API server (or an admission webhook) rejected these manifests."
|
||||
exit 1
|
||||
fi
|
||||
echo "server-side dry-run: all active manifests accepted by the API server"
|
||||
|
||||
build:
|
||||
needs: [lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
|
||||
needs:
|
||||
[lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
|
||||
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 60
|
||||
outputs:
|
||||
services: ${{ steps.services.outputs.services }}
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
@@ -206,7 +550,7 @@ jobs:
|
||||
case "$service" in
|
||||
dtek_notif)
|
||||
image="${REGISTRY}/forust/dtek-notif"
|
||||
tags=("latest")
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
@@ -219,14 +563,17 @@ jobs:
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" dtek_notif
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" dtek_notif
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
;;
|
||||
errorpages)
|
||||
image="${REGISTRY}/forust/error-pages"
|
||||
tags=("latest")
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
@@ -239,13 +586,16 @@ jobs:
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" errorpages
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" errorpages
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
;;
|
||||
userbot)
|
||||
tags=("latest")
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
@@ -269,15 +619,18 @@ jobs:
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" "$context"
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" "$context"
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
homepages)
|
||||
for service in forust xdfnx; do
|
||||
case "$service" in
|
||||
for variant in forust xdfnx; do
|
||||
case "$variant" in
|
||||
forust)
|
||||
image="${REGISTRY}/forust/forust-homepage"
|
||||
;;
|
||||
@@ -285,7 +638,7 @@ jobs:
|
||||
image="${REGISTRY}/forust/xdfnx-homepage"
|
||||
;;
|
||||
esac
|
||||
tags=("latest")
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
@@ -298,15 +651,18 @@ jobs:
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" -f "homepages/Dockerfile.${variant}" homepages
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
edu_master)
|
||||
for service in session-keeper webinar-checker; do
|
||||
case "$service" in
|
||||
for variant in session-keeper webinar-checker; do
|
||||
case "$variant" in
|
||||
session-keeper)
|
||||
context="edu_master/phpsessid-bot"
|
||||
image="${REGISTRY}/forust/session-keeper"
|
||||
@@ -316,7 +672,7 @@ jobs:
|
||||
image="${REGISTRY}/forust/webinar-checker"
|
||||
;;
|
||||
esac
|
||||
tags=("latest")
|
||||
tags=()
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
@@ -329,7 +685,10 @@ jobs:
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" "$context"
|
||||
docker build \
|
||||
--cache-from "type=registry,ref=${image}:buildcache" \
|
||||
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
|
||||
"${build_args[@]}" "$context"
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
@@ -337,23 +696,3 @@ jobs:
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
deploy-userbot-panel:
|
||||
needs: build
|
||||
if: github.ref_name == 'main' && contains(needs.build.outputs.services, 'userbot')
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Apply and roll out userbot panel
|
||||
shell: bash
|
||||
run: |
|
||||
kubectl apply -f userbot/k8s/base/panel.yaml
|
||||
kubectl get secret userbot-common-secrets -n default -o json \
|
||||
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
|
||||
| kubectl apply -f -
|
||||
# Keep legacy deployments (forust/anna) in sync with manifests; they have no replicas field, so apply leaves scaling to the user manager only.
|
||||
kubectl apply -f userbot/k8s/base/userbots.yaml
|
||||
kubectl rollout restart deployment/userbot-panel -n userbot
|
||||
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
|
||||
@@ -0,0 +1,46 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared helpers for validating Compose files. Sourced both by steps in
|
||||
# .gitea/workflows/ci.yaml and by deploy-lib.sh on the workstation.
|
||||
#
|
||||
# Two levels of checking, matching how the repo is structured:
|
||||
#
|
||||
# general every committed Compose file, active or not. Pure structure check:
|
||||
# no ${VAR} interpolation, no .env lookup, no bind-mount path
|
||||
# resolution. Disabled stacks deliberately have no .env in the repo
|
||||
# and no values on the CI runner, so a full `config` run would fail on
|
||||
# their `${VAR:?}` guards for reasons that have nothing to do with the
|
||||
# change under review.
|
||||
#
|
||||
# full active stacks only, with interpolation and env-file resolution, so
|
||||
# required variables and referenced files are actually resolved. Needs
|
||||
# the gitignored .env files, so this only runs in the deploy workflow
|
||||
# on the workstation.
|
||||
#
|
||||
# This file is meant to be sourced, not executed.
|
||||
|
||||
# All committed Compose files, including the ones deploy never starts.
|
||||
compose_files() {
|
||||
git ls-files \
|
||||
'*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \
|
||||
'*/docker-compose.yaml' '*/docker-compose.yml'
|
||||
}
|
||||
|
||||
# Prints the flags that turn `docker compose config` into the general check.
|
||||
# Probed rather than hardcoded so an older Compose without --no-env-resolution
|
||||
# still gets the flags it does support.
|
||||
compose_safe_flags() {
|
||||
local help flag
|
||||
help="$(docker compose config --help 2>/dev/null || true)"
|
||||
for flag in --no-interpolate --no-env-resolution --no-path-resolution; do
|
||||
if printf '%s' "$help" | grep -q -- "$flag"; then
|
||||
printf '%s\n' "$flag"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# validate_compose_file <file> [extra docker compose config flags...]
|
||||
validate_compose_file() {
|
||||
local file="$1"
|
||||
shift
|
||||
docker compose -f "$file" config --quiet "$@"
|
||||
}
|
||||
@@ -0,0 +1,937 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared stages for the deploy workflow. Runs on the workstation, invoked as:
|
||||
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
|
||||
# source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||
# run_stage "$STAGE"
|
||||
# EOF
|
||||
set -euo pipefail
|
||||
|
||||
: "${REPO:?REPO must be set}"
|
||||
APPLY_PRUNE="${APPLY_PRUNE:-false}"
|
||||
# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to
|
||||
# the current origin/main.
|
||||
DEPLOY_SHA="${DEPLOY_SHA:-}"
|
||||
# Handoff point between the apply stage (writes) and the verify stage (reads).
|
||||
# Under the deploy user's own XDG state directory rather than /var/backups: the
|
||||
# deploy is unprivileged, /var/backups does not exist on a minimal Arch host, and
|
||||
# creating it would need root — which is why the first real deploy died here with
|
||||
# "is not writable" before touching a single workload. $HOME comes from sshd.
|
||||
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state}/homelab-deploy}"
|
||||
# Per-workload rollout budget and how many workloads to watch at once. The whole
|
||||
# apply job has its own timeout-minutes as a backstop.
|
||||
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
|
||||
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}"
|
||||
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
|
||||
|
||||
log() {
|
||||
echo "== $* =="
|
||||
}
|
||||
|
||||
warn() {
|
||||
echo "WARNING: $*" >&2
|
||||
}
|
||||
|
||||
collect_k8s() {
|
||||
git -C "$REPO" ls-files -- "$1" \
|
||||
| grep -E '\.ya?ml$' \
|
||||
| grep -Ev '/overlays/' \
|
||||
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
|
||||
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
|
||||
| sort
|
||||
}
|
||||
|
||||
kustomize_overlay() {
|
||||
if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
|
||||
echo "$1/overlays/prod"
|
||||
elif [ -f "$1/base/kustomization.yaml" ]; then
|
||||
echo "$1/base"
|
||||
elif [ -f "$1/kustomization.yaml" ]; then
|
||||
echo "$1"
|
||||
fi
|
||||
}
|
||||
|
||||
select_manifests() {
|
||||
K8S_MANIFESTS=()
|
||||
KUSTOMIZE_APPS=()
|
||||
COMPOSE_STACKS=()
|
||||
local kd_rel kd overlay cf_rel cf f
|
||||
while IFS= read -r kd_rel; do
|
||||
kd="$REPO/$kd_rel"
|
||||
if [ ! -f "$kd/active" ]; then
|
||||
echo "skip (no k8s/active): $kd_rel"
|
||||
continue
|
||||
fi
|
||||
overlay="$(kustomize_overlay "$kd" || true)"
|
||||
if [ -n "${overlay:-}" ]; then
|
||||
echo "kustomize app: ${overlay#"$REPO"/}"
|
||||
KUSTOMIZE_APPS+=("$overlay")
|
||||
else
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] && K8S_MANIFESTS+=("$REPO/$f")
|
||||
done < <(collect_k8s "$kd_rel" || true)
|
||||
fi
|
||||
done < <(
|
||||
git -C "$REPO" ls-files '*.yaml' '*.yml' \
|
||||
| grep -E '(^|/)k8s/' \
|
||||
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
|
||||
| sort -u
|
||||
)
|
||||
while IFS= read -r cf_rel; do
|
||||
cf="$REPO/$cf_rel"
|
||||
if [ -f "$(dirname "$cf")/active" ]; then
|
||||
echo "compose: $cf_rel"
|
||||
COMPOSE_STACKS+=("$cf")
|
||||
else
|
||||
echo "skip (no root active): $cf_rel"
|
||||
fi
|
||||
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
|
||||
}
|
||||
|
||||
# --- post-apply verification and rollback -------------------------------------
|
||||
#
|
||||
# A green `kubectl apply` says nothing about the cluster being healthy. These
|
||||
# helpers watch exactly the workloads whose spec changed during this apply, and
|
||||
# on failure roll them back to the revision that was running before, so a bad
|
||||
# push to main cannot leave a service crash-looping.
|
||||
#
|
||||
# Verification lives in its own workflow job, not at the end of the apply stage.
|
||||
# Inside a single process it is worthless exactly when it is needed most: a job
|
||||
# killed by timeout-minutes or cancelled mid-apply never reaches the rollback
|
||||
# code, and leaves a half-applied cluster behind. Split out, the apply job can
|
||||
# die in any way and the verify job still runs.
|
||||
#
|
||||
# That split needs a handoff point on the workstation, because the two stages are
|
||||
# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written
|
||||
# before anything is applied, read by the verify stage afterwards.
|
||||
|
||||
# Creates this run's snapshot directory and publishes it as the handoff point for
|
||||
# the verify stage. Fails hard by design: a deploy that cannot record what it is
|
||||
# about to change must not start, because then nothing can be rolled back for it
|
||||
# automatically. Publishing happens before the first apply, so an apply killed
|
||||
# mid-flight still leaves a usable baseline behind.
|
||||
snapshot_dir() {
|
||||
local stamp dir
|
||||
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
|
||||
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
|
||||
|
||||
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
|
||||
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
|
||||
echo "The verify job needs it to learn which workloads this deploy touches." >&2
|
||||
echo "Refusing to deploy without a way to roll back." >&2
|
||||
return 1
|
||||
fi
|
||||
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
|
||||
echo "ERROR: cannot create snapshot dir $dir" >&2
|
||||
return 1
|
||||
fi
|
||||
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
|
||||
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
printf '%s\n' "$dir"
|
||||
}
|
||||
|
||||
save_snapshot() {
|
||||
local dir="$1"
|
||||
log "Saving pre-apply snapshot to $dir"
|
||||
workload_generations >"$dir/generations.before" 2>/dev/null \
|
||||
|| warn "could not snapshot workload generations"
|
||||
kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \
|
||||
|| warn "could not snapshot workloads"
|
||||
for release in prometheus-stack loki alloy; do
|
||||
if helm status "$release" -n prometheus >/dev/null 2>&1; then
|
||||
{
|
||||
echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)"
|
||||
helm get values "$release" -n prometheus --all 2>/dev/null
|
||||
} >"$dir/helm-$release.txt"
|
||||
fi
|
||||
done
|
||||
# The verify stage compares this against the commit it is deploying, to refuse
|
||||
# rolling back against a baseline left by an earlier run. A snapshot we cannot
|
||||
# attribute to a commit is unusable for that, so fail before anything is applied.
|
||||
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
|
||||
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
|
||||
workload_generations() {
|
||||
kubectl get "$WORKLOAD_KINDS" -A \
|
||||
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
|
||||
--no-headers 2>/dev/null \
|
||||
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
|
||||
}
|
||||
|
||||
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
|
||||
# moved since the snapshot, i.e. the ones this apply actually touched.
|
||||
changed_workloads() {
|
||||
local before="$1"
|
||||
local ns name kind gen old
|
||||
while read -r ns name kind gen; do
|
||||
[ -n "${gen:-}" ] || continue
|
||||
old="$(awk -v want_ns="$ns" -v want_name="$name" \
|
||||
'$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)"
|
||||
if [ "$old" != "$gen" ]; then
|
||||
printf '%s %s %s\n' "$kind" "$ns" "$name"
|
||||
fi
|
||||
done < <(workload_generations)
|
||||
}
|
||||
|
||||
# Prints "<ns> <kind>/<name> <image>" for every workload this repository owns that
|
||||
# runs an image from our own registry.
|
||||
#
|
||||
# The repository is the scope, deliberately. The cluster also holds workloads on
|
||||
# our registry that no manifest here declares (they are applied out of band), and
|
||||
# those are somebody else's to deploy. Walking the manifests rather than the
|
||||
# cluster means those can never be restarted by this pipeline, now or later.
|
||||
owned_registry_workloads() {
|
||||
local kd_rel f
|
||||
while IFS= read -r kd_rel; do
|
||||
[ -f "$REPO/$kd_rel/active" ] || continue
|
||||
while IFS= read -r f; do
|
||||
[ -n "$f" ] || continue
|
||||
# A file that does not mention the registry cannot declare a workload on it,
|
||||
# and parsing costs ~2.5s per file against a millisecond for the grep. The
|
||||
# filter keeps this at a handful of parses instead of one per manifest.
|
||||
grep -q 'gcr\.forust\.xyz/forust/' "$REPO/$f" 2>/dev/null || continue
|
||||
# kubectl prints a bare object for a single-document file and a List for a
|
||||
# multi-document one, so normalise both shapes before filtering.
|
||||
kubectl apply --dry-run=client -f "$REPO/$f" -o json 2>/dev/null \
|
||||
| jq -r '
|
||||
(if .items then .items[] else . end)
|
||||
| select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
|
||||
| select(any((.spec.template.spec.containers // [])[]?;
|
||||
(.image // "") | test("^gcr\\.forust\\.xyz/forust/")))
|
||||
| (.metadata.namespace // "default") as $ns
|
||||
| ([.spec.template.spec.containers[].image
|
||||
| select(test("^gcr\\.forust\\.xyz/forust/"))][0]) as $img
|
||||
| "\($ns) \(.kind | ascii_downcase)/\(.metadata.name) \($img)"
|
||||
' 2>/dev/null || true
|
||||
done < <(collect_k8s "$kd_rel" || true)
|
||||
done < <(
|
||||
git -C "$REPO" ls-files '*.yaml' '*.yml' \
|
||||
| grep -E '(^|/)k8s/' \
|
||||
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
|
||||
| sort -u
|
||||
)
|
||||
}
|
||||
|
||||
# Prints the digest an image tag resolves to for this cluster's architecture, or
|
||||
# nothing when it cannot be resolved.
|
||||
#
|
||||
# Only the manifest entry matching the node architecture counts. A multi-arch tag
|
||||
# also carries `unknown/unknown` entries for the build attestation, and a pod's
|
||||
# imageID is always the per-platform digest, so comparing the wrong entry would
|
||||
# mark every workload stale forever and restart the whole cluster on every deploy.
|
||||
registry_digest() {
|
||||
local arch
|
||||
arch="$(kubectl get nodes -o jsonpath='{.items[0].status.nodeInfo.architecture}' 2>/dev/null || true)"
|
||||
[ -n "$arch" ] || arch=amd64
|
||||
# The || true is load-bearing. Every caller runs under set -euo pipefail, and
|
||||
# pipefail reports the rightmost non-zero stage, so a ref the registry does not
|
||||
# have would abort the caller at the assignment instead of yielding an empty
|
||||
# string. The callers check for empty themselves and report it by name.
|
||||
docker manifest inspect "$1" 2>/dev/null \
|
||||
| jq -r --arg arch "$arch" '
|
||||
.manifests[]?
|
||||
| select(.platform.os == "linux" and .platform.architecture == $arch)
|
||||
| .digest
|
||||
' 2>/dev/null \
|
||||
| head -1 || true
|
||||
}
|
||||
|
||||
# Rewrites our own images to immutable digests on the way into the cluster.
|
||||
# Reads a manifest stream on stdin, writes the pinned stream to stdout.
|
||||
#
|
||||
# A digest is not knowable when a manifest is written, so it is resolved here, at
|
||||
# apply time, and never committed: git keeps a readable `:prod` tag. That is what
|
||||
# makes rollback mean something. `kubectl rollout undo` restores the previous
|
||||
# ReplicaSet's pod template verbatim, and a template naming a digest restores the
|
||||
# exact bytes that were serving before. A template naming a moving tag does not —
|
||||
# the tag has already moved by the time the rollback runs, so the "rollback"
|
||||
# re-pulls the very image that just failed and the cluster stays broken.
|
||||
#
|
||||
# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a
|
||||
# reference that is not `:latest` defaults to IfNotPresent, which is what the
|
||||
# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot
|
||||
# change, so pulling again buys nothing.
|
||||
#
|
||||
# An image that cannot be resolved is fatal. Carrying on would quietly apply a
|
||||
# mutable tag again, which is the exact failure this function exists to remove.
|
||||
render_pinned() {
|
||||
local src refs map ref digest missing=0
|
||||
src="$(mktemp)"
|
||||
refs="$(mktemp)"
|
||||
map="$(mktemp)"
|
||||
|
||||
cat >"$src"
|
||||
grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true
|
||||
|
||||
while read -r ref; do
|
||||
[ -n "$ref" ] || continue
|
||||
digest="$(registry_digest "$ref")"
|
||||
if [ -z "$digest" ]; then
|
||||
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
|
||||
echo " The build job has to push that tag before the deploy resolves it." >&2
|
||||
missing=$((missing + 1))
|
||||
continue
|
||||
fi
|
||||
printf '%s\t%s\n' "$ref" "$digest" >>"$map"
|
||||
done <"$refs"
|
||||
if [ "$missing" -gt 0 ]; then
|
||||
rm -f "$src" "$refs" "$map"
|
||||
return 1
|
||||
fi
|
||||
|
||||
awk -v mapfile="$map" '
|
||||
BEGIN {
|
||||
while ((getline line < mapfile) > 0) {
|
||||
i = index(line, "\t")
|
||||
d[substr(line, 1, i - 1)] = substr(line, i + 1)
|
||||
}
|
||||
}
|
||||
{
|
||||
if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) {
|
||||
name = $0
|
||||
sub(/^[[:space:]]*image:[[:space:]]*/, "", name)
|
||||
sub(/[[:space:]]*$/, "", name)
|
||||
if (name in d) {
|
||||
pad = $0
|
||||
sub(/image:.*/, "", pad)
|
||||
# Drop the tag: the canonical form used in the docs is repo@sha256:...,
|
||||
# and leaving :prod next to the digest reads like it still matters.
|
||||
repo = name
|
||||
sub(/:[A-Za-z0-9._-]+$/, "", repo)
|
||||
print pad "image: " repo "@" d[name]
|
||||
next
|
||||
}
|
||||
}
|
||||
print
|
||||
}
|
||||
' "$src"
|
||||
rm -f "$src" "$refs" "$map"
|
||||
}
|
||||
|
||||
# Restarts every owned workload whose running image is not the one its tag
|
||||
# resolves to now.
|
||||
#
|
||||
# This used to be how a rebuild reached the cluster at all: the manifests pinned
|
||||
# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply`
|
||||
# decided there was nothing to do, and the cluster served the previous build
|
||||
# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves
|
||||
# the pod template and rolls out on its own.
|
||||
#
|
||||
# What is left is the drift check: a hand-run `kubectl set image`, or anything
|
||||
# else that edits a live workload behind the deploy's back, is the only way to end
|
||||
# up serving a digest the tag has moved past. It stays idempotent, so a redeploy
|
||||
# that changed no image still does not bounce healthy services.
|
||||
#
|
||||
# The container is matched on its repository rather than on the exact reference:
|
||||
# once render_pinned has run, a pod's status reports `repo@sha256:...` while this
|
||||
# still reads the repository's `:prod` tag out of the manifest.
|
||||
restart_stale_images() {
|
||||
local ns target image want selector running entry one
|
||||
local unchecked=0
|
||||
local -A digests=()
|
||||
local -a stale=()
|
||||
while read -r ns target image; do
|
||||
[ -n "${target:-}" ] || continue
|
||||
if [ -z "${digests[$image]:-}" ]; then
|
||||
digests[$image]="$(registry_digest "$image")"
|
||||
fi
|
||||
want="${digests[$image]}"
|
||||
if [ -z "$want" ]; then
|
||||
warn "cannot resolve ${image##*/} in the registry, leaving $target alone"
|
||||
unchecked=$((unchecked + 1))
|
||||
continue
|
||||
fi
|
||||
selector="$(kubectl get "$target" -n "$ns" -o jsonpath='{.spec.selector.matchLabels}' 2>/dev/null \
|
||||
| jq -r 'to_entries | map("\(.key)=\(.value)") | join(",")' 2>/dev/null)"
|
||||
if [ -z "$selector" ]; then
|
||||
warn "cannot read the pod selector of $target, skipping"
|
||||
unchecked=$((unchecked + 1))
|
||||
continue
|
||||
fi
|
||||
running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \
|
||||
| jq -r --arg repo "${image%%:*}" '
|
||||
.items[] | .status.containerStatuses[]?
|
||||
| select(.image == $repo
|
||||
or (.image | startswith($repo + ":"))
|
||||
or (.image | startswith($repo + "@")))
|
||||
| .imageID
|
||||
' 2>/dev/null)"
|
||||
if [ -z "$running" ]; then
|
||||
# Scaled to zero. Nothing is serving stale code, and imagePullPolicy
|
||||
# resolves the tag when it is scaled back up.
|
||||
continue
|
||||
fi
|
||||
entry=""
|
||||
while IFS= read -r one; do
|
||||
[ -n "$one" ] || continue
|
||||
entry="${one##*@}"
|
||||
if [ "$entry" != "$want" ]; then
|
||||
stale+=("$ns $target")
|
||||
break
|
||||
fi
|
||||
done <<<"$running"
|
||||
done < <(owned_registry_workloads)
|
||||
if [ "${#stale[@]}" -eq 0 ]; then
|
||||
if [ "$unchecked" -gt 0 ]; then
|
||||
# Say so plainly. Reporting "everything is current" after checking nothing
|
||||
# would tell the operator the deploy is fine when it may not be.
|
||||
warn "No workload needed a restart, but $unchecked could not be checked"
|
||||
else
|
||||
log "All owned workloads already run the image their tag points at"
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
log "Restarting ${#stale[@]} workload(s) running an image their tag has moved past"
|
||||
for ref in "${stale[@]}"; do
|
||||
log " $ref"
|
||||
done
|
||||
local failed=()
|
||||
for ref in "${stale[@]}"; do
|
||||
ns="${ref%% *}"
|
||||
target="${ref#* }"
|
||||
if ! kubectl rollout restart "$target" -n "$ns" >/dev/null 2>&1; then
|
||||
failed+=("$ref")
|
||||
fi
|
||||
done
|
||||
if [ "${#failed[@]}" -gt 0 ]; then
|
||||
warn "could not restart: ${failed[*]}"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# verify_workloads <failed-file> <kind> <ns> <name> ...
|
||||
# Watches every workload in parallel and records the ones that never became
|
||||
# healthy. Returns non-zero if any of them failed.
|
||||
verify_workloads() {
|
||||
local failed_file="$1"
|
||||
shift
|
||||
[ "$#" -gt 0 ] || return 0
|
||||
: >"$failed_file"
|
||||
local running=0 pid kind ns name
|
||||
local -a pids=()
|
||||
for entry in "$@"; do
|
||||
read -r kind ns name <<<"$entry"
|
||||
(
|
||||
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
||||
echo " ok: ${kind}/${ns}/${name}"
|
||||
else
|
||||
echo " FAILED: ${kind}/${ns}/${name}"
|
||||
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
|
||||
fi
|
||||
) &
|
||||
pids+=($!)
|
||||
running=$((running + 1))
|
||||
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
|
||||
wait -n 2>/dev/null || true
|
||||
running=$((running - 1))
|
||||
fi
|
||||
done
|
||||
for pid in ${pids[@]+"${pids[@]}"}; do
|
||||
wait "$pid" || true
|
||||
done
|
||||
# Non-zero when the file holds at least one failure, i.e. a workload never
|
||||
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
|
||||
# every rollback this stage is meant to perform.
|
||||
[ ! -s "$failed_file" ]
|
||||
}
|
||||
|
||||
# rollback_workloads <failed-file>
|
||||
# Restores the previous revision of every failed workload and waits for it to
|
||||
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
|
||||
# so the operator knows manual recovery is required.
|
||||
rollback_workloads() {
|
||||
local failed_file="$1"
|
||||
local kind ns name unrecovered=()
|
||||
local -a recovered=()
|
||||
while read -r kind ns name; do
|
||||
[ -n "${kind:-}" ] || continue
|
||||
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
|
||||
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
|
||||
echo " rolled back: ${kind}/${ns}/${name}"
|
||||
recovered+=("${kind}/${ns}/${name}")
|
||||
else
|
||||
echo " NOT RECOVERED: ${kind}/${ns}/${name}"
|
||||
unrecovered+=("${kind}/${ns}/${name}")
|
||||
fi
|
||||
done <"$failed_file"
|
||||
echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file"
|
||||
echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file"
|
||||
[ "${#unrecovered[@]}" -eq 0 ]
|
||||
}
|
||||
|
||||
# Helm releases owned by this stage, one line each:
|
||||
#
|
||||
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
|
||||
#
|
||||
# The chart version is the field Renovate keeps current. The helmv3 manager only
|
||||
# understands Chart.yaml and the helm-values manager only values files, so a pin
|
||||
# written straight into a `helm upgrade` command would never be updated: these
|
||||
# have to be declared as custom.regex managers in renovate/renovate.json.
|
||||
HELM_RELEASES=(
|
||||
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
|
||||
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
|
||||
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
|
||||
"reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active"
|
||||
)
|
||||
|
||||
# "name url" for the Helm repository hosting a chart, empty if unknown.
|
||||
helm_repo_for() {
|
||||
case "$1" in
|
||||
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
|
||||
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
|
||||
stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
upgrade_helm_releases() {
|
||||
local entry release chart namespace version values marker repo
|
||||
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
|
||||
IFS='|' read -r release chart namespace version values marker <<<"$entry"
|
||||
if [ ! -f "$REPO/$marker" ]; then
|
||||
echo "skip (no $marker): $release"
|
||||
continue
|
||||
fi
|
||||
if [ ! -f "$REPO/$values" ]; then
|
||||
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
|
||||
return 1
|
||||
fi
|
||||
repo="$(helm_repo_for "$chart")"
|
||||
if [ -z "$repo" ]; then
|
||||
echo "ERROR: no Helm repository configured for chart $chart"
|
||||
return 1
|
||||
fi
|
||||
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
|
||||
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
|
||||
log "Upgrading $release ($chart $version)"
|
||||
# --atomic rolls the release back when the upgrade times out or the workloads
|
||||
# it touches never become ready, so a bad chart bump is not left half applied.
|
||||
helm upgrade --install "$release" "$chart" \
|
||||
--namespace "$namespace" \
|
||||
--version "$version" \
|
||||
--values "$REPO/$values" \
|
||||
--atomic --cleanup-on-fail --timeout 10m
|
||||
done
|
||||
}
|
||||
|
||||
stage_preflight() {
|
||||
if [ ! -d "$REPO/.git" ]; then
|
||||
echo "Repository not found at $REPO"
|
||||
exit 1
|
||||
fi
|
||||
if [ -n "$DEPLOY_SHA" ]; then
|
||||
log "Checking out the commit CI validated ($DEPLOY_SHA)"
|
||||
git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \
|
||||
|| git -C "$REPO" fetch origin main
|
||||
else
|
||||
git -C "$REPO" fetch origin main
|
||||
fi
|
||||
target="${DEPLOY_SHA:-origin/main}"
|
||||
log "Workstation state"
|
||||
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
|
||||
echo " target: $(git -C "$REPO" rev-parse --short "$target")"
|
||||
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
|
||||
echo "ERROR: workstation has local tracked modifications, refusing reset:"
|
||||
git -C "$REPO" status --porcelain --untracked-files=no
|
||||
git -C "$REPO" diff --stat
|
||||
echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy."
|
||||
exit 1
|
||||
fi
|
||||
git -C "$REPO" reset --hard "$target"
|
||||
}
|
||||
|
||||
stage_validate() {
|
||||
cd "$REPO"
|
||||
select_manifests
|
||||
local m k cf
|
||||
# Compose .env files and secret files are gitignored by design, so the
|
||||
# workstation never has real values for the inactive stacks. This stage only
|
||||
# runs the full check on active stacks; the general structure check for every
|
||||
# committed Compose file (active or not) lives in the ci workflow, which has no
|
||||
# .env at all.
|
||||
#
|
||||
# Active stacks are still validated with interpolation and env-file resolution
|
||||
# off, so required-variable guards (:?) and missing local files do not fail the
|
||||
# deploy. Normalization and consistency checks stay enabled.
|
||||
# shellcheck source=compose-lint.sh
|
||||
source "$REPO/.gitea/workflows/compose-lint.sh"
|
||||
local compose_validate_flags=()
|
||||
mapfile -t compose_validate_flags < <(compose_safe_flags)
|
||||
log "Validate compose stacks"
|
||||
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
||||
echo " config: $cf"
|
||||
validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"}
|
||||
done
|
||||
log "Validate k8s manifests (kubectl dry-run=client)"
|
||||
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
||||
kubectl apply --dry-run=client -f "$m" >/dev/null
|
||||
done
|
||||
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
||||
kubectl apply -k "$k" --dry-run=client >/dev/null
|
||||
done
|
||||
log "Validate k8s manifests (kubectl dry-run=server)"
|
||||
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
||||
kubectl apply --dry-run=server -f "$m" >/dev/null
|
||||
done
|
||||
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
||||
kubectl apply -k "$k" --dry-run=server >/dev/null
|
||||
done
|
||||
log "Checking referenced Secrets exist"
|
||||
echo " (deploy never applies *secret*.yaml; create missing ones manually)"
|
||||
local ref_secrets=() missing_secrets=() all_secrets s
|
||||
if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then
|
||||
while IFS= read -r s; do
|
||||
[ -n "$s" ] && ref_secrets+=("$s")
|
||||
done < <(
|
||||
{
|
||||
grep -h -A1 -E 'secretRef:|secretKeyRef:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
|
||||
grep -h -E 'secretName:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
|
||||
} | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true
|
||||
)
|
||||
fi
|
||||
all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)"
|
||||
for s in ${ref_secrets[@]+"${ref_secrets[@]}"}; do
|
||||
if printf '%s\n' "$all_secrets" | grep -qx "$s"; then
|
||||
echo " ok: $s"
|
||||
else
|
||||
echo " MISSING: $s"
|
||||
missing_secrets+=("$s")
|
||||
fi
|
||||
done
|
||||
if [ "${#missing_secrets[@]}" -gt 0 ]; then
|
||||
echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:"
|
||||
printf ' - %s\n' "${missing_secrets[@]}"
|
||||
echo "Create them manually from the laptop, e.g.:"
|
||||
echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
stage_apply_k8s() {
|
||||
cd "$REPO"
|
||||
select_manifests >/dev/null
|
||||
local ns_files=() other_files=() m k prune_opts=()
|
||||
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
||||
case "$m" in
|
||||
*/namespace.y?ml) ns_files+=("$m") ;;
|
||||
*) other_files+=("$m") ;;
|
||||
esac
|
||||
done
|
||||
if [ "$APPLY_PRUNE" = "true" ]; then
|
||||
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
|
||||
fi
|
||||
|
||||
# Record what is about to change, and publish it for the verify job, before
|
||||
# the first apply. Both are fatal on failure: see snapshot_dir.
|
||||
local snapshot
|
||||
snapshot="$(snapshot_dir)" || return 1
|
||||
save_snapshot "$snapshot" || return 1
|
||||
|
||||
if [ "${#ns_files[@]}" -gt 0 ]; then
|
||||
log "Applying namespaces (${#ns_files[@]} files)"
|
||||
for m in "${ns_files[@]}"; do
|
||||
kubectl apply -f "$m"
|
||||
done
|
||||
fi
|
||||
if [ -f "$REPO/prometheus-stack/k8s/active" ]; then
|
||||
if [ ! -f "$REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then
|
||||
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
upgrade_helm_releases
|
||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
|
||||
for m in "${other_files[@]}"; do
|
||||
if ! render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -; then
|
||||
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
fi
|
||||
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
||||
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
|
||||
if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then
|
||||
echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2
|
||||
exit 1
|
||||
fi
|
||||
done
|
||||
if [ -f "$REPO/userbot/k8s/active" ]; then
|
||||
log "userbot panel hook"
|
||||
if kubectl get secret userbot-common-secrets -n userbot >/dev/null 2>&1; then
|
||||
echo " userbot-common-secrets already present in userbot ns, not touching"
|
||||
elif kubectl get secret userbot-common-secrets -n default >/dev/null 2>&1; then
|
||||
echo " bootstrapping userbot-common-secrets into userbot ns"
|
||||
kubectl get secret userbot-common-secrets -n default -o json \
|
||||
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
|
||||
| kubectl apply -f -
|
||||
else
|
||||
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
|
||||
fi
|
||||
fi
|
||||
restart_stale_images
|
||||
|
||||
# No verification here on purpose. This stage may be killed at any point by
|
||||
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
|
||||
# connection, and any code below that line would simply not run. stage_verify_k8s
|
||||
# picks the work up from the snapshot instead.
|
||||
log "Applied. Verification and rollback are the verify job's job, not this one's."
|
||||
}
|
||||
|
||||
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
|
||||
# including when they failed, timed out or were cancelled. Reads the baseline the
|
||||
# apply stage published and works out what it changed, watches those workloads,
|
||||
# and rolls back the ones that never became healthy.
|
||||
stage_verify_k8s() {
|
||||
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
|
||||
local snapshot want have generations
|
||||
local -a touched=()
|
||||
|
||||
if [ ! -s "$pointer" ]; then
|
||||
echo "ERROR: no snapshot pointer at $pointer."
|
||||
echo "The apply stage died before publishing any state, so there is no baseline to"
|
||||
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
|
||||
echo "inspect the cluster by hand."
|
||||
return 1
|
||||
fi
|
||||
snapshot="$(head -1 "$pointer")"
|
||||
if [ ! -d "$snapshot" ]; then
|
||||
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Never trust the pointer blindly. If the apply stage was killed before it
|
||||
# published its own snapshot, `current` still points at the previous deploy's
|
||||
# baseline. Verifying against that would watch the wrong workloads and the
|
||||
# rollback would revert the wrong revisions, so refuse instead.
|
||||
want="${DEPLOY_SHA:-}"
|
||||
if [ -z "$want" ]; then
|
||||
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
|
||||
fi
|
||||
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
|
||||
if [ -z "$want" ] || [ "$have" != "$want" ]; then
|
||||
echo "ERROR: refusing to verify or roll back against a stale snapshot."
|
||||
echo " snapshot: $snapshot"
|
||||
echo " snapshot commit: ${have:-<missing>}"
|
||||
echo " deploy commit: ${want:-<unknown>}"
|
||||
return 1
|
||||
fi
|
||||
echo " snapshot: $snapshot (commit ${have:0:12})"
|
||||
|
||||
generations="$snapshot/generations.before"
|
||||
if [ ! -s "$generations" ]; then
|
||||
# Without a baseline we cannot tell which workloads the apply touched, so
|
||||
# fall back to watching everything rather than silently skipping the check.
|
||||
warn "no pre-apply baseline, verifying every workload in the cluster"
|
||||
: >"$generations"
|
||||
fi
|
||||
|
||||
while read -r kind ns name; do
|
||||
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
|
||||
done < <(changed_workloads "$generations")
|
||||
|
||||
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
|
||||
if [ "${#touched[@]}" -eq 0 ]; then
|
||||
echo " nothing to verify"
|
||||
return 0
|
||||
fi
|
||||
printf ' watching: %s\n' "${touched[@]/#/ }"
|
||||
|
||||
local failed_file="$snapshot/failed-workloads"
|
||||
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
|
||||
echo
|
||||
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
|
||||
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
|
||||
echo
|
||||
log "Rolling back to the previous revision"
|
||||
if rollback_workloads "$failed_file"; then
|
||||
echo
|
||||
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
|
||||
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
|
||||
echo "externally created resources from this commit are still in place. Review the failed"
|
||||
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
|
||||
else
|
||||
echo
|
||||
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
|
||||
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
|
||||
echo "Pre-apply snapshot: $snapshot"
|
||||
fi
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# verify_compose_stack <compose-file>
|
||||
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
|
||||
# come back broken with a green pipeline. Require every long-running service to
|
||||
# actually be running.
|
||||
verify_compose_stack() {
|
||||
local cf="$1"
|
||||
local expected running missing=()
|
||||
expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)"
|
||||
running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)"
|
||||
[ -n "$expected" ] || return 0
|
||||
while IFS= read -r svc; do
|
||||
[ -n "$svc" ] || continue
|
||||
# restart:"no" services are allowed to have exited.
|
||||
if ! printf '%s\n' "$running" | grep -qx "$svc" \
|
||||
&& ! docker compose -f "$cf" config 2>/dev/null \
|
||||
| grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then
|
||||
missing+=("$svc")
|
||||
fi
|
||||
done <<<"$expected"
|
||||
if [ "${#missing[@]}" -gt 0 ]; then
|
||||
echo " NOT RUNNING: ${missing[*]}"
|
||||
docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
|
||||
return 1
|
||||
fi
|
||||
echo " all ${#expected} service(s) running"
|
||||
return 0
|
||||
}
|
||||
|
||||
# The public hostname of every active service, one per line.
|
||||
#
|
||||
# Comments are stripped first, and deliberately so: a route that someone
|
||||
# disabled by commenting it out is not a service to probe, and naio and xui are
|
||||
# both still in the tree that way. A `#` only starts a comment when it is at the
|
||||
# start of a line or after whitespace, so `s/#.*//` alone would also cut a
|
||||
# legitimate value in half.
|
||||
#
|
||||
# Only the public names. The *.internal names are the same Traefik and the same
|
||||
# Services, reached by a different label, so probing both would double the run
|
||||
# to learn the same thing. The public name is also the one a user types.
|
||||
smoke_hosts() {
|
||||
local m k
|
||||
# The backticks below are literal. They are Traefik's Host() delimiter, and the
|
||||
# single quotes are precisely what keeps the shell from reading them as a
|
||||
# command substitution, so the warning is the opposite of a real problem.
|
||||
# shellcheck disable=SC2016
|
||||
{
|
||||
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
|
||||
[ -f "$m" ] && cat "$m"
|
||||
done
|
||||
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
|
||||
kubectl kustomize "$k" 2>/dev/null || true
|
||||
done
|
||||
} | sed -E 's/(^|[[:space:]])#.*$//' \
|
||||
| grep -oE 'Host\(`[^`]+`\)' \
|
||||
| sed -E 's/^Host\(`//; s/`\)$//' \
|
||||
| grep -E '(^|\.)forust\.xyz$' \
|
||||
| grep -v '\${' \
|
||||
| sort -u
|
||||
}
|
||||
|
||||
# stage_verify_k8s watches the rollout, which reports that the pods converged.
|
||||
# It cannot tell a converged pod from a serving one: a route pointing at the
|
||||
# wrong port, a Service selector that matches nothing the app listens on, a 500
|
||||
# from the app itself, an OOMKill loop that still counts as Available for long
|
||||
# enough to pass. All of those are green at the rollout level.
|
||||
#
|
||||
# So ask the thing users ask. Any HTTP response proves Traefik matched the
|
||||
# host, the Service resolved to a pod and the pod answered -- a 302 to a login
|
||||
# or a 404 from a path the service does not serve still means the chain is
|
||||
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
|
||||
# the service is not serving, and only those fail the run.
|
||||
stage_smoke() {
|
||||
cd "$REPO"
|
||||
select_manifests >/dev/null
|
||||
local -a hosts=()
|
||||
# Not named failed: an array of that name already exists in restart_stale_images
|
||||
# above, and a scalar shadowing an array is a trap rather than a shadow.
|
||||
local h code rc bad=0
|
||||
while IFS= read -r h; do
|
||||
[ -n "$h" ] && hosts+=("$h")
|
||||
done < <(smoke_hosts)
|
||||
|
||||
if [ "${#hosts[@]}" -eq 0 ]; then
|
||||
# Nothing to probe means the extraction broke, not that the cluster is empty.
|
||||
echo "ERROR: no public hostnames found in active manifests, refusing to report success"
|
||||
return 1
|
||||
fi
|
||||
|
||||
log "Probing ${#hosts[@]} public route(s)"
|
||||
for h in "${hosts[@]}"; do
|
||||
code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo " UNREACHABLE $h (curl exit $rc)"
|
||||
bad=1
|
||||
continue
|
||||
fi
|
||||
# A glob, not a string compare. `case` on the leading digit is the only one
|
||||
# of these that survives a three-digit code, and the obvious expansion to
|
||||
# try first -- ${code%%[0-9]*} -- is empty for every input, so it silently
|
||||
# reports a 500 as healthy.
|
||||
case "$code" in
|
||||
5*)
|
||||
echo " SERVER ERROR $h $code"
|
||||
bad=1
|
||||
;;
|
||||
000)
|
||||
# curl exited 0 and still no status, so nothing on the far end replied.
|
||||
# Not a pass, whatever the transport thought.
|
||||
echo " NO RESPONSE $h"
|
||||
bad=1
|
||||
;;
|
||||
*)
|
||||
echo " ok $h $code"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [ "$bad" -ne 0 ]; then
|
||||
echo "ERROR: at least one active service is not serving over its public route"
|
||||
return 1
|
||||
fi
|
||||
echo "all ${#hosts[@]} route(s) answered"
|
||||
}
|
||||
|
||||
stage_apply_compose() {
|
||||
cd "$REPO"
|
||||
select_manifests >/dev/null
|
||||
local cf
|
||||
log "Redeploying docker compose stacks (${#COMPOSE_STACKS[@]} stacks)"
|
||||
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
||||
echo " compose: $cf"
|
||||
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
|
||||
docker compose -f "$cf" build
|
||||
docker compose -f "$cf" push
|
||||
fi
|
||||
docker compose -f "$cf" up -d --pull always --remove-orphans
|
||||
done
|
||||
|
||||
local -a broken=()
|
||||
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
|
||||
echo " verifying: $cf"
|
||||
if ! verify_compose_stack "$cf"; then
|
||||
broken+=("$cf")
|
||||
fi
|
||||
done
|
||||
if [ "${#broken[@]}" -gt 0 ]; then
|
||||
echo
|
||||
echo "ERROR: ${#broken[@]} compose stack(s) did not come up:"
|
||||
printf ' - %s\n' "${broken[@]}"
|
||||
echo "Compose stacks are not rolled back automatically: their images use mutable"
|
||||
echo "':latest' tags, so there is no previous version to return to. Check the logs"
|
||||
echo "above, then re-run the deploy once the cause is fixed."
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
run_stage() {
|
||||
case "${1:?stage required}" in
|
||||
preflight) stage_preflight ;;
|
||||
validate) stage_validate ;;
|
||||
apply-k8s) stage_apply_k8s ;;
|
||||
verify-k8s) stage_verify_k8s ;;
|
||||
smoke) stage_smoke ;;
|
||||
apply-compose) stage_apply_compose ;;
|
||||
*)
|
||||
echo "ERROR: unknown stage: $1"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
+125
-129
@@ -1,155 +1,151 @@
|
||||
name: deploy
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
# Deploy only what CI already validated. workflow_run is used instead of
|
||||
# workflow_dispatch so a red lint/validate run can never reach the cluster.
|
||||
workflow_run:
|
||||
workflows: [ci]
|
||||
types: [completed]
|
||||
workflow_dispatch:
|
||||
|
||||
# The deploy jobs read the tree, then reach the cluster over SSH with the
|
||||
# deploy key. The Actions token itself is not part of that path, so it gets
|
||||
# read-only contents and no more.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: deploy-main
|
||||
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
|
||||
# takes the verify job down with it, so a superseded deploy would leave the
|
||||
# cluster half-applied and unchecked — the exact failure the verify job exists
|
||||
# to catch. kubectl apply and docker compose up are both idempotent, so letting
|
||||
# the older run finish and then deploying the newer commit costs little.
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
||||
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
|
||||
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
|
||||
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
||||
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
||||
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
||||
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
|
||||
# finished ci run checked. Pin the exact validated commit instead, so a push
|
||||
# landing mid-deploy cannot make the workstation deploy something else. Also
|
||||
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
|
||||
# which falls back to the current origin/main.
|
||||
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
|
||||
jobs:
|
||||
redeploy:
|
||||
preflight:
|
||||
if: >-
|
||||
github.event_name != 'workflow_run' ||
|
||||
(github.event.workflow_run.conclusion == 'success' &&
|
||||
github.event.workflow_run.head_branch == 'main')
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Redeploy workstation
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Fetch and reset workstation
|
||||
shell: bash
|
||||
env:
|
||||
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
||||
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
|
||||
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
|
||||
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
||||
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
||||
# Set APPLY_PRUNE=true to enable kubectl apply --prune. Requires every
|
||||
# manifest to carry label app.kubernetes.io/managed-by=homelab-deploy,
|
||||
# otherwise previously applied resources get deleted on the next run.
|
||||
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh preflight
|
||||
|
||||
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
|
||||
: "${DEPLOY_USER:?missing DEPLOY_USER}"
|
||||
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
|
||||
validate:
|
||||
needs: [preflight]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
deploy_port="${DEPLOY_PORT:-22}"
|
||||
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
||||
|
||||
ssh_key="$RUNNER_TEMP/deploy_key"
|
||||
mkdir -p "$RUNNER_TEMP"
|
||||
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
||||
chmod 600 "$ssh_key"
|
||||
|
||||
ssh_opts=(
|
||||
-i "$ssh_key"
|
||||
-p "$deploy_port"
|
||||
-o BatchMode=yes
|
||||
-o StrictHostKeyChecking=accept-new
|
||||
)
|
||||
|
||||
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
"DEPLOY_PATH=$(printf '%q' \"$deploy_path\") APPLY_PRUNE=$(printf '%q' \"${APPLY_PRUNE:-false}\") bash -se" <<'EOF'
|
||||
- name: Dry-run manifests and check Secrets
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh validate
|
||||
|
||||
repo="${DEPLOY_PATH:-/srv/homelab}"
|
||||
apply-k8s:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# Apply only, no verification, so this is just the work itself: snapshot,
|
||||
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
|
||||
# apply loop. Verification has its own job and its own budget.
|
||||
timeout-minutes: 45
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
if [ ! -d "$repo/.git" ]; then
|
||||
echo "Repository not found at $repo"
|
||||
exit 1
|
||||
fi
|
||||
- name: Apply Kubernetes manifests
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh apply-k8s
|
||||
|
||||
git -C "$repo" fetch origin main
|
||||
git -C "$repo" reset --hard origin/main
|
||||
apply-compose:
|
||||
needs: [validate]
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# Runtime selection: a service is k8s-managed when $SERVICE/k8s/active
|
||||
# exists. Otherwise it is compose-managed, and only k8s/routing/*
|
||||
# manifests (external Services / EndpointSlices / ServersTransport /
|
||||
# Ingresses that route to docker backends) are applied.
|
||||
# migrate: touch SERVICE/k8s/active (+ move routing files up)
|
||||
# rollback: rm SERVICE/k8s/active
|
||||
collect_k8s() {
|
||||
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
|
||||
! -path '*/routing/*' ! -path '*/overlays/*' \
|
||||
! -name 'kustomization.y*ml' ! -name '*.example.y*ml' \
|
||||
! -name '*values.y*ml' ! -name 'patch-*.y*ml' \
|
||||
| sort
|
||||
}
|
||||
- name: Redeploy docker compose stacks
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh apply-compose
|
||||
|
||||
collect_k8s_inactive() {
|
||||
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
|
||||
\( -name 'namespace.y*ml' -o -path '*/routing/*' \) \
|
||||
! -path '*/overlays/*' ! -name '*.example.y*ml' \
|
||||
| sort
|
||||
}
|
||||
# Watches the workloads this deploy changed and rolls back the ones that never
|
||||
# became healthy. Runs even when the apply jobs failed, timed out or were
|
||||
# cancelled — that is the whole point of splitting it out. `always()` is what
|
||||
# lets it start after a failed dependency; the needs on apply-compose are a
|
||||
# barrier, so verification begins only once both applies are done.
|
||||
verify-k8s:
|
||||
needs: [apply-k8s, apply-compose]
|
||||
if: >-
|
||||
always() &&
|
||||
needs.apply-k8s.result != 'skipped' &&
|
||||
needs.apply-compose.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
mapfile -t compose_stacks < <(
|
||||
find "$repo" -type f \( -name 'compose.yaml' -o -name 'compose.yml' \) | sort
|
||||
)
|
||||
- name: Verify workloads and roll back on failure
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh verify-k8s
|
||||
|
||||
mapfile -t k8s_manifests < <(
|
||||
for kd in $(find "$repo" -type d -name k8s ! -path '*/.git/*' | sort); do
|
||||
if [ -f "$kd/active" ]; then
|
||||
collect_k8s "$kd"
|
||||
else
|
||||
collect_k8s_inactive "$kd"
|
||||
fi
|
||||
done
|
||||
)
|
||||
# Asks the public route of every active service whether it is actually
|
||||
# serving, which the rollout check above structurally cannot: a pod can
|
||||
# converge and still be crash-looping, or be listening on a port no Service
|
||||
# points at, or answer 500.
|
||||
#
|
||||
# `always()` for the same reason verify-k8s has it, and it runs after that job
|
||||
# specifically because a rollback is when a route most needs re-checking. The
|
||||
# needs is a barrier, not a filter: whether verify-k8s passed, failed or was
|
||||
# cancelled, the probes are what say whether the cluster is serving, and
|
||||
# suppressing them on a rollback would hide the one run where the answer
|
||||
# matters most.
|
||||
smoke:
|
||||
needs: [verify-k8s]
|
||||
if: always() && needs.verify-k8s.result != 'skipped'
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
echo "== Validate compose stacks =="
|
||||
for cf in "${compose_stacks[@]}"; do
|
||||
dir=$(dirname "$cf")
|
||||
if [ -f "$dir/k8s/active" ]; then
|
||||
echo " skip (k8s-managed): $dir"
|
||||
continue
|
||||
fi
|
||||
echo " config: $cf"
|
||||
docker compose -f "$cf" config --quiet
|
||||
done
|
||||
|
||||
echo "== Validate k8s manifests (kubectl dry-run) =="
|
||||
for m in "${k8s_manifests[@]}"; do
|
||||
echo " apply --dry-run=client $m"
|
||||
kubectl apply --dry-run=client -f "$m" >/dev/null
|
||||
done
|
||||
|
||||
echo "== Applying Kubernetes manifests =="
|
||||
ns_files=()
|
||||
other_files=()
|
||||
for m in "${k8s_manifests[@]}"; do
|
||||
case "$m" in
|
||||
*/namespace.y?ml) ns_files+=("$m") ;;
|
||||
*) other_files+=("$m") ;;
|
||||
esac
|
||||
done
|
||||
|
||||
prune_opts=()
|
||||
if [ "${APPLY_PRUNE:-false}" = "true" ]; then
|
||||
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
|
||||
fi
|
||||
|
||||
if [ "${#ns_files[@]}" -gt 0 ]; then
|
||||
echo " namespaces first: ${ns_files[*]}"
|
||||
kubectl apply -f "${ns_files[@]}"
|
||||
fi
|
||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||
echo " resources: ${other_files[*]}"
|
||||
kubectl apply "${prune_opts[@]}" -f "${other_files[@]}"
|
||||
fi
|
||||
|
||||
echo "== Redeploying docker compose stacks =="
|
||||
for cf in "${compose_stacks[@]}"; do
|
||||
dir=$(dirname "$cf")
|
||||
if [ -f "$dir/k8s/active" ]; then
|
||||
echo " skip (k8s-managed): $dir"
|
||||
continue
|
||||
fi
|
||||
echo " compose: $dir"
|
||||
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
|
||||
docker compose -f "$cf" build
|
||||
docker compose -f "$cf" push
|
||||
fi
|
||||
docker compose -f "$cf" up -d --pull always --remove-orphans
|
||||
done
|
||||
EOF
|
||||
- name: Probe the public route of every active service
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/ssh-run.sh smoke
|
||||
Executable
+233
@@ -0,0 +1,233 @@
|
||||
#!/usr/bin/env bash
|
||||
# Installs the pinned CI tools into "$TOOLS_DIR/bin" and echoes that directory
|
||||
# on stdout, so callers can do:
|
||||
#
|
||||
# export PATH="$(bash .gitea/workflows/install-ci-tools.sh kubeconform shellcheck):$PATH"
|
||||
#
|
||||
# Versions come from tool-versions.env next to this script and are kept fresh by
|
||||
# Renovate. Re-running is cheap: an already-installed tool at the pinned version
|
||||
# is left alone.
|
||||
set -euo pipefail
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=tool-versions.env
|
||||
. "$here/tool-versions.env"
|
||||
|
||||
TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}"
|
||||
BIN_DIR="$TOOLS_DIR/bin"
|
||||
mkdir -p "$BIN_DIR"
|
||||
|
||||
arch="$(uname -m)"
|
||||
# Upstream projects disagree on arch spelling: kubeconform and actionlint use
|
||||
# Go names (amd64/arm64), shellcheck uses uname names (x86_64/aarch64), node
|
||||
# uses neither (x64/arm64), and hadolint mixes the two in a single release
|
||||
# (x86_64 but arm64).
|
||||
case "$arch" in
|
||||
x86_64 | amd64)
|
||||
goarch=amd64
|
||||
sharch=x86_64
|
||||
nodearch=x64
|
||||
hadolintarch=x86_64
|
||||
;;
|
||||
aarch64 | arm64)
|
||||
goarch=arm64
|
||||
sharch=aarch64
|
||||
nodearch=arm64
|
||||
hadolintarch=arm64
|
||||
;;
|
||||
*)
|
||||
echo "install-ci-tools: unsupported architecture: $arch" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
fetch() {
|
||||
# fetch <url> <dest>
|
||||
if command -v curl >/dev/null 2>&1; then
|
||||
curl -sSLf --retry 3 -o "$2" "$1"
|
||||
elif command -v wget >/dev/null 2>&1; then
|
||||
wget -q -O "$2" "$1"
|
||||
else
|
||||
echo "install-ci-tools: neither curl nor wget is available" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
# installed_version <command>
|
||||
# Prints the version of an already-installed tool, or nothing. Each tool spells
|
||||
# its version flag differently, hence the case.
|
||||
installed_version() {
|
||||
local out
|
||||
case "$1" in
|
||||
kubeconform) out="$("$1" -v 2>/dev/null | head -1 || true)" ;;
|
||||
*) out="$("$1" --version 2>/dev/null | head -1 || true)" ;;
|
||||
esac
|
||||
printf '%s' "$out"
|
||||
}
|
||||
|
||||
# at_version <command> <expected>
|
||||
at_version() {
|
||||
case "$(installed_version "$1")" in
|
||||
*"$2"*) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
install_kubeconform() {
|
||||
if at_version kubeconform "v${KUBECONFORM_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
fetch "https://github.com/yannh/kubeconform/releases/download/v${KUBECONFORM_VERSION}/kubeconform-linux-${goarch}.tar.gz" \
|
||||
"$tmp/kubeconform.tar.gz"
|
||||
tar -xzf "$tmp/kubeconform.tar.gz" -C "$tmp" kubeconform
|
||||
install -m 0755 "$tmp/kubeconform" "$BIN_DIR/kubeconform"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
install_shellcheck() {
|
||||
if at_version shellcheck "${SHELLCHECK_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
fetch "https://github.com/koalaman/shellcheck/releases/download/v${SHELLCHECK_VERSION}/shellcheck-v${SHELLCHECK_VERSION}.linux.${sharch}.tar.xz" \
|
||||
"$tmp/shellcheck.tar.xz"
|
||||
tar -xJf "$tmp/shellcheck.tar.xz" -C "$tmp" --strip-components=1 "shellcheck-v${SHELLCHECK_VERSION}/shellcheck"
|
||||
install -m 0755 "$tmp/shellcheck" "$BIN_DIR/shellcheck"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
install_uv() {
|
||||
if at_version uv "${UV_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
# uv release tags carry no leading v, unlike every other tool installed here.
|
||||
fetch "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-${sharch}-unknown-linux-gnu.tar.gz" \
|
||||
"$tmp/uv.tar.gz"
|
||||
tar -xzf "$tmp/uv.tar.gz" -C "$tmp" --strip-components=1 "uv-${sharch}-unknown-linux-gnu/uv"
|
||||
install -m 0755 "$tmp/uv" "$BIN_DIR/uv"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
install_hadolint() {
|
||||
if at_version hadolint "${HADOLINT_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
# A bare binary, no archive: hadolint ships one file per platform.
|
||||
fetch "https://github.com/hadolint/hadolint/releases/download/v${HADOLINT_VERSION}/hadolint-linux-${hadolintarch}" \
|
||||
"$BIN_DIR/hadolint"
|
||||
chmod 0755 "$BIN_DIR/hadolint"
|
||||
}
|
||||
|
||||
# ruff and yamllint both come from PyPI as wheels, which uv unpacks for us.
|
||||
install_uv_tool() {
|
||||
# <package> <pinned version>
|
||||
if at_version "$1" "$2"; then
|
||||
return 0
|
||||
fi
|
||||
install_uv
|
||||
UV_TOOL_BIN_DIR="$BIN_DIR" uv tool install --force "$1==$2" >/dev/null
|
||||
}
|
||||
|
||||
install_ruff() {
|
||||
install_uv_tool ruff "${RUFF_VERSION}"
|
||||
}
|
||||
|
||||
install_yamllint() {
|
||||
install_uv_tool yamllint "${YAMLLINT_VERSION}"
|
||||
}
|
||||
|
||||
install_pip_audit() {
|
||||
install_uv_tool pip-audit "${PIP_AUDIT_VERSION}"
|
||||
}
|
||||
|
||||
install_prettier() {
|
||||
if at_version prettier "${PRETTIER_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
# Not a standalone binary: prettier's entry point requires ../package.json
|
||||
# relative to its own real path, so the package directory has to survive
|
||||
# next to it. Hence a versioned directory plus a relative symlink, rather
|
||||
# than copying the one file out as the other installers do.
|
||||
local dir="$BIN_DIR/prettier-${PRETTIER_VERSION}"
|
||||
if [ ! -f "$dir/package/package.json" ]; then
|
||||
rm -rf "$dir"
|
||||
mkdir -p "$dir"
|
||||
fetch "https://registry.npmjs.org/prettier/-/prettier-${PRETTIER_VERSION}.tgz" "$dir/prettier.tgz"
|
||||
tar -xzf "$dir/prettier.tgz" -C "$dir"
|
||||
rm -f "$dir/prettier.tgz"
|
||||
# npm strips the exec bit from bin/ on the way into the tarball.
|
||||
chmod 0755 "$dir/package/bin/prettier.cjs"
|
||||
fi
|
||||
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
|
||||
ln -sfn "prettier-${PRETTIER_VERSION}/package/bin/prettier.cjs" "$BIN_DIR/prettier"
|
||||
}
|
||||
|
||||
install_node() {
|
||||
# npm gets checked by running it, not by looking it up: what matters is that
|
||||
# it answers, so a stub, a half-removed Arch package or a name that resolves
|
||||
# to something broken all have to read as "not installed". The runner's npm
|
||||
# is a symlink into /usr/lib/node_modules/npm, which is exactly the kind of
|
||||
# thing that disappears between runs.
|
||||
if at_version node "v${NODE_VERSION}" && [ -n "$(installed_version npm)" ]; then
|
||||
return 0
|
||||
fi
|
||||
# Same shape as prettier above: the tarball's bin/npm and bin/npx are links
|
||||
# into lib/node_modules, so the whole tree has to survive next to them.
|
||||
local dir="$BIN_DIR/node-${NODE_VERSION}"
|
||||
if [ ! -x "$dir/bin/node" ]; then
|
||||
rm -rf "$dir"
|
||||
mkdir -p "$dir"
|
||||
fetch "https://nodejs.org/dist/v${NODE_VERSION}/node-v${NODE_VERSION}-linux-${nodearch}.tar.xz" \
|
||||
"$dir/node.tar.xz"
|
||||
tar -xJf "$dir/node.tar.xz" -C "$dir" --strip-components=1 "node-v${NODE_VERSION}-linux-${nodearch}"
|
||||
rm -f "$dir/node.tar.xz"
|
||||
fi
|
||||
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
|
||||
for bin in node npm npx; do
|
||||
ln -sfn "node-${NODE_VERSION}/bin/${bin}" "$BIN_DIR/${bin}"
|
||||
done
|
||||
}
|
||||
|
||||
install_actionlint() {
|
||||
if at_version actionlint "${ACTIONLINT_VERSION}"; then
|
||||
return 0
|
||||
fi
|
||||
local tmp
|
||||
tmp="$(mktemp -d)"
|
||||
fetch "https://github.com/rhysd/actionlint/releases/download/v${ACTIONLINT_VERSION}/actionlint_${ACTIONLINT_VERSION}_linux_${goarch}.tar.gz" \
|
||||
"$tmp/actionlint.tar.gz"
|
||||
tar -xzf "$tmp/actionlint.tar.gz" -C "$tmp" actionlint
|
||||
install -m 0755 "$tmp/actionlint" "$BIN_DIR/actionlint"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
wanted=("$@")
|
||||
if [ "${#wanted[@]}" -eq 0 ]; then
|
||||
wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint)
|
||||
fi
|
||||
|
||||
for tool in "${wanted[@]}"; do
|
||||
case "$tool" in
|
||||
kubeconform) install_kubeconform ;;
|
||||
shellcheck) install_shellcheck ;;
|
||||
actionlint) install_actionlint ;;
|
||||
prettier) install_prettier ;;
|
||||
ruff) install_ruff ;;
|
||||
yamllint) install_yamllint ;;
|
||||
pip-audit) install_pip_audit ;;
|
||||
hadolint) install_hadolint ;;
|
||||
node) install_node ;;
|
||||
uv) install_uv ;;
|
||||
*)
|
||||
echo "install-ci-tools: unknown tool: $tool" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
printf '%s\n' "$BIN_DIR"
|
||||
@@ -0,0 +1,76 @@
|
||||
name: renovate-ci
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
validate-renovate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag,
|
||||
# so the same version that runs in the cluster is the one validated here.
|
||||
- name: Resolve the deployed Renovate image
|
||||
id: image
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
|
||||
renovate/k8s/cronjob.yaml | head -1)"
|
||||
if [ -z "$image" ]; then
|
||||
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
|
||||
exit 1
|
||||
fi
|
||||
echo "using $image"
|
||||
echo "image=$image" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Validate Renovate repository config
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker run --rm \
|
||||
-v "$PWD/renovate:/opt/renovate:ro" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
|
||||
"${{ steps.image.outputs.image }}" \
|
||||
renovate-config-validator /opt/renovate/renovate.json
|
||||
|
||||
# The CronJob cannot read the repository, so renovate/k8s/configmap.yaml
|
||||
# carries an inlined copy of the config. Fail if it no longer matches.
|
||||
- name: Check the generated Renovate ConfigMap
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
./.gitea/workflows/sync-renovate-configmap.sh --check
|
||||
|
||||
- name: Validate Renovate Kubernetes manifests
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
|
||||
export PATH="$tools_dir:$PATH"
|
||||
kubeconform \
|
||||
-strict \
|
||||
-ignore-missing-schemas \
|
||||
-summary \
|
||||
renovate/k8s/namespace.yaml \
|
||||
renovate/k8s/configmap.yaml \
|
||||
renovate/k8s/cronjob.yaml
|
||||
|
||||
- name: Validate Renovate Compose file
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
source .gitea/workflows/compose-lint.sh
|
||||
mapfile -t safe_flags < <(compose_safe_flags)
|
||||
validate_compose_file renovate/renovate-compose.yaml \
|
||||
${safe_flags[@]+"${safe_flags[@]}"}
|
||||
@@ -0,0 +1,92 @@
|
||||
name: renovate-run
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
repositories:
|
||||
description: "Repositories to scan (comma-separated)"
|
||||
required: false
|
||||
default: "forust/homelab"
|
||||
log_level:
|
||||
description: "Renovate log level"
|
||||
required: false
|
||||
default: "info"
|
||||
type: choice
|
||||
options:
|
||||
- info
|
||||
- debug
|
||||
dry_run:
|
||||
description: "Plan only, do not open or update PRs"
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
|
||||
# Renovate writes through its own bot PAT, passed in as RENOVATE_TOKEN, so the
|
||||
# Actions token is only ever used to read the checkout.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: renovate-run
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
run-renovate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag.
|
||||
# Reading it here means this workflow validates and runs the exact version
|
||||
# that is deployed, instead of a copy that silently goes stale.
|
||||
- name: Resolve the deployed Renovate image
|
||||
id: image
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
|
||||
renovate/k8s/cronjob.yaml | head -1)"
|
||||
if [ -z "$image" ]; then
|
||||
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
|
||||
exit 1
|
||||
fi
|
||||
echo "using $image"
|
||||
echo "image=$image" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Validate Renovate config
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker run --rm \
|
||||
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
|
||||
"${{ steps.image.outputs.image }}" \
|
||||
renovate-config-validator
|
||||
|
||||
- name: Run Renovate
|
||||
shell: bash
|
||||
env:
|
||||
RENOVATE_TOKEN: ${{ secrets.RENOVATE_TOKEN }}
|
||||
RENOVATE_GITHUB_COM_TOKEN: ${{ secrets.RENOVATE_GITHUB_COM_TOKEN }}
|
||||
RENOVATE_REPOSITORIES: ${{ inputs.repositories }}
|
||||
RENOVATE_DRY_RUN: ${{ inputs.dry_run && 'full' || '' }}
|
||||
LOG_LEVEL: ${{ inputs.log_level }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
: "${RENOVATE_TOKEN:?missing RENOVATE_TOKEN secret — add a renovate-bot PAT in repo/org Actions secrets}"
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
|
||||
-e RENOVATE_PLATFORM=gitea \
|
||||
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
|
||||
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
|
||||
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
|
||||
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
|
||||
-e RENOVATE_DRY_RUN="${RENOVATE_DRY_RUN:-}" \
|
||||
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
|
||||
-e RENOVATE_BASE_DIR=/tmp/renovate \
|
||||
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
|
||||
"${{ steps.image.outputs.image }}"
|
||||
Executable
+30
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env bash
|
||||
# usage: ssh-run.sh <stage>
|
||||
# Runs one deploy-lib.sh stage on the workstation over SSH.
|
||||
set -euo pipefail
|
||||
|
||||
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
|
||||
: "${DEPLOY_USER:?missing DEPLOY_USER}"
|
||||
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
|
||||
|
||||
deploy_port="${DEPLOY_PORT:-22}"
|
||||
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
||||
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
|
||||
|
||||
# The private key is written to a per-run directory that is removed on exit, so a
|
||||
# failed or cancelled job cannot leave deploy credentials in the runner's temp
|
||||
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
|
||||
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
|
||||
trap 'rm -rf "$key_dir"' EXIT INT TERM
|
||||
|
||||
ssh_key="$key_dir/deploy_key"
|
||||
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
||||
chmod 600 "$ssh_key"
|
||||
|
||||
ssh -i "$ssh_key" -p "$deploy_port" \
|
||||
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
|
||||
"${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-} STAGE=$1 bash -se" <<'EOF'
|
||||
source "$REPO/.gitea/workflows/deploy-lib.sh"
|
||||
run_stage "$STAGE"
|
||||
EOF
|
||||
Executable
+55
@@ -0,0 +1,55 @@
|
||||
#!/usr/bin/env bash
|
||||
# Regenerates renovate/k8s/configmap.yaml from renovate/renovate.json.
|
||||
#
|
||||
# renovate/renovate.json is the single source of truth: the CronJob, the Compose
|
||||
# file and the renovate-run workflow all mount that exact file. A ConfigMap cannot
|
||||
# read a file from the repository, so the same bytes are inlined here as a literal
|
||||
# block. This script keeps the copy honest:
|
||||
#
|
||||
# .gitea/workflows/sync-renovate-configmap.sh # rewrite in place
|
||||
# .gitea/workflows/sync-renovate-configmap.sh --check # fail if out of date
|
||||
#
|
||||
# renovate-ci runs the --check form on every PR and push, so a config change that
|
||||
# forgets to regenerate the ConfigMap cannot be merged.
|
||||
set -euo pipefail
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
repo="$(git -C "$here" rev-parse --show-toplevel)"
|
||||
|
||||
src="$repo/renovate/renovate.json"
|
||||
dst="$repo/renovate/k8s/configmap.yaml"
|
||||
[ -f "$src" ] || {
|
||||
echo "missing $src" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
render() {
|
||||
cat <<'HEADER'
|
||||
# GENERATED FILE - do not edit by hand.
|
||||
# Source: renovate/renovate.json
|
||||
# Regenerate: .gitea/workflows/sync-renovate-configmap.sh
|
||||
# Verify: .gitea/workflows/sync-renovate-configmap.sh --check
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: renovate-config
|
||||
namespace: renovate
|
||||
data:
|
||||
renovate.json: |
|
||||
HEADER
|
||||
sed 's/^/ /' "$src"
|
||||
}
|
||||
|
||||
if [ "${1:-}" = "--check" ]; then
|
||||
if ! diff -u "$dst" <(render) >/dev/null 2>&1; then
|
||||
echo "ERROR: $dst is out of sync with renovate/renovate.json"
|
||||
echo "Run: .gitea/workflows/sync-renovate-configmap.sh"
|
||||
diff -u "$dst" <(render) || true
|
||||
exit 1
|
||||
fi
|
||||
echo "renovate/k8s/configmap.yaml is in sync with renovate/renovate.json"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
render >"$dst"
|
||||
echo "wrote $dst"
|
||||
@@ -0,0 +1,33 @@
|
||||
# Pinned versions of the CI tools installed by install-ci-tools.sh.
|
||||
# Renovate keeps these up to date (see customManagers in renovate/renovate.json).
|
||||
#
|
||||
# Every version here except NODE_VERSION matches what was already installed on
|
||||
# the runner, so pinning them changes what CI does not at all. It changes what
|
||||
# CI does when the runner is rebuilt with something else: today
|
||||
# install-ci-tools.sh finds the pinned version already on PATH and installs
|
||||
# nothing, and a runner that drifts gets the pinned one installed over it.
|
||||
#
|
||||
# The renovate image version is NOT pinned here: renovate/k8s/cronjob.yaml is the
|
||||
# single source of truth and the workflows read the tag from it, so there is
|
||||
# nothing to drift.
|
||||
ACTIONLINT_VERSION="1.7.7"
|
||||
SHELLCHECK_VERSION="0.11.0"
|
||||
KUBECONFORM_VERSION="0.8.0"
|
||||
PRETTIER_VERSION="3.8.1"
|
||||
RUFF_VERSION="0.16.8"
|
||||
YAMLLINT_VERSION="1.38.0"
|
||||
HADOLINT_VERSION="2.14.0"
|
||||
# pip-audit reads the advisory database over the network, so a floating version
|
||||
# would make the same commit report different things on different days. Pin it
|
||||
# like the rest: the advisories themselves are the moving part, not the tool.
|
||||
PIP_AUDIT_VERSION="2.10.1"
|
||||
# uv builds the throwaway venv the pytest job runs in, and unpacks the PyPI
|
||||
# wheels for ruff, yamllint and pip-audit.
|
||||
UV_VERSION="0.12.17"
|
||||
# node runs `npm ci` for the frontend tests and the npm audit, and it is the one
|
||||
# pin here that does NOT come from the runner: the runner's system node is a
|
||||
# rolling Arch package (it was node 26 with no npm at all when this was pinned),
|
||||
# and the panel image is node:22-alpine. Pinned to the image's major on purpose,
|
||||
# so the tree that gets tested is the tree that gets built. Renovate keeps this
|
||||
# in step with the Dockerfile's node: tag via the "node runtime" group.
|
||||
NODE_VERSION="22.23.3"
|
||||
@@ -1,359 +0,0 @@
|
||||
name: ci
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- "**"
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
env:
|
||||
REGISTRY: gcr.forust.xyz
|
||||
|
||||
jobs:
|
||||
lint-prettier:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Check formatting with Prettier
|
||||
shell: bash
|
||||
run: |
|
||||
mapfile -t prettier_files < <(
|
||||
git ls-files \
|
||||
| grep -E '\.(md|json|ya?ml|html|css)$' \
|
||||
| grep -Ev '^(\.docs/|\.zed/|errorpages/html/|homepages/(forust_files|xdfnx_files)/)'
|
||||
)
|
||||
|
||||
if [ "${#prettier_files[@]}" -eq 0 ]; then
|
||||
echo "No Prettier-managed files found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
node:22-alpine \
|
||||
sh -lc 'npx --yes prettier@3 --check --ignore-unknown "$@"' sh "${prettier_files[@]}"
|
||||
|
||||
lint-ruff:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Lint Python with Ruff
|
||||
shell: bash
|
||||
run: |
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
ghcr.io/astral-sh/ruff:latest \
|
||||
check .
|
||||
|
||||
lint-yaml:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Lint YAML syntax
|
||||
shell: bash
|
||||
run: |
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
cytopia/yamllint:latest \
|
||||
-c .yamllint .
|
||||
|
||||
lint-dockerfiles:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Lint Dockerfiles
|
||||
shell: bash
|
||||
run: |
|
||||
mapfile -t dockerfiles < <(
|
||||
git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*'
|
||||
)
|
||||
|
||||
if [ "${#dockerfiles[@]}" -eq 0 ]; then
|
||||
echo "No Dockerfiles found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
--entrypoint hadolint \
|
||||
hadolint/hadolint:latest-debian \
|
||||
-c .hadolint.yaml "${dockerfiles[@]}"
|
||||
|
||||
validate:
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Validate Kubernetes manifests
|
||||
shell: bash
|
||||
run: |
|
||||
mapfile -t manifests < <(
|
||||
git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \
|
||||
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
|
||||
)
|
||||
|
||||
if [ "${#manifests[@]}" -eq 0 ]; then
|
||||
echo "No Kubernetes manifests found."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
docker run --rm \
|
||||
-v "$PWD:/work" \
|
||||
-w /work \
|
||||
ghcr.io/yannh/kubeconform:latest \
|
||||
-strict \
|
||||
-ignore-missing-schemas \
|
||||
-summary \
|
||||
"${manifests[@]}"
|
||||
|
||||
build:
|
||||
needs: [lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
|
||||
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
|
||||
runs-on: [self-hosted, linux, arch, homelab]
|
||||
outputs:
|
||||
services: ${{ steps.services.outputs.services }}
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Detect changed docker-built services
|
||||
id: services
|
||||
shell: bash
|
||||
run: |
|
||||
base="${{ github.event.before }}"
|
||||
if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then
|
||||
base="$(git rev-list --max-parents=0 HEAD)"
|
||||
fi
|
||||
|
||||
mapfile -t changed_files < <(git diff --name-only "$base" "${GITHUB_SHA}")
|
||||
|
||||
services=()
|
||||
|
||||
add_service() {
|
||||
local name="$1"
|
||||
local seen=0
|
||||
for existing in "${services[@]}"; do
|
||||
if [ "$existing" = "$name" ]; then
|
||||
seen=1
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ "$seen" -eq 0 ]; then
|
||||
services+=("$name")
|
||||
fi
|
||||
}
|
||||
|
||||
for file in "${changed_files[@]}"; do
|
||||
case "$file" in
|
||||
dtek_notif/*)
|
||||
add_service dtek_notif
|
||||
;;
|
||||
errorpages/*)
|
||||
add_service errorpages
|
||||
;;
|
||||
userbot/*)
|
||||
add_service userbot
|
||||
;;
|
||||
homepages/*)
|
||||
add_service homepages
|
||||
;;
|
||||
edu_master/phpsessid-bot/*|edu_master/webinar-checker/*|edu_master/compose.yaml)
|
||||
add_service edu_master
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [ "${#services[@]}" -eq 0 ]; then
|
||||
echo "No docker-built services changed."
|
||||
echo "services=" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
printf '%s\n' "${services[@]}" | tee /tmp/services.txt
|
||||
echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Log in to registry
|
||||
if: steps.services.outputs.services != ''
|
||||
shell: bash
|
||||
run: |
|
||||
echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login "${REGISTRY}" \
|
||||
-u "${{ secrets.REGISTRY_USERNAME }}" \
|
||||
--password-stdin
|
||||
|
||||
- name: Build and push changed images
|
||||
if: steps.services.outputs.services != ''
|
||||
shell: bash
|
||||
run: |
|
||||
IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}"
|
||||
|
||||
for service in "${services[@]}"; do
|
||||
case "$service" in
|
||||
dtek_notif)
|
||||
image="${REGISTRY}/forust/dtek-notif"
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" dtek_notif
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
;;
|
||||
errorpages)
|
||||
image="${REGISTRY}/forust/error-pages"
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" errorpages
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
;;
|
||||
userbot)
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
for target in runtime panel; do
|
||||
case "$target" in
|
||||
runtime)
|
||||
context="userbot"
|
||||
image="${REGISTRY}/forust/userbot"
|
||||
;;
|
||||
panel)
|
||||
context="userbot/panel"
|
||||
image="${REGISTRY}/forust/userbot-panel"
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" "$context"
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
homepages)
|
||||
for service in forust xdfnx; do
|
||||
case "$service" in
|
||||
forust)
|
||||
image="${REGISTRY}/forust/forust-homepage"
|
||||
;;
|
||||
xdfnx)
|
||||
image="${REGISTRY}/forust/xdfnx-homepage"
|
||||
;;
|
||||
esac
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
edu_master)
|
||||
for service in session-keeper webinar-checker; do
|
||||
case "$service" in
|
||||
session-keeper)
|
||||
context="edu_master/phpsessid-bot"
|
||||
image="${REGISTRY}/forust/session-keeper"
|
||||
;;
|
||||
webinar-checker)
|
||||
context="edu_master/webinar-checker"
|
||||
image="${REGISTRY}/forust/webinar-checker"
|
||||
;;
|
||||
esac
|
||||
tags=("latest")
|
||||
case "${GITHUB_REF_NAME}" in
|
||||
main)
|
||||
tags+=("main" "prod")
|
||||
;;
|
||||
dev)
|
||||
tags+=("dev")
|
||||
;;
|
||||
esac
|
||||
build_args=()
|
||||
for tag in "${tags[@]}"; do
|
||||
build_args+=(-t "${image}:${tag}")
|
||||
done
|
||||
docker build "${build_args[@]}" "$context"
|
||||
for tag in "${tags[@]}"; do
|
||||
docker push "${image}:${tag}"
|
||||
done
|
||||
done
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
deploy-userbot-panel:
|
||||
needs: build
|
||||
if: github.ref_name == 'main' && contains(needs.build.outputs.services, 'userbot')
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Apply and roll out userbot panel
|
||||
shell: bash
|
||||
run: |
|
||||
kubectl apply -f userbot/k8s/base/panel.yaml
|
||||
kubectl get secret userbot-common-secrets -n default -o json \
|
||||
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
|
||||
| kubectl apply -f -
|
||||
# Keep legacy deployments (forust/anna) in sync with manifests; they have no replicas field, so apply leaves scaling to the user manager only.
|
||||
kubectl apply -f userbot/k8s/base/userbots.yaml
|
||||
kubectl rollout restart deployment/userbot-panel -n userbot
|
||||
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
|
||||
@@ -1,155 +0,0 @@
|
||||
name: deploy
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: deploy-main
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
redeploy:
|
||||
runs-on: [self-hosted, linux, arch, homelab, prod]
|
||||
steps:
|
||||
- name: Redeploy workstation
|
||||
shell: bash
|
||||
env:
|
||||
DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }}
|
||||
DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }}
|
||||
DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
|
||||
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
|
||||
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
|
||||
# Set APPLY_PRUNE=true to enable kubectl apply --prune. Requires every
|
||||
# manifest to carry label app.kubernetes.io/managed-by=homelab-deploy,
|
||||
# otherwise previously applied resources get deleted on the next run.
|
||||
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
: "${DEPLOY_HOST:?missing DEPLOY_HOST}"
|
||||
: "${DEPLOY_USER:?missing DEPLOY_USER}"
|
||||
: "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}"
|
||||
|
||||
deploy_port="${DEPLOY_PORT:-22}"
|
||||
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
|
||||
|
||||
ssh_key="$RUNNER_TEMP/deploy_key"
|
||||
mkdir -p "$RUNNER_TEMP"
|
||||
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
|
||||
chmod 600 "$ssh_key"
|
||||
|
||||
ssh_opts=(
|
||||
-i "$ssh_key"
|
||||
-p "$deploy_port"
|
||||
-o BatchMode=yes
|
||||
-o StrictHostKeyChecking=accept-new
|
||||
)
|
||||
|
||||
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
|
||||
"DEPLOY_PATH=$(printf '%q' \"$deploy_path\") APPLY_PRUNE=$(printf '%q' \"${APPLY_PRUNE:-false}\") bash -se" <<'EOF'
|
||||
set -euo pipefail
|
||||
|
||||
repo="${DEPLOY_PATH:-/srv/homelab}"
|
||||
|
||||
if [ ! -d "$repo/.git" ]; then
|
||||
echo "Repository not found at $repo"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
git -C "$repo" fetch origin main
|
||||
git -C "$repo" reset --hard origin/main
|
||||
|
||||
# Runtime selection: a service is k8s-managed when $SERVICE/k8s/active
|
||||
# exists. Otherwise it is compose-managed, and only k8s/routing/*
|
||||
# manifests (external Services / EndpointSlices / ServersTransport /
|
||||
# Ingresses that route to docker backends) are applied.
|
||||
# migrate: touch SERVICE/k8s/active (+ move routing files up)
|
||||
# rollback: rm SERVICE/k8s/active
|
||||
collect_k8s() {
|
||||
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
|
||||
! -path '*/routing/*' ! -path '*/overlays/*' \
|
||||
! -name 'kustomization.y*ml' ! -name '*.example.y*ml' \
|
||||
! -name '*values.y*ml' ! -name 'patch-*.y*ml' \
|
||||
| sort
|
||||
}
|
||||
|
||||
collect_k8s_inactive() {
|
||||
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \
|
||||
\( -name 'namespace.y*ml' -o -path '*/routing/*' \) \
|
||||
! -path '*/overlays/*' ! -name '*.example.y*ml' \
|
||||
| sort
|
||||
}
|
||||
|
||||
mapfile -t compose_stacks < <(
|
||||
find "$repo" -type f \( -name 'compose.yaml' -o -name 'compose.yml' \) | sort
|
||||
)
|
||||
|
||||
mapfile -t k8s_manifests < <(
|
||||
for kd in $(find "$repo" -type d -name k8s ! -path '*/.git/*' | sort); do
|
||||
if [ -f "$kd/active" ]; then
|
||||
collect_k8s "$kd"
|
||||
else
|
||||
collect_k8s_inactive "$kd"
|
||||
fi
|
||||
done
|
||||
)
|
||||
|
||||
echo "== Validate compose stacks =="
|
||||
for cf in "${compose_stacks[@]}"; do
|
||||
dir=$(dirname "$cf")
|
||||
if [ -f "$dir/k8s/active" ]; then
|
||||
echo " skip (k8s-managed): $dir"
|
||||
continue
|
||||
fi
|
||||
echo " config: $cf"
|
||||
docker compose -f "$cf" config --quiet
|
||||
done
|
||||
|
||||
echo "== Validate k8s manifests (kubectl dry-run) =="
|
||||
for m in "${k8s_manifests[@]}"; do
|
||||
echo " apply --dry-run=client $m"
|
||||
kubectl apply --dry-run=client -f "$m" >/dev/null
|
||||
done
|
||||
|
||||
echo "== Applying Kubernetes manifests =="
|
||||
ns_files=()
|
||||
other_files=()
|
||||
for m in "${k8s_manifests[@]}"; do
|
||||
case "$m" in
|
||||
*/namespace.y?ml) ns_files+=("$m") ;;
|
||||
*) other_files+=("$m") ;;
|
||||
esac
|
||||
done
|
||||
|
||||
prune_opts=()
|
||||
if [ "${APPLY_PRUNE:-false}" = "true" ]; then
|
||||
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
|
||||
fi
|
||||
|
||||
if [ "${#ns_files[@]}" -gt 0 ]; then
|
||||
echo " namespaces first: ${ns_files[*]}"
|
||||
kubectl apply -f "${ns_files[@]}"
|
||||
fi
|
||||
if [ "${#other_files[@]}" -gt 0 ]; then
|
||||
echo " resources: ${other_files[*]}"
|
||||
kubectl apply "${prune_opts[@]}" -f "${other_files[@]}"
|
||||
fi
|
||||
|
||||
echo "== Redeploying docker compose stacks =="
|
||||
for cf in "${compose_stacks[@]}"; do
|
||||
dir=$(dirname "$cf")
|
||||
if [ -f "$dir/k8s/active" ]; then
|
||||
echo " skip (k8s-managed): $dir"
|
||||
continue
|
||||
fi
|
||||
echo " compose: $dir"
|
||||
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
|
||||
docker compose -f "$cf" build
|
||||
docker compose -f "$cf" push
|
||||
fi
|
||||
docker compose -f "$cf" up -d --pull always --remove-orphans
|
||||
done
|
||||
EOF
|
||||
+10
@@ -21,6 +21,9 @@ checkmk/checkmk/*
|
||||
downtify/Downtify_downloads
|
||||
headscale/config/*
|
||||
headscale/data/*
|
||||
# NetBird local hostnames and generated secrets
|
||||
netbird/.env
|
||||
netbird/secrets/
|
||||
searxng/core-config/*
|
||||
|
||||
# Steaming services files
|
||||
@@ -41,6 +44,7 @@ traefik/dynamic/fileservers.yml
|
||||
traefik/dynamic/*.local.y*ml.*
|
||||
traefik/dynamic/*.external.y*ml
|
||||
traefik/k8s/fileservers.y*ml
|
||||
traefik/k8s/aliasHeadersStrategy.md
|
||||
|
||||
traefik/logs/*
|
||||
|
||||
@@ -92,6 +96,8 @@ replacements.txt
|
||||
# Temp files
|
||||
edu_master/temp/
|
||||
temp/*
|
||||
# Local-only tooling scratch space (pinned CI tools, verification scripts)
|
||||
tmp/
|
||||
|
||||
# Environment
|
||||
.env
|
||||
@@ -103,6 +109,10 @@ temp/*
|
||||
# kubernetes
|
||||
*/k8s/*secret*
|
||||
!*/k8s/*secret*.example
|
||||
**/k8s/*secret*
|
||||
!**/k8s/*secret*.example
|
||||
# Local-only tweaks, not for upstream
|
||||
prometheus-stack/k8s/grafana-values.yaml
|
||||
traefik/k8s/local-tls.yaml
|
||||
converters/k8s/config.yaml
|
||||
convertx/k8s/config.yaml
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
adguard:
|
||||
image: adguard/adguardhome:latest
|
||||
image: adguard/adguardhome:v0.107.79
|
||||
container_name: adguardhome
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
|
||||
@@ -62,10 +62,12 @@ spec:
|
||||
metadata:
|
||||
labels:
|
||||
app: adguard
|
||||
annotations:
|
||||
reloader.stakater.com/auto: "true"
|
||||
spec:
|
||||
containers:
|
||||
- name: adguard
|
||||
image: adguard/adguardhome:latest
|
||||
image: adguard/adguardhome:v0.107.79
|
||||
resources:
|
||||
limits:
|
||||
memory: "1.5Gi"
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: adguard-certs
|
||||
namespace: adguard
|
||||
spec:
|
||||
secretName: adguard-certs
|
||||
dnsNames:
|
||||
- dns.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
@@ -7,18 +7,21 @@ spec:
|
||||
entryPoints:
|
||||
- websecure
|
||||
routes:
|
||||
- match: Host(`adguard.forust.xyz`) || Host(`dns.forust.xyz`)
|
||||
- match: Host(`dns.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: adguard-service
|
||||
port: 3000
|
||||
- match: (Host(`adguard.forust.xyz`) || Host(`dns.forust.xyz`)) && PathPrefix(`/dns-query`)
|
||||
- match: (Host(`dns.forust.xyz`)) && PathPrefix(`/dns-query`)
|
||||
kind: Rule
|
||||
services:
|
||||
- name: adguard-service
|
||||
port: 3000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: adguard-certs
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -39,3 +42,5 @@ spec:
|
||||
services:
|
||||
- name: adguard-service
|
||||
port: 3000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1,15 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: adguard
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
postgresql:
|
||||
image: docker.io/library/postgres:15-alpine
|
||||
image: docker.io/library/postgres:15.19-alpine
|
||||
restart: unless-stopped
|
||||
env_file:
|
||||
- .env
|
||||
|
||||
@@ -41,7 +41,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: authentik-server
|
||||
image: ghcr.io/goauthentik/server:2025.10.2
|
||||
image: ghcr.io/goauthentik/server:2026.8.3
|
||||
args: ["server"]
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
@@ -75,7 +75,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: authentik-worker
|
||||
image: ghcr.io/goauthentik/server:2025.10.2
|
||||
image: ghcr.io/goauthentik/server:2026.8.3
|
||||
args: ["worker"]
|
||||
securityContext:
|
||||
runAsUser: 0
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: authentik-prod-tls
|
||||
namespace: authentik
|
||||
spec:
|
||||
secretName: authentik-prod-tls
|
||||
dnsNames:
|
||||
- auth.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: authentik
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -6,6 +6,6 @@ metadata:
|
||||
data:
|
||||
AUTHENTIK_IMAGE: ghcr.io/goauthentik/server
|
||||
AUTHENTIK_TAG: "2025.10.2"
|
||||
AUTHENTIK_POSTGRESQL__HOST: authentik-postgres-service
|
||||
AUTHENTIK_POSTGRESQL__HOST: postgres.database.svc.cluster.local
|
||||
AUTHENTIK_POSTGRESQL__NAME: authentik
|
||||
AUTHENTIK_ERROR_REPORTING__ENABLED: "true"
|
||||
@@ -9,11 +9,14 @@ spec:
|
||||
routes:
|
||||
- match: Host(`auth.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: authentik-server-service
|
||||
port: 9000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: authentik-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -29,3 +32,5 @@ spec:
|
||||
services:
|
||||
- name: authentik-server-service
|
||||
port: 9000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -1,66 +0,0 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: authentik-postgres-service
|
||||
namespace: authentik
|
||||
spec:
|
||||
clusterIP: None
|
||||
selector:
|
||||
app: authentik-postgres
|
||||
ports:
|
||||
- port: 5432
|
||||
targetPort: 5432
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: authentik-postgres-statefulset
|
||||
namespace: authentik
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: authentik-postgres
|
||||
serviceName: authentik-postgres-service
|
||||
replicas: 1
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: authentik-postgres
|
||||
spec:
|
||||
containers:
|
||||
- name: postgres
|
||||
image: docker.io/library/postgres:15-alpine
|
||||
env:
|
||||
- name: POSTGRES_DB
|
||||
value: authentik
|
||||
- name: POSTGRES_USER
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: authentik-secrets
|
||||
key: AUTHENTIK_POSTGRESQL__USER
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: authentik-secrets
|
||||
key: AUTHENTIK_POSTGRESQL__PASSWORD
|
||||
ports:
|
||||
- containerPort: 5432
|
||||
name: postgres
|
||||
volumeMounts:
|
||||
- name: postgres-data
|
||||
mountPath: /var/lib/postgresql/data
|
||||
resources:
|
||||
requests:
|
||||
memory: "256Mi"
|
||||
cpu: "200m"
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
cpu: "500m"
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
name: postgres-data
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
resources:
|
||||
requests:
|
||||
storage: 5Gi
|
||||
@@ -0,0 +1,9 @@
|
||||
crds:
|
||||
enabled: true
|
||||
prometheus:
|
||||
servicemonitor:
|
||||
enabled: true
|
||||
interval: 60s
|
||||
scrapeTimeout: 30s
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: ClusterIssuer
|
||||
metadata:
|
||||
name: letsencrypt-staging
|
||||
spec:
|
||||
acme:
|
||||
email: bobrovod@national.shitposting.agency
|
||||
server: https://acme-staging-v02.api.letsencrypt.org/directory
|
||||
privateKeySecretRef:
|
||||
name: letsencrypt-staging-account-key
|
||||
solvers:
|
||||
- http01:
|
||||
ingress:
|
||||
class: traefik
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: ClusterIssuer
|
||||
metadata:
|
||||
name: letsencrypt-prod
|
||||
spec:
|
||||
acme:
|
||||
email: bobrovod@national.shitposting.agency
|
||||
server: https://acme-v02.api.letsencrypt.org/directory
|
||||
privateKeySecretRef:
|
||||
name: letsencrypt-prod-account-key
|
||||
solvers:
|
||||
- http01:
|
||||
ingress:
|
||||
class: traefik
|
||||
@@ -0,0 +1,30 @@
|
||||
-----BEGIN CERTIFICATE-----
|
||||
MIIFFjCCAv6gAwIBAgIUetKpTfEDOn2985FFMu6G26itT+wwDQYJKoZIhvcNAQEN
|
||||
BQAwIzEhMB8GA1UEAxMYaG9tZWxhYiBpbnRlcm5hbCByb290IENBMB4XDTI2MDky
|
||||
MzEyNDA0N1oXDTM2MDkyMDEyNDA0N1owIzEhMB8GA1UEAxMYaG9tZWxhYiBpbnRl
|
||||
cm5hbCByb290IENBMIICIjANBgkqhkiG9w0BAQEFAAOCAg8AMIICCgKCAgEAvmNP
|
||||
ZCOoD8NtNuYJKVXBlTPjX7D7sJCSK5neH7ZbYV5+lmUlEErY8Mik7j37V5k5NfpF
|
||||
Ig85pOjP7RckTPz5V6ek3yaN40s4AL053sN5ZPauDVYjalaEHTgj5sEMqlLACQWI
|
||||
yZmJOZspZykae8dIpQnqCoFpRT4FurJ78v4a0ylnFVLMQn/lyCHedwTjkEdtYWYr
|
||||
ccJy8vQwqkzs/rWvEH1lDqZhennLOrmcCjfonG7D/pruMn4z+6E28p4+ejkRrI6x
|
||||
luak3KnpT1XMeHtgU21hiRGaMDBchHMFgAhnY1qosymKenXvfTZItwgjZbwa1hJI
|
||||
GAiDm+jQDKMjzRZ3rH6Xfc0auUcykNz73PpNu1NGm78nndXwCXcXn1LFKNQJ+r1U
|
||||
sJiyAmUZmXVn4aM4OMf2F38k7wTYIKg7nRGaUkNeKDlNkjA4HvgWw+jwO1KmdHQ/
|
||||
mOem1rosDWHRK01wg+Gga9mQCnhNhxglg3t/UeSic6uOaRsvaz4qkzHq8MbCujVz
|
||||
DpKQjqdikYOAXZOs4KlBLWrS7NaK4NzfSD02pBUErh54ruJfY/bWz9KyXzBD/lQZ
|
||||
VUTKyvUVB0bkVHEdf1jJmX3H4IZRQSF5JPqOBotW6bJI5fEGNBvj9Zxy4nm2WWGz
|
||||
yyP3uWsQz8U/Wdx9nXZLHInTZBsvgLYtKUAWA30CAwEAAaNCMEAwDgYDVR0PAQH/
|
||||
BAQDAgKkMA8GA1UdEwEB/wQFMAMBAf8wHQYDVR0OBBYEFEkKm2rxPaK6+O9WD80z
|
||||
BLC6F9QsMA0GCSqGSIb3DQEBDQUAA4ICAQAnFyHz97Umf5VIu+dKTJid7C73VugJ
|
||||
TIar/xJBs/4CxP+znBxhJjXygRoyIfzoVGWcB2ZSL//vL78Qlts79K/Imc9a4RFF
|
||||
wMvCxsRXAEQ4TpeWi3ophPNcs4rhsP+gQKQFtnyKP9519bqpfxp0bTqwOV2o18fn
|
||||
za7rlQViiEnNV58j7CVoM9+mJvVVfBEX1Km+GyJL9GadzbIQ7FxClVJZefCbft93
|
||||
zHVk9gDOw8ys1XGSR2OUCyCLinXO6mqS16CmBb2MAKXq/YyH7E0N8iotAPGtfA8V
|
||||
M/0ddy947rY0xCrtECfWwvGQpJS7NRv/Z9b2jCfXrI5LXmL2nfQRg0y9GE4Vjwr+
|
||||
WxtGU5jOeFt0jQ+xRzcgG0Op+qK3x55l5LSo2hOcOVYbiHxcHEJFgwNi1ADeBFwb
|
||||
q/HdysfURSOghqjIpMMAUabBp+DBUg2EUF7pIaUqbdqExFYcr9EYisEMiNsmKmN+
|
||||
8ZbcOeerFKDQj+t/R0bFXa7UBn2UWsjI8zlR74aa2kLDXwtyz/XlO/FlYm66eBFo
|
||||
2/eYUSeU+S4ej+wUAs/dvjF7f190/DUQGuwOTlLTahqWDztmhCk7qzbECu56CwKT
|
||||
E5Ect2P72UleYwdblkVOVd352AmiwEzdOaziIRrPh8uenEknH6JBYPO3mjk7cCg+
|
||||
GPGcNIBctjdXhg==
|
||||
-----END CERTIFICATE-----
|
||||
@@ -0,0 +1,33 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: ClusterIssuer
|
||||
metadata:
|
||||
name: selfsigned
|
||||
spec:
|
||||
selfSigned: {}
|
||||
---
|
||||
# Homelab internal root CA (10y). Install the .crt on clients (see below).
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-ca-root
|
||||
namespace: cert-manager
|
||||
spec:
|
||||
isCA: true
|
||||
commonName: homelab internal root CA
|
||||
duration: 87600h
|
||||
renewBefore: 7200h
|
||||
secretName: internal-ca-root
|
||||
privateKey:
|
||||
algorithm: RSA
|
||||
size: 4096
|
||||
issuerRef:
|
||||
name: selfsigned
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: ClusterIssuer
|
||||
metadata:
|
||||
name: internal-ca
|
||||
spec:
|
||||
ca:
|
||||
secretName: internal-ca-root
|
||||
@@ -0,0 +1,4 @@
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: cert-manager
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
cloudflare-ddns:
|
||||
image: timothyjmiller/cloudflare-ddns:latest
|
||||
image: timothyjmiller/cloudflare-ddns:2.2.0
|
||||
container_name: cloudflare-ddns
|
||||
restart: unless-stopped
|
||||
security_opt:
|
||||
|
||||
@@ -18,7 +18,7 @@ spec:
|
||||
dnsPolicy: ClusterFirstWithHostNet
|
||||
containers:
|
||||
- name: cloudflare-ddns
|
||||
image: timothyjmiller/cloudflare-ddns:latest
|
||||
image: timothyjmiller/cloudflare-ddns:2.2.0
|
||||
imagePullPolicy: Always
|
||||
resources:
|
||||
requests:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
checkmk:
|
||||
image: "checkmk/check-mk-raw:2.4.0-latest"
|
||||
image: "checkmk/check-mk-raw:2.4.0-2026.09.14"
|
||||
container_name: "checkmk"
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: checkmk-prod-tls
|
||||
namespace: checkmk
|
||||
spec:
|
||||
secretName: checkmk-prod-tls
|
||||
dnsNames:
|
||||
- cmk.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: checkmk
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -7,8 +7,12 @@ spec:
|
||||
selector:
|
||||
app: checkmk
|
||||
ports:
|
||||
- port: 5000
|
||||
- name: web
|
||||
port: 5000
|
||||
targetPort: 5000
|
||||
- name: agent-receiver
|
||||
port: 8000
|
||||
targetPort: 8000
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
@@ -27,14 +31,17 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: checkmk
|
||||
image: checkmk/check-mk-raw:2.4.0-latest
|
||||
image: checkmk/check-mk-raw:2.4.0-2026.09.14
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: checkmk-secrets
|
||||
- configMapRef:
|
||||
name: checkmk-config
|
||||
ports:
|
||||
- containerPort: 5000
|
||||
- name: web
|
||||
containerPort: 5000
|
||||
- name: agent-receiver
|
||||
containerPort: 8000
|
||||
volumeMounts:
|
||||
- name: sites
|
||||
mountPath: /omd/sites
|
||||
|
||||
@@ -9,11 +9,30 @@ spec:
|
||||
routes:
|
||||
- match: Host(`cmk.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: checkmk-service
|
||||
port: 5000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: checkmk-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRouteTCP
|
||||
metadata:
|
||||
name: checkmk-agent-receiver
|
||||
namespace: checkmk
|
||||
spec:
|
||||
entryPoints:
|
||||
- checkmk-agent
|
||||
routes:
|
||||
- match: HostSNI(`*`)
|
||||
services:
|
||||
- name: checkmk-service
|
||||
port: 8000
|
||||
tls:
|
||||
passthrough: true
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -29,3 +48,5 @@ spec:
|
||||
services:
|
||||
- name: checkmk-service
|
||||
port: 5000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1 @@
|
||||
secret.yaml
|
||||
File renamed without changes.
@@ -0,0 +1,37 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: cloudflared
|
||||
labels:
|
||||
app: cloudflared
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
app: cloudflared
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: cloudflared
|
||||
spec:
|
||||
containers:
|
||||
- name: cloudflared
|
||||
image: cloudflare/cloudflared:2026.9.3
|
||||
imagePullPolicy: IfNotPresent
|
||||
args:
|
||||
- tunnel
|
||||
- --no-autoupdate
|
||||
- run
|
||||
env:
|
||||
- name: TUNNEL_TOKEN
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: cloudflared-secrets
|
||||
key: TUNNEL_TOKEN
|
||||
resources:
|
||||
requests:
|
||||
memory: "32Mi"
|
||||
cpu: "30m"
|
||||
limits:
|
||||
memory: "128Mi"
|
||||
cpu: "200m"
|
||||
@@ -0,0 +1,7 @@
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: cloudflared-secrets
|
||||
type: Opaque
|
||||
stringData:
|
||||
TUNNEL_TOKEN: your_tunnel_token_here
|
||||
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
convertx:
|
||||
container_name: convertx
|
||||
image: ghcr.io/c4illin/convertx:latest
|
||||
image: ghcr.io/c4illin/convertx:v0.18.0
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- "9992:3000"
|
||||
@@ -42,7 +42,7 @@ services:
|
||||
|
||||
bentopdf:
|
||||
container_name: bentopdf
|
||||
image: bentopdf/bentopdf:latest
|
||||
image: bentopdf/bentopdf@sha256:4eb4ec8f5030faf87c29a73d3d5a2781f28a597cf440c3ab111eb96aee550871
|
||||
restart: unless-stopped
|
||||
labels:
|
||||
- "traefik.enable=true"
|
||||
@@ -54,7 +54,7 @@ services:
|
||||
- "traefik.http.routers.bentopdf.tls.certresolver=letsencrypt"
|
||||
- "traefik.http.routers.bentopdf.tls=true"
|
||||
# Local router
|
||||
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.wokstation.internal`)"
|
||||
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.workstation.internal`)"
|
||||
- "traefik.http.routers.bentopdf-local.entrypoints=websecure"
|
||||
- "traefik.http.routers.bentopdf-local.tls=true"
|
||||
# Dev router
|
||||
|
||||
@@ -26,7 +26,7 @@ spec:
|
||||
app: bentopdf
|
||||
spec:
|
||||
containers:
|
||||
- image: bentopdf/bentopdf:latest
|
||||
- image: bentopdf/bentopdf@sha256:4eb4ec8f5030faf87c29a73d3d5a2781f28a597cf440c3ab111eb96aee550871
|
||||
imagePullPolicy: Always
|
||||
name: bentopdf
|
||||
ports:
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: convertx-prod-tls
|
||||
namespace: converters
|
||||
spec:
|
||||
secretName: convertx-prod-tls
|
||||
dnsNames:
|
||||
- forust.xyz
|
||||
- www.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: bentopdf-prod-tls
|
||||
namespace: converters
|
||||
spec:
|
||||
secretName: bentopdf-prod-tls
|
||||
dnsNames:
|
||||
- pdf.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: converters
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -26,7 +26,7 @@ spec:
|
||||
app: convertx
|
||||
spec:
|
||||
containers:
|
||||
- image: ghcr.io/c4illin/convertx:latest
|
||||
- image: ghcr.io/c4illin/convertx:v0.18.0
|
||||
name: convertx
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
|
||||
@@ -14,7 +14,7 @@ spec:
|
||||
- name: convertx-service
|
||||
port: 3000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: convertx-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -31,6 +31,9 @@ spec:
|
||||
services:
|
||||
- name: convertx-service
|
||||
port: 3000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -47,7 +50,7 @@ spec:
|
||||
- name: bentopdf-service
|
||||
port: 8080
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: bentopdf-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -63,3 +66,5 @@ spec:
|
||||
services:
|
||||
- name: bentopdf-service
|
||||
port: 8080
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1,24 @@
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: Middleware
|
||||
metadata:
|
||||
name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
spec:
|
||||
plugin:
|
||||
crowdsec-bouncer:
|
||||
enabled: true
|
||||
LogLevel: INFO
|
||||
CrowdsecMode: live
|
||||
CrowdsecLapiScheme: http
|
||||
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
|
||||
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
|
||||
# LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on
|
||||
# `GET /v1/decisions?ip=...&banned=true` before the request reaches
|
||||
# the backend, and fails CLOSED (403) if the lookup exceeds the
|
||||
# timeout. Unset, the fork defaults to 10s, which is an eternity for
|
||||
# a request path: a single slow LAPI (idle 1.3-7.4s here) turned
|
||||
# every request into a 10s hang and then a self-inflicted 403.
|
||||
# 2s keeps the fail-closed path fast and bounded; with the LAPI
|
||||
# resourced properly (see crowdsec-values.yaml) the lookup is
|
||||
# sub-100ms and this budget is never hit.
|
||||
CrowdsecLapiTimeout: "2s"
|
||||
@@ -0,0 +1,126 @@
|
||||
container_runtime: containerd
|
||||
|
||||
agent:
|
||||
acquisition: []
|
||||
additionalAcquisition:
|
||||
- labels:
|
||||
type: traefik
|
||||
limit: 1000
|
||||
query: |
|
||||
{namespace="traefik"}
|
||||
source: loki
|
||||
url: http://loki.prometheus.svc.cluster.local:3100/
|
||||
wait_for_ready: 30s
|
||||
env:
|
||||
- name: COLLECTIONS
|
||||
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
|
||||
- name: DISABLE_COLLECTIONS
|
||||
value: crowdsecurity/sshd
|
||||
metrics:
|
||||
enabled: true
|
||||
serviceMonitor:
|
||||
additionalLabels:
|
||||
release: prometheus-stack
|
||||
enabled: true
|
||||
# Static machine identity: agent pods mount pre-created LAPI credentials
|
||||
# (Secret crowdsec-agent-credentials, key local_api_credentials.yaml)
|
||||
# at the exact path the agent entrypoint expects. Together with the
|
||||
# patched register-init (enforced by janitor-cronjob.yaml) the agent
|
||||
# never calls `cscli lapi register` in steady state, so pod names,
|
||||
# restarts and reboots can no longer break it.
|
||||
extraVolumes:
|
||||
- name: static-creds
|
||||
secret:
|
||||
secretName: crowdsec-agent-credentials
|
||||
items:
|
||||
- key: local_api_credentials.yaml
|
||||
path: local_api_credentials.yaml
|
||||
extraVolumeMounts:
|
||||
- name: static-creds
|
||||
mountPath: /tmp_config/local_api_credentials.yaml
|
||||
subPath: local_api_credentials.yaml
|
||||
readOnly: true
|
||||
resources:
|
||||
limits:
|
||||
cpu: 200m
|
||||
memory: 500Mi
|
||||
requests:
|
||||
cpu: 50m
|
||||
memory: 100Mi
|
||||
|
||||
config:
|
||||
parsers:
|
||||
s02-enrich:
|
||||
mobile-whitelist.yaml: |
|
||||
name: forust/mobile-whitelist
|
||||
description: "Whitelist SWAN/4ka mobile network"
|
||||
whitelist:
|
||||
reason: "Mobile IP whitelist"
|
||||
cidr:
|
||||
- "84.245.64.0/18"
|
||||
|
||||
postoverflows:
|
||||
s01-whitelist:
|
||||
home-dynamic-ip.yaml: |
|
||||
name: forust/home-dynamic-ip
|
||||
description: "Whitelist home dynamic IP"
|
||||
whitelist:
|
||||
reason: "Home dynamic IP"
|
||||
expression:
|
||||
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
|
||||
# The hairpin-NAT address of the router (192.168.88.1) is what the
|
||||
# Gitea Actions runner presents to Traefik - it is NOT the home
|
||||
# dynamic IP, so the whitelist above did not cover it. During a
|
||||
# deploy the runner POSTs to the Actions API many times a second;
|
||||
# a single 403 storm was enough to earn it a 4h ban and break every
|
||||
# later job. Whitelisting the whole LAN also covers phones and
|
||||
# tablets browsing over 192.168.88.0/24.
|
||||
lan.yaml: |
|
||||
name: forust/lan
|
||||
description: "Whitelist local network"
|
||||
whitelist:
|
||||
reason: "Local network"
|
||||
cidr:
|
||||
- "127.0.0.0/8"
|
||||
- "10.0.0.0/8"
|
||||
- "172.16.0.0/12"
|
||||
- "192.168.0.0/16"
|
||||
|
||||
lapi:
|
||||
env:
|
||||
- name: COLLECTIONS
|
||||
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
|
||||
- name: DISABLE_COLLECTIONS
|
||||
value: crowdsecurity/linux crowdsecurity/sshd
|
||||
metrics:
|
||||
enabled: true
|
||||
serviceMonitor:
|
||||
additionalLabels:
|
||||
release: prometheus-stack
|
||||
enabled: true
|
||||
persistentVolume:
|
||||
config:
|
||||
enabled: true
|
||||
size: 100Mi
|
||||
storageClassName: local-path-retain
|
||||
data:
|
||||
enabled: true
|
||||
size: 1Gi
|
||||
storageClassName: local-path-retain
|
||||
# LAPI answers a blocking /v1/decisions lookup for EVERY bouncer-protected
|
||||
# request (whole Traefik front door), so it is the hot path of the proxy.
|
||||
# At 400m/500Mi it went CPU-throttled and idle lookups measured 1.3-7.4s,
|
||||
# which pushed requests into the bouncer's fail-closed 403.
|
||||
# Single replica on purpose: LAPI is stateful (BoltDB on the `data` PVC,
|
||||
# credentials on the `config` PVC) - two replicas sharing those RWO
|
||||
# volumes would corrupt the decision store. Scale up CPU, not replicas.
|
||||
resources:
|
||||
limits:
|
||||
cpu: 1500m
|
||||
memory: 1Gi
|
||||
requests:
|
||||
cpu: 250m
|
||||
memory: 500Mi
|
||||
service:
|
||||
type: ClusterIP
|
||||
storeLAPICscliCredentialsInSecret: true
|
||||
@@ -0,0 +1,32 @@
|
||||
apiVersion: v1
|
||||
data:
|
||||
crowdsec-overview.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"graph\",\n \"name\": \"Graph (old)\",\n \"version\": \"\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"stat\",\n \"name\": \"Stat\",\n \"version\": \"\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"timeseries\",\n \"name\": \"Time series\",\n \"version\": \"\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 24,\n \"panels\": [],\n \"title\": \"Summary\",\n \"type\": \"row\"\n },\n {\n \"cacheTimeout\": null,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [\n {\n \"options\": {\n \"match\": \"null\",\n \"result\": {\n \"text\": \"N/A\"\n }\n },\n \"type\": \"special\"\n }\n ],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"#E02F44\",\n \"value\": null\n },\n {\n \"color\": \"#E02F44\",\n \"value\": 10\n },\n {\n \"color\": \"#299c46\",\n \"value\": 10\n }\n ]\n },\n \"unit\": \"none\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 6,\n \"x\": 0,\n \"y\": 1\n },\n \"id\": 2,\n \"interval\": null,\n \"links\": [],\n \"maxDataPoints\": 100,\n \"options\": {\n \"colorMode\": \"background\",\n \"graphMode\": \"none\",\n \"justifyMode\": \"auto\",\n \"orientation\": \"horizontal\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"text\": {},\n \"textMode\": \"auto\"\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"count(cs_info)\",\n \"interval\": \"\",\n \"legendFormat\": \"\",\n \"refId\": \"A\"\n }\n ],\n \"timeFrom\": null,\n \"timeShift\": null,\n \"title\": \"Running Crowdsec\",\n \"transparent\": true,\n \"type\": \"stat\"\n },\n {\n \"aliasColors\": {},\n \"bars\": false,\n \"dashLength\": 10,\n \"dashes\": false,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"decimals\": 1,\n \"fieldConfig\": {\n \"defaults\": {\n \"links\": []\n },\n \"overrides\": []\n },\n \"fill\": 1,\n \"fillGradient\": 0,\n \"gridPos\": {\n \"h\": 8,\n \"w\": 18,\n \"x\": 6,\n \"y\": 1\n },\n \"hiddenSeries\": false,\n \"id\": 8,\n \"legend\": {\n \"alignAsTable\": true,\n \"avg\": false,\n \"current\": false,\n \"max\": false,\n \"min\": false,\n \"rightSide\": true,\n \"show\": true,\n \"sort\": \"total\",\n \"sortDesc\": true,\n \"total\": true,\n \"values\": true\n },\n \"lines\": true,\n \"linewidth\": 1,\n \"nullPointMode\": \"null\",\n \"options\": {\n \"alertThreshold\": true\n },\n \"percentage\": false,\n \"pluginVersion\": \"8.1.2\",\n \"pointradius\": 2,\n \"points\": false,\n \"renLine truncated
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/managed-by: manual
|
||||
grafana_dashboard: "1"
|
||||
name: crowdsec-crowdsec-overview
|
||||
namespace: prometheus
|
||||
---
|
||||
apiVersion: v1
|
||||
data:
|
||||
crowdsec-lapi-metrics.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"panel\",\n \"id\": \"bargauge\",\n \"name\": \"Bar gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"iteration\": 1655915193937,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 10,\n \"panels\": [],\n \"title\": \"Agents\",\n \"type\": \"row\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n },\n {\n \"color\": \"red\",\n \"value\": 80\n }\n ]\n }\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 12,\n \"x\": 0,\n \"y\": 1\n },\n \"id\": 2,\n \"options\": {\n \"displayMode\": \"gradient\",\n \"orientation\": \"vertical\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"showUnfilled\": false,\n \"text\": {}\n },\n \"pluginVersion\": \"8.1.2\",\n \"repeat\": \"query0\",\n \"repeatDirection\": \"h\",\n \"targets\": [\n {\n \"exemplar\": false,\n \"expr\": \"sum(rate(cs_lapi_request_duration_seconds_bucket{endpoint=\\\"/v1/watchers/login\\\", instance=\\\"$lapi\\\"}[$__rate_interval])) by (le)\",\n \"format\": \"heatmap\",\n \"interval\": \"\",\n \"legendFormat\": \"{{le}}\",\n \"refId\": \"A\"\n }\n ],\n \"title\": \"Agents Login\",\n \"type\": \"heatmap\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n }\n ]\n },\n \"unit\": \"none\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 8,\n \"w\": 12,\n \"x\": 12,\n \"y\": 1\n },\n \"id\": 6,\n \"options\": {\n \"displayMode\": \"gradient\",\n \"orientation\": \"auto\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"showUnfilled\": false,\n \"text\": {}\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"sum(rate(cs_lapi_request_duration_seconds_bucket{endpoint=\\\"/v1/watchers/login\\\"}[$__rate_interval])) by (le)\",\n \"format\": \"heatmap\",\n \"interval\": \"\",\n \"legendFormat\": \"{{le}}\",\n \"refId\": \"A\"\n }\n ],\n \"title\": \"Heartbeat\",\n \"type\": \"heatmap\"\n },\n {\n \"collapsed\": false,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 9\n },\n \"id\": 12,\n \"panels\": [],\n \"title\": \"Decisions\",\n \"type\": \"row\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n Line truncated
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/managed-by: manual
|
||||
grafana_dashboard: "1"
|
||||
name: crowdsec-crowdsec-lapi-metrics
|
||||
namespace: prometheus
|
||||
---
|
||||
apiVersion: v1
|
||||
data:
|
||||
crowdsec-insight.json: "{\n \"__inputs\": [\n {\n \"name\": \"DS_PROMETHEUS\",\n \"label\": \"Prometheus\",\n \"description\": \"\",\n \"type\": \"datasource\",\n \"pluginId\": \"prometheus\",\n \"pluginName\": \"Prometheus\"\n }\n ],\n \"__requires\": [\n {\n \"type\": \"panel\",\n \"id\": \"bargauge\",\n \"name\": \"Bar gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"gauge\",\n \"name\": \"Gauge\",\n \"version\": \"\"\n },\n {\n \"type\": \"grafana\",\n \"id\": \"grafana\",\n \"name\": \"Grafana\",\n \"version\": \"8.1.2\"\n },\n {\n \"type\": \"datasource\",\n \"id\": \"prometheus\",\n \"name\": \"Prometheus\",\n \"version\": \"1.0.0\"\n },\n {\n \"type\": \"panel\",\n \"id\": \"stat\",\n \"name\": \"Stat\",\n \"version\": \"\"\n }\n ],\n \"annotations\": {\n \"list\": [\n {\n \"builtIn\": 1,\n \"datasource\": \"-- Grafana --\",\n \"enable\": true,\n \"hide\": true,\n \"iconColor\": \"rgba(0, 211, 255, 1)\",\n \"name\": \"Annotations & Alerts\",\n \"target\": {\n \"limit\": 100,\n \"matchAny\": false,\n \"tags\": [],\n \"type\": \"dashboard\"\n },\n \"type\": \"dashboard\"\n }\n ]\n },\n \"editable\": true,\n \"gnetId\": null,\n \"graphTooltip\": 0,\n \"id\": null,\n \"iteration\": 1655915159751,\n \"links\": [],\n \"panels\": [\n {\n \"collapsed\": true,\n \"datasource\": null,\n \"gridPos\": {\n \"h\": 1,\n \"w\": 24,\n \"x\": 0,\n \"y\": 0\n },\n \"id\": 22,\n \"panels\": [\n {\n \"cacheTimeout\": null,\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"color\": {\n \"mode\": \"thresholds\"\n },\n \"mappings\": [\n {\n \"options\": {\n \"match\": \"null\",\n \"result\": {\n \"text\": \"N/A\"\n }\n },\n \"type\": \"special\"\n }\n ],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n },\n {\n \"color\": \"red\",\n \"value\": 80\n }\n ]\n },\n \"unit\": \"dateTimeAsIso\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 9,\n \"w\": 5,\n \"x\": 2,\n \"y\": 1\n },\n \"id\": 2,\n \"interval\": null,\n \"links\": [],\n \"maxDataPoints\": 100,\n \"options\": {\n \"colorMode\": \"none\",\n \"graphMode\": \"none\",\n \"justifyMode\": \"auto\",\n \"orientation\": \"horizontal\",\n \"reduceOptions\": {\n \"calcs\": [\n \"lastNotNull\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\n \"text\": {},\n \"textMode\": \"auto\"\n },\n \"pluginVersion\": \"8.1.2\",\n \"targets\": [\n {\n \"exemplar\": true,\n \"expr\": \"(process_start_time_seconds{instance=\\\"$instance\\\"})*1000\",\n \"interval\": \"\",\n \"legendFormat\": \"{{instance}}\",\n \"refId\": \"A\"\n }\n ],\n \"timeFrom\": null,\n \"timeShift\": null,\n \"title\": \"Up since\",\n \"type\": \"stat\"\n },\n {\n \"datasource\": \"${DS_PROMETHEUS}\",\n \"fieldConfig\": {\n \"defaults\": {\n \"displayName\": \"\",\n \"mappings\": [],\n \"thresholds\": {\n \"mode\": \"absolute\",\n \"steps\": [\n {\n \"color\": \"green\",\n \"value\": null\n }\n ]\n },\n \"unit\": \"decbytes\"\n },\n \"overrides\": []\n },\n \"gridPos\": {\n \"h\": 9,\n \"w\": 5,\n \"x\": 7,\n \"y\": 1\n },\n \"id\": 4,\n \"options\": {\n \"orientation\": \"auto\",\n \"reduceOptions\": {\n \"calcs\": [\n \"mean\"\n ],\n \"fields\": \"\",\n \"values\": false\n },\Line truncated
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/managed-by: manual
|
||||
grafana_dashboard: "1"
|
||||
name: crowdsec-crowdsec-insight
|
||||
namespace: prometheus
|
||||
@@ -0,0 +1,214 @@
|
||||
# CrowdSec self-healing: static machine identity + enforcement loops.
|
||||
#
|
||||
# Problem it fixes: the chart's agent init container runs
|
||||
# `cscli lapi register --machine "$POD_NAME" ...`
|
||||
# unconditionally. Credentials live in an emptyDir, the machine row lives
|
||||
# in LAPI's persistent DB. Any init re-run for an already-known pod name
|
||||
# (kubelet restart, node reboot) dies with
|
||||
# 403 Forbidden: user '<pod>' already exist
|
||||
# and the DaemonSet pod sticks in Init forever. Every DS restart also
|
||||
# leaves an orphan machine row that is never cleaned.
|
||||
#
|
||||
# Design (name-independent):
|
||||
# * Agent identity is a STATIC machine `crowdsec-agent-workstation`
|
||||
# whose password lives in Secret `crowdsec-agent-credentials`
|
||||
# (created once, manually - like all other secrets in this repo).
|
||||
# The secret is mounted into agent pods at
|
||||
# /tmp_config/local_api_credentials.yaml (see extraVolumeMounts in
|
||||
# crowdsec-values.yaml), which is exactly the path the agent's main
|
||||
# container copies into place at startup.
|
||||
# * The DS init command is patched (strategic merge, by container name)
|
||||
# to SKIP registration when that file exists, keeping the legacy
|
||||
# register path only as fallback. Detection marker in the patched
|
||||
# command: `[ -s /tmp_config`.
|
||||
# * This CronJob enforces the desired state hourly, so recovery is
|
||||
# automatic even after `helm upgrade` reverts the DS patch or the
|
||||
# LAPI database is wiped:
|
||||
# 1. patch DS init if it still has the unconditional register
|
||||
# (no-op otherwise - no restart churn);
|
||||
# 2. prune machines with no heartbeat for 2h (orphan hygiene);
|
||||
# 3. ensure the static machine exists, recreating it with the
|
||||
# Secret password if missing (agent retry loops reconnect
|
||||
# on their own - same name + same password);
|
||||
# 4. prune bouncer entries idle for 30d;
|
||||
# 5. delete decisions from LePresidente/http-generic-403-bf, a hub
|
||||
# scenario that bans an IP for 4h after 5 POST-403s in 10s and
|
||||
# therefore bans us for our own bouncer's fail-closed 403s.
|
||||
#
|
||||
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
|
||||
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
|
||||
# Force a run:
|
||||
# kubectl create job -n crowdsec --from=cronjob/crowdsec-janitor janitor-now
|
||||
#
|
||||
# Helm upgrades: the janitor's strategic patch puts the DS field under
|
||||
# the `kubectl-patch` field manager, so a plain `helm upgrade` FAILS
|
||||
# with an SSA conflict on initContainers[].command. Procedure:
|
||||
# 1. revert init to chart state (kills the conflict):
|
||||
# helm template crowdsec crowdsec/crowdsec --version <ver> \
|
||||
# -n crowdsec -f crowdsec/k8s/crowdsec-values.yaml > /tmp/r.yaml
|
||||
# python3 -c "import yaml,json; ..." # build revert patch from
|
||||
# the rendered DaemonSet init command, then
|
||||
# kubectl patch ds crowdsec-agent -n crowdsec \
|
||||
# --type strategic -p "\$(cat /tmp/revert_patch.json)"
|
||||
# 2. helm upgrade --install crowdsec ... (no --force needed)
|
||||
# 3. janitor-now right away (upgrade reverts init; new pods would
|
||||
# sit in Init until the next hourly run otherwise).
|
||||
#
|
||||
# One-time bootstrap (order matters):
|
||||
# 1. Create Secret + static machine (see commands in chat).
|
||||
# 2. Apply this file, trigger janitor-now, wait for agent 1/1.
|
||||
# 3. One-time orphan cleanup:
|
||||
# kubectl exec -n crowdsec deploy/crowdsec-lapi -- \
|
||||
# cscli machines prune --duration 1h --force
|
||||
# 4. Only then `helm upgrade` crowdsec with the extraVolumes values.
|
||||
# Upgrade reverts the DS patch; trigger janitor-now right after it
|
||||
# (otherwise new pods sit in Init until the next hourly run, then
|
||||
# self-heal anyway).
|
||||
#
|
||||
# Password rotation: update the Secret, delete the machine
|
||||
# (`cscli machines delete crowdsec-agent-workstation`), trigger
|
||||
# janitor-now (recreates it), then `kubectl rollout restart
|
||||
# ds/crowdsec-agent -n crowdsec` (agent reads the file at startup only).
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: crowdsec-janitor
|
||||
namespace: crowdsec
|
||||
labels:
|
||||
app.kubernetes.io/part-of: crowdsec
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: Role
|
||||
metadata:
|
||||
name: crowdsec-janitor
|
||||
namespace: crowdsec
|
||||
labels:
|
||||
app.kubernetes.io/part-of: crowdsec
|
||||
rules:
|
||||
- apiGroups: [""]
|
||||
resources: ["pods"]
|
||||
verbs: ["get", "list"]
|
||||
- apiGroups: [""]
|
||||
resources: ["pods/exec"]
|
||||
verbs: ["create"]
|
||||
- apiGroups: ["apps"]
|
||||
resources: ["daemonsets"]
|
||||
verbs: ["get", "patch"]
|
||||
# `kubectl exec deploy/<name>` resolves deploy -> replicaset -> pod,
|
||||
# which needs read access to these (exec itself is pods/exec above).
|
||||
- apiGroups: ["apps"]
|
||||
resources: ["deployments", "replicasets"]
|
||||
verbs: ["get", "list"]
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
metadata:
|
||||
name: crowdsec-janitor
|
||||
namespace: crowdsec
|
||||
labels:
|
||||
app.kubernetes.io/part-of: crowdsec
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: crowdsec-janitor
|
||||
namespace: crowdsec
|
||||
roleRef:
|
||||
kind: Role
|
||||
name: crowdsec-janitor
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
---
|
||||
apiVersion: batch/v1
|
||||
kind: CronJob
|
||||
metadata:
|
||||
name: crowdsec-janitor
|
||||
namespace: crowdsec
|
||||
labels:
|
||||
app.kubernetes.io/part-of: crowdsec
|
||||
spec:
|
||||
schedule: "17 * * * *"
|
||||
concurrencyPolicy: Forbid
|
||||
successfulJobsHistoryLimit: 3
|
||||
failedJobsHistoryLimit: 3
|
||||
jobTemplate:
|
||||
spec:
|
||||
activeDeadlineSeconds: 300
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app.kubernetes.io/part-of: crowdsec
|
||||
spec:
|
||||
serviceAccountName: crowdsec-janitor
|
||||
restartPolicy: OnFailure
|
||||
containers:
|
||||
- name: janitor
|
||||
# Same image the chart itself uses for registration jobs;
|
||||
# IfNotPresent so it works while the node is offline
|
||||
# (layer cached from the chart install).
|
||||
image: alpine/kubectl:latest
|
||||
imagePullPolicy: IfNotPresent
|
||||
env:
|
||||
- name: AGENT_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: crowdsec-agent-credentials
|
||||
key: password
|
||||
command:
|
||||
- /bin/sh
|
||||
- -c
|
||||
- |
|
||||
set -eu
|
||||
LAPI_EXEC="kubectl exec -n crowdsec deploy/crowdsec-lapi --"
|
||||
echo "== 1. enforce patched agent init =="
|
||||
CUR=$(kubectl get ds crowdsec-agent -n crowdsec \
|
||||
-o jsonpath='{.spec.template.spec.initContainers[0].command[2]}')
|
||||
case "$CUR" in
|
||||
*'-s /tmp_config'*)
|
||||
echo "init already patched"
|
||||
;;
|
||||
*)
|
||||
echo "patching init"
|
||||
WAIT='until nc "$LAPI_HOST" "$LAPI_PORT" -z'
|
||||
WAIT="$WAIT; do echo waiting for lapi to start; sleep 5; done"
|
||||
LINK='ln -s /staging/etc/crowdsec /etc/crowdsec'
|
||||
REG='cscli lapi register --machine "$USERNAME"'
|
||||
REG="$REG -u \"\$LAPI_URL\" --token \"\$REGISTRATION_TOKEN\""
|
||||
CREDS=/tmp_config/local_api_credentials.yaml
|
||||
CMD="$WAIT; $LINK; [ -s $CREDS ] || {"
|
||||
CMD="$CMD $REG && cp"
|
||||
CMD="$CMD /etc/crowdsec/local_api_credentials.yaml $CREDS; }"
|
||||
ESC=$(printf '%s' "$CMD" | sed 's/"/\\"/g')
|
||||
PATCH='{"spec":{"template":{"spec":{"initContainers":'
|
||||
PATCH=$PATCH'[{"name":"wait-for-lapi-and-register",'
|
||||
PATCH=$PATCH'"command":["sh","-c","'$ESC'"]}]}}}}'
|
||||
kubectl patch ds crowdsec-agent -n crowdsec \
|
||||
--type strategic -p "$PATCH"
|
||||
;;
|
||||
esac
|
||||
echo "== 2. prune orphan machines (no heartbeat for 2h) =="
|
||||
$LAPI_EXEC cscli machines prune --duration 2h --force
|
||||
echo "== 3. ensure static machine exists =="
|
||||
if $LAPI_EXEC cscli machines inspect \
|
||||
crowdsec-agent-workstation >/dev/null 2>&1; then
|
||||
echo "static machine present"
|
||||
else
|
||||
echo "recreating static machine"
|
||||
$LAPI_EXEC cscli machines add crowdsec-agent-workstation \
|
||||
--password "$AGENT_PASSWORD" --force
|
||||
fi
|
||||
echo "== 4. prune stale bouncers (no pull for 30d) =="
|
||||
$LAPI_EXEC cscli bouncers prune -d 720h --force
|
||||
echo "== 5. drop http-403-bf decisions (4h self-bans) =="
|
||||
# `LePresidente/http-generic-403-bf` (hub item
|
||||
# crowdsecurity/http-generic-bf v0.9) bans any source IP
|
||||
# after 5 POSTs answered 403 within 10s, for 4h. That
|
||||
# includes 403s this homelab generates ITSELF (any
|
||||
# bouncer fail-closed, any app CSRF/rate-limit 403), and a
|
||||
# 4h ban on the runner/home IP silently breaks deploys and
|
||||
# browsing. The scenario cannot be removed per-scenario -
|
||||
# it is baked into a hub item, and disabling the whole
|
||||
# base-http-scenarios collection would drop ~40 useful
|
||||
# detections. Instead we keep the detection and drop its
|
||||
# decisions hourly; the LAN/home whitelists in
|
||||
# crowdsec-values.yaml handle the legit sources, so this
|
||||
# only ever hits real scanners (who are re-banned anyway).
|
||||
$LAPI_EXEC cscli decisions delete \
|
||||
--scenario LePresidente/http-generic-403-bf --all || true
|
||||
@@ -0,0 +1,6 @@
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: crowdsec
|
||||
labels:
|
||||
app.kubernetes.io/part-of: crowdsec
|
||||
@@ -0,0 +1,34 @@
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: crowdsec-lapi
|
||||
namespace: crowdsec
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels:
|
||||
k8s-app: crowdsec
|
||||
type: lapi
|
||||
policyTypes:
|
||||
- Ingress
|
||||
ingress:
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: traefik
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: traefik
|
||||
- podSelector:
|
||||
matchLabels:
|
||||
k8s-app: crowdsec
|
||||
type: agent
|
||||
ports:
|
||||
- protocol: TCP
|
||||
port: 8080
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: prometheus
|
||||
ports:
|
||||
- protocol: TCP
|
||||
port: 6060
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
dockmon:
|
||||
image: darthnorse/dockmon:latest
|
||||
image: darthnorse/dockmon:2.5.0
|
||||
container_name: dockmon
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: dockmon-prod-tls
|
||||
namespace: dockmon
|
||||
spec:
|
||||
secretName: dockmon-prod-tls
|
||||
dnsNames:
|
||||
- dockmon.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: dockmon
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -29,7 +29,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: dockmon
|
||||
image: darthnorse/dockmon:latest
|
||||
image: darthnorse/dockmon:2.5.0
|
||||
ports:
|
||||
- containerPort: 443
|
||||
volumeMounts:
|
||||
|
||||
@@ -18,13 +18,15 @@ spec:
|
||||
- match: Host(`dockmon.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
- name: security-headers@file
|
||||
services:
|
||||
- name: dockmon-service
|
||||
port: 443
|
||||
serversTransport: dockmon-transport
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: dockmon-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -41,3 +43,5 @@ spec:
|
||||
- name: dockmon-service
|
||||
port: 443
|
||||
serversTransport: dockmon-transport
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
downtify:
|
||||
container_name: downtify
|
||||
image: ghcr.io/henriquesebastiao/downtify:latest
|
||||
image: ghcr.io/henriquesebastiao/downtify:3.1.0
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
# - '7077:8000'
|
||||
|
||||
@@ -27,7 +27,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: downtify
|
||||
image: ghcr.io/henriquesebastiao/downtify:latest
|
||||
image: ghcr.io/henriquesebastiao/downtify:3.1.0
|
||||
ports:
|
||||
- containerPort: 8000
|
||||
volumeMounts:
|
||||
|
||||
@@ -10,12 +10,14 @@ spec:
|
||||
- match: Host(`downtify.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
- name: security-chain@file
|
||||
services:
|
||||
- name: downtify-service
|
||||
port: 8000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: downtify-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -31,3 +33,5 @@ spec:
|
||||
services:
|
||||
- name: downtify-service
|
||||
port: 8000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1 @@
|
||||
1.56.0
|
||||
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
redis:
|
||||
image: redis:alpine
|
||||
image: redis:8.10.2-alpine
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- redis-data:/data
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: edu-master-webinar
|
||||
namespace: edu-master
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
spec:
|
||||
groups:
|
||||
- name: edu_master.webinar
|
||||
rules:
|
||||
# No successful webinar check for 5m (~2-3 missed 2-min checks).
|
||||
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
|
||||
# The last_success > 0 guard is mandatory: checker.py initialises
|
||||
# last_success to 0, so without it `time() - 0` equals the current epoch
|
||||
# and humanizeDuration renders ~20722d on every pod restart. Keep the
|
||||
# duration expression on the left so $value stays the real gap.
|
||||
- alert: WebinarCheckerNoSuccessfulCheck
|
||||
expr: |
|
||||
((time() - webinar_check_last_success_timestamp_seconds) > 300)
|
||||
and (webinar_check_last_success_timestamp_seconds > 0)
|
||||
and (webinar_check_last_run_timestamp_seconds > 0)
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker has no successful check for 5m"
|
||||
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
|
||||
|
||||
# Checks are running but none has ever succeeded since pod start.
|
||||
# Split out from the rule above so a zeroed gauge never feeds
|
||||
# humanizeDuration.
|
||||
- alert: WebinarCheckerNeverSucceeded
|
||||
expr: |
|
||||
(webinar_check_last_success_timestamp_seconds == 0)
|
||||
and (webinar_check_last_run_timestamp_seconds > 0)
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker has never completed a successful check"
|
||||
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
|
||||
|
||||
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
|
||||
- alert: WebinarCheckerConsecutiveFailures
|
||||
expr: |
|
||||
webinar_check_consecutive_failures >= 3
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker failing consecutively"
|
||||
description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
|
||||
|
||||
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
|
||||
- alert: WebinarCheckerScrapeDown
|
||||
expr: |
|
||||
absent(webinar_check_last_run_timestamp_seconds) == 1
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker metrics missing"
|
||||
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
|
||||
|
||||
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
|
||||
- alert: EduPhpsessidMissing
|
||||
expr: |
|
||||
edu_phpsessid_present == 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "EDU_PHPSESSID missing"
|
||||
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
|
||||
|
||||
# Hard deps: checker and playwright deployments unavailable.
|
||||
- alert: WebinarCheckerDeploymentDown
|
||||
expr: |
|
||||
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Webinar checker deployment unavailable"
|
||||
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
|
||||
|
||||
- alert: PlaywrightServiceDown
|
||||
expr: |
|
||||
kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "Playwright service unavailable"
|
||||
description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it."
|
||||
@@ -17,6 +17,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: playwright
|
||||
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
|
||||
image: mcr.microsoft.com/playwright:v1.56.0-jammy
|
||||
imagePullPolicy: IfNotPresent
|
||||
command:
|
||||
|
||||
@@ -1,11 +1,12 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: redis
|
||||
namespace: edu-master
|
||||
labels:
|
||||
app: edu-master-redis
|
||||
spec:
|
||||
serviceName: redis
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
@@ -17,7 +18,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: redis
|
||||
image: redis:alpine
|
||||
image: redis:8.10.2-alpine
|
||||
imagePullPolicy: IfNotPresent
|
||||
ports:
|
||||
- containerPort: 6379
|
||||
|
||||
@@ -21,6 +21,8 @@ stringData:
|
||||
WEBINAR_TELEGRAM_TOKEN: ""
|
||||
WEBINAR_ADMIN_ID: ""
|
||||
WEBINAR_CHECK_INTERVAL: "60"
|
||||
# Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml)
|
||||
METRICS_PORT: "8000"
|
||||
# Database
|
||||
REDIS_HOST: "redis"
|
||||
REDIS_PORT: "6379"
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: webinar-checker
|
||||
namespace: edu-master
|
||||
labels:
|
||||
app: edu-master-webinar-checker
|
||||
spec:
|
||||
selector:
|
||||
app: edu-master-webinar-checker
|
||||
ports:
|
||||
- name: metrics
|
||||
port: 8000
|
||||
targetPort: metrics
|
||||
protocol: TCP
|
||||
@@ -0,0 +1,16 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: ServiceMonitor
|
||||
metadata:
|
||||
name: webinar-checker
|
||||
namespace: edu-master
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: edu-master-webinar-checker
|
||||
endpoints:
|
||||
- port: metrics
|
||||
path: /metrics
|
||||
interval: 30s
|
||||
scrapeTimeout: 10s
|
||||
@@ -17,7 +17,7 @@ spec:
|
||||
spec:
|
||||
initContainers:
|
||||
- name: wait-redis
|
||||
image: redis:alpine
|
||||
image: redis:8.10.2-alpine
|
||||
command:
|
||||
- /bin/sh
|
||||
- -ec
|
||||
@@ -31,8 +31,7 @@ spec:
|
||||
echo "redis is ready"
|
||||
containers:
|
||||
- name: session-keeper
|
||||
image: gcr.forust.xyz/forust/session-keeper:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/session-keeper:prod
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: edu-master-secrets
|
||||
|
||||
@@ -19,7 +19,7 @@ spec:
|
||||
# redis healthy -> session-keeper healthy (EXISTS EDU_PHPSESSID) -> playwright started
|
||||
initContainers:
|
||||
- name: wait-deps
|
||||
image: redis:alpine
|
||||
image: redis:8.10.2-alpine
|
||||
command:
|
||||
- /bin/sh
|
||||
- -ec
|
||||
@@ -45,8 +45,19 @@ spec:
|
||||
echo "playwright ok"
|
||||
containers:
|
||||
- name: webinar-checker
|
||||
image: gcr.forust.xyz/forust/webinar-checker:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/webinar-checker:prod
|
||||
ports:
|
||||
- name: metrics
|
||||
containerPort: 8000
|
||||
protocol: TCP
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: metrics
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 3
|
||||
failureThreshold: 12
|
||||
initialDelaySeconds: 10
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: edu-master-secrets
|
||||
@@ -55,8 +66,8 @@ spec:
|
||||
value: "Europe/Kyiv"
|
||||
resources:
|
||||
requests:
|
||||
cpu: 25m
|
||||
memory: 128Mi
|
||||
cpu: "50m"
|
||||
memory: "128Mi"
|
||||
limits:
|
||||
cpu: 300m
|
||||
memory: 384Mi
|
||||
cpu: "600m"
|
||||
memory: "512Mi"
|
||||
@@ -2,8 +2,11 @@ FROM python:3.11-slim
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Install dependencies
|
||||
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==1.56.0 redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
|
||||
# renovate: datasource=pypi depName=playwright versioning=pep440
|
||||
ARG PLAYWRIGHT_VERSION=1.56.0
|
||||
|
||||
# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers
|
||||
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
|
||||
|
||||
COPY checker.py .
|
||||
|
||||
|
||||
@@ -1,12 +1,15 @@
|
||||
import asyncio
|
||||
import contextlib
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
from datetime import datetime, timedelta
|
||||
from html import escape
|
||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||
|
||||
import redis
|
||||
from playwright.async_api import async_playwright
|
||||
@@ -48,6 +51,106 @@ USER_AGENT = _env(
|
||||
)
|
||||
WEBINAR_TELEGRAM_TOKEN = _env('WEBINAR_TELEGRAM_TOKEN')
|
||||
ADMIN_ID = int(_env('WEBINAR_ADMIN_ID', '0'))
|
||||
METRICS_PORT = int(_env('METRICS_PORT', '8000'))
|
||||
|
||||
# --- Prometheus metrics (stdlib only, no extra deps) ---
|
||||
# Scraped by prometheus-stack via ServiceMonitor (edu_master/k8s/servicemonitor.yaml).
|
||||
# Critical alerts in edu_master/k8s/alerts.yaml fire to Telegram via Alertmanager.
|
||||
_METRICS_LOCK = threading.Lock()
|
||||
_METRICS = {
|
||||
'last_run': 0.0, # Unix ts of last check start
|
||||
'last_success': 0.0, # Unix ts of last successful check
|
||||
'last_duration': 0.0, # Duration of last check in seconds
|
||||
'success_total': 0,
|
||||
'failure_total': 0,
|
||||
'consecutive_failures': 0,
|
||||
'phpsessid_present': 1, # 1 if EDU_PHPSESSID found in redis, else 0
|
||||
}
|
||||
|
||||
|
||||
def _metric_check_start():
|
||||
with _METRICS_LOCK:
|
||||
_METRICS['last_run'] = time.time()
|
||||
|
||||
|
||||
def _metric_check_ok(duration: float):
|
||||
now = time.time()
|
||||
with _METRICS_LOCK:
|
||||
_METRICS['last_success'] = now
|
||||
_METRICS['last_duration'] = duration
|
||||
_METRICS['success_total'] += 1
|
||||
_METRICS['consecutive_failures'] = 0
|
||||
_METRICS['phpsessid_present'] = 1
|
||||
|
||||
|
||||
def _metric_check_fail(duration: float, phpsessid_missing: bool = False):
|
||||
with _METRICS_LOCK:
|
||||
_METRICS['last_duration'] = duration
|
||||
_METRICS['failure_total'] += 1
|
||||
_METRICS['consecutive_failures'] += 1
|
||||
_METRICS['phpsessid_present'] = 0 if phpsessid_missing else 1
|
||||
|
||||
|
||||
def _metrics_render() -> bytes:
|
||||
with _METRICS_LOCK:
|
||||
m = dict(_METRICS)
|
||||
lines = [
|
||||
'# HELP webinar_check_last_run_timestamp_seconds Unix timestamp of last webinar check start.',
|
||||
'# TYPE webinar_check_last_run_timestamp_seconds gauge',
|
||||
f'webinar_check_last_run_timestamp_seconds {m["last_run"]}',
|
||||
'# HELP webinar_check_last_success_timestamp_seconds Unix timestamp of last successful webinar check.',
|
||||
'# TYPE webinar_check_last_success_timestamp_seconds gauge',
|
||||
f'webinar_check_last_success_timestamp_seconds {m["last_success"]}',
|
||||
'# HELP webinar_check_last_duration_seconds Duration of last webinar check in seconds.',
|
||||
'# TYPE webinar_check_last_duration_seconds gauge',
|
||||
f'webinar_check_last_duration_seconds {m["last_duration"]}',
|
||||
'# HELP webinar_check_success_total Total successful webinar checks.',
|
||||
'# TYPE webinar_check_success_total counter',
|
||||
f'webinar_check_success_total {m["success_total"]}',
|
||||
'# HELP webinar_check_failure_total Total failed webinar checks (timeout, playwright error, page error).',
|
||||
'# TYPE webinar_check_failure_total counter',
|
||||
f'webinar_check_failure_total {m["failure_total"]}',
|
||||
'# HELP webinar_check_consecutive_failures Consecutive failed webinar checks (reset on success).',
|
||||
'# TYPE webinar_check_consecutive_failures gauge',
|
||||
f'webinar_check_consecutive_failures {m["consecutive_failures"]}',
|
||||
'# HELP edu_phpsessid_present 1 if EDU_PHPSESSID exists in redis, 0 otherwise.',
|
||||
'# TYPE edu_phpsessid_present gauge',
|
||||
f'edu_phpsessid_present {m["phpsessid_present"]}',
|
||||
]
|
||||
return ('\n'.join(lines) + '\n').encode()
|
||||
|
||||
|
||||
class _MetricsHandler(BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
if self.path == '/metrics':
|
||||
body = _metrics_render()
|
||||
self.send_response(200)
|
||||
self.send_header('Content-Type', 'text/plain; version=0.0.4')
|
||||
self.send_header('Content-Length', str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
elif self.path in ('/healthz', '/health'):
|
||||
body = b'ok\n'
|
||||
self.send_response(200)
|
||||
self.send_header('Content-Type', 'text/plain')
|
||||
self.send_header('Content-Length', str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
else:
|
||||
self.send_response(404)
|
||||
self.end_headers()
|
||||
|
||||
def log_message(self, *args):
|
||||
pass # keep bot logs clean
|
||||
|
||||
|
||||
def start_metrics_server(port: int = METRICS_PORT):
|
||||
server = HTTPServer(('0.0.0.0', port), _MetricsHandler) # noqa: S104 - k8s ServiceMonitor scrapes pod IP
|
||||
thread = threading.Thread(target=server.serve_forever, name='metrics-server', daemon=True)
|
||||
thread.start()
|
||||
logger.info(f'Metrics server listening on :{port}/metrics')
|
||||
return server
|
||||
|
||||
|
||||
# Redis Keys
|
||||
KEY_WHITELIST = 'bot:whitelist'
|
||||
@@ -597,59 +700,66 @@ async def _collect_event_times(page) -> dict:
|
||||
async def fetch_diary_data(phpsessid: str) -> dict | None:
|
||||
logger.info('Fetching diary data via Playwright...')
|
||||
try:
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.connect(PLAYWRIGHT_WS)
|
||||
try:
|
||||
context_browser = await browser.new_context(user_agent=USER_AGENT)
|
||||
await context_browser.add_cookies(
|
||||
[{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}]
|
||||
)
|
||||
page = await context_browser.new_page()
|
||||
|
||||
async with asyncio.timeout(60):
|
||||
async with async_playwright() as p:
|
||||
browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
|
||||
try:
|
||||
await page.goto(DIARY_URL, wait_until='domcontentloaded')
|
||||
await page.wait_for_selector('table.calendar', timeout=10000)
|
||||
await page.wait_for_timeout(1500)
|
||||
|
||||
table_html = await page.evaluate("""
|
||||
() => {
|
||||
const t = document.querySelector('table.calendar');
|
||||
return t ? t.outerHTML : null;
|
||||
}
|
||||
""")
|
||||
if not table_html:
|
||||
logger.error('table.calendar not found in DOM')
|
||||
return None
|
||||
|
||||
# Debug: save HTML for troubleshooting
|
||||
with contextlib.suppress(Exception), open('/tmp/diary_debug.html', 'w', encoding='utf-8') as f: # noqa: S108
|
||||
f.write(table_html)
|
||||
|
||||
month_text, days = _parse_calendar_html(table_html)
|
||||
|
||||
# Read event times by opening each event's AJAX popup.
|
||||
times_by_id = await _collect_event_times(page)
|
||||
if times_by_id:
|
||||
for day_data in days.values():
|
||||
for ev in day_data.get('events', []):
|
||||
eid = ev.get('id')
|
||||
if eid and eid in times_by_id:
|
||||
ev['time'] = times_by_id[eid]
|
||||
|
||||
logger.info(
|
||||
f'Diary parsed: month={month_text!r}, days_with_events={sum(1 for d in days.values() if d["events"])}/{len(days)}'
|
||||
context_browser = await browser.new_context(user_agent=USER_AGENT)
|
||||
await context_browser.add_cookies(
|
||||
[{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}]
|
||||
)
|
||||
page = await context_browser.new_page()
|
||||
|
||||
return {'monthFullText': month_text, 'days': days}
|
||||
try:
|
||||
await asyncio.wait_for(page.goto(DIARY_URL, wait_until='domcontentloaded'), timeout=30)
|
||||
await page.wait_for_selector('table.calendar', timeout=10000)
|
||||
await page.wait_for_timeout(1500)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'Error parsing diary: {e}')
|
||||
return None
|
||||
table_html = await page.evaluate("""
|
||||
() => {
|
||||
const t = document.querySelector('table.calendar');
|
||||
return t ? t.outerHTML : null;
|
||||
}
|
||||
""")
|
||||
if not table_html:
|
||||
logger.error('table.calendar not found in DOM')
|
||||
return None
|
||||
|
||||
# Debug: save HTML for troubleshooting
|
||||
with contextlib.suppress(Exception), open('/tmp/diary_debug.html', 'w', encoding='utf-8') as f: # noqa: S108
|
||||
f.write(table_html)
|
||||
|
||||
month_text, days = _parse_calendar_html(table_html)
|
||||
|
||||
# Read event times by opening each event's AJAX popup.
|
||||
times_by_id = await _collect_event_times(page)
|
||||
if times_by_id:
|
||||
for day_data in days.values():
|
||||
for ev in day_data.get('events', []):
|
||||
eid = ev.get('id')
|
||||
if eid and eid in times_by_id:
|
||||
ev['time'] = times_by_id[eid]
|
||||
|
||||
logger.info(
|
||||
f'Diary parsed: month={month_text!r}, days_with_events={sum(1 for d in days.values() if d["events"])}/{len(days)}'
|
||||
)
|
||||
|
||||
return {'monthFullText': month_text, 'days': days}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'Error parsing diary: {e}')
|
||||
return None
|
||||
finally:
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(page.close(), timeout=5)
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(context_browser.close(), timeout=5)
|
||||
finally:
|
||||
await page.close()
|
||||
await context_browser.close()
|
||||
finally:
|
||||
await browser.close()
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(browser.close(), timeout=5)
|
||||
except TimeoutError:
|
||||
logger.error('Diary fetch timed out (60s)')
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f'Playwright error in diary fetch: {e}')
|
||||
return None
|
||||
@@ -931,48 +1041,55 @@ def _parse_schedule_html(table_html: str) -> dict:
|
||||
async def fetch_schedule_data(phpsessid: str) -> dict | None:
|
||||
logger.info('Fetching schedule data via Playwright...')
|
||||
try:
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.connect(PLAYWRIGHT_WS)
|
||||
try:
|
||||
context_browser = await browser.new_context(user_agent=USER_AGENT)
|
||||
await context_browser.add_cookies(
|
||||
[{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}]
|
||||
)
|
||||
page = await context_browser.new_page()
|
||||
|
||||
async with asyncio.timeout(60):
|
||||
async with async_playwright() as p:
|
||||
browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
|
||||
try:
|
||||
await page.goto(SCHEDULE_URL, wait_until='domcontentloaded')
|
||||
await page.wait_for_selector('table.schedule-table', timeout=10000)
|
||||
await page.wait_for_timeout(1500)
|
||||
context_browser = await browser.new_context(user_agent=USER_AGENT)
|
||||
await context_browser.add_cookies(
|
||||
[{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}]
|
||||
)
|
||||
page = await context_browser.new_page()
|
||||
|
||||
table_html = await page.evaluate("""
|
||||
() => {
|
||||
const t = document.querySelector('table.schedule-table');
|
||||
return t ? t.outerHTML : null;
|
||||
}
|
||||
""")
|
||||
if not table_html:
|
||||
logger.error('table.schedule-table not found in DOM')
|
||||
try:
|
||||
await asyncio.wait_for(page.goto(SCHEDULE_URL, wait_until='domcontentloaded'), timeout=30)
|
||||
await page.wait_for_selector('table.schedule-table', timeout=10000)
|
||||
await page.wait_for_timeout(1500)
|
||||
|
||||
table_html = await page.evaluate("""
|
||||
() => {
|
||||
const t = document.querySelector('table.schedule-table');
|
||||
return t ? t.outerHTML : null;
|
||||
}
|
||||
""")
|
||||
if not table_html:
|
||||
logger.error('table.schedule-table not found in DOM')
|
||||
return None
|
||||
|
||||
debug_path = os.path.join(tempfile.gettempdir(), 'schedule_debug.html')
|
||||
with contextlib.suppress(Exception), open(debug_path, 'w', encoding='utf-8') as f:
|
||||
f.write(table_html)
|
||||
|
||||
data = _parse_schedule_html(table_html)
|
||||
logger.info(list(data['weekdays'].keys()))
|
||||
logger.info(f'Schedule parsed: {len(data["weekdays"])} days, classes={data["classes"]}')
|
||||
|
||||
return data
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'Error parsing schedule: {e}')
|
||||
return None
|
||||
|
||||
debug_path = os.path.join(tempfile.gettempdir(), 'schedule_debug.html')
|
||||
with contextlib.suppress(Exception), open(debug_path, 'w', encoding='utf-8') as f:
|
||||
f.write(table_html)
|
||||
|
||||
data = _parse_schedule_html(table_html)
|
||||
logger.info(list(data['weekdays'].keys()))
|
||||
logger.info(f'Schedule parsed: {len(data["weekdays"])} days, classes={data["classes"]}')
|
||||
|
||||
return data
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'Error parsing schedule: {e}')
|
||||
return None
|
||||
finally:
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(page.close(), timeout=5)
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(context_browser.close(), timeout=5)
|
||||
finally:
|
||||
await page.close()
|
||||
await context_browser.close()
|
||||
finally:
|
||||
await browser.close()
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(browser.close(), timeout=5)
|
||||
except TimeoutError:
|
||||
logger.error('Schedule fetch timed out (60s)')
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f'Playwright error in schedule fetch: {e}')
|
||||
return None
|
||||
@@ -1485,10 +1602,13 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
|
||||
int: Number of webinars found, or None if check failed
|
||||
"""
|
||||
logger.info('Running webinar check...')
|
||||
_t0 = time.time()
|
||||
_metric_check_start()
|
||||
|
||||
phpsessid = redis_client.get(KEY_PHPSESSID)
|
||||
if not phpsessid:
|
||||
logger.warning('PHPSESSID missing. Skipping check.')
|
||||
_metric_check_fail(time.time() - _t0, phpsessid_missing=True)
|
||||
# --- DEBUG LOGGING ---
|
||||
try:
|
||||
with open('phpsessid_missing.log', 'a') as f:
|
||||
@@ -1502,78 +1622,88 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
|
||||
content = ''
|
||||
|
||||
try:
|
||||
async with async_playwright() as p:
|
||||
# Connect to remote Playwright service
|
||||
browser = await p.chromium.connect(PLAYWRIGHT_WS)
|
||||
|
||||
try:
|
||||
# Create browser context with user agent
|
||||
context_browser = await browser.new_context(user_agent=USER_AGENT)
|
||||
|
||||
# Add PHPSESSID cookie
|
||||
await context_browser.add_cookies(
|
||||
[{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}]
|
||||
)
|
||||
|
||||
# Create new page
|
||||
page = await context_browser.new_page()
|
||||
async with asyncio.timeout(90):
|
||||
async with async_playwright() as p:
|
||||
# Connect to remote Playwright service
|
||||
browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
|
||||
|
||||
try:
|
||||
# Navigate to webinar page
|
||||
await page.goto(WEBINAR_URL, wait_until='domcontentloaded')
|
||||
# Create browser context with user agent
|
||||
context_browser = await browser.new_context(user_agent=USER_AGENT)
|
||||
|
||||
# Wait for the table to load
|
||||
await page.wait_for_selector('#meetings table', timeout=10000)
|
||||
await page.wait_for_timeout(2000)
|
||||
# Add PHPSESSID cookie
|
||||
await context_browser.add_cookies(
|
||||
[{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}]
|
||||
)
|
||||
|
||||
# Get page content
|
||||
content = await page.content()
|
||||
# Create new page
|
||||
page = await context_browser.new_page()
|
||||
|
||||
# Check if "no webinar" message is present
|
||||
if NO_WEBINAR_MARKER not in content:
|
||||
logger.info('!!! WEBINAR FOUND !!!')
|
||||
try:
|
||||
# Navigate to webinar page
|
||||
await asyncio.wait_for(page.goto(WEBINAR_URL, wait_until='domcontentloaded'), timeout=30)
|
||||
|
||||
# Extract webinar details from table rows
|
||||
rows = page.locator('#meetings table tbody tr')
|
||||
count = await rows.count()
|
||||
# Wait for the table to load
|
||||
await page.wait_for_selector('#meetings table', timeout=10000)
|
||||
await page.wait_for_timeout(2000)
|
||||
|
||||
for i in range(count):
|
||||
row = rows.nth(i)
|
||||
text = await row.inner_text()
|
||||
# Get page content
|
||||
content = await page.content()
|
||||
|
||||
if NO_WEBINAR_MARKER not in text:
|
||||
# Extract name (topic) from first column
|
||||
name_elem = row.locator('td').nth(0)
|
||||
name = await name_elem.inner_text()
|
||||
name = name.strip()
|
||||
# Check if "no webinar" message is present
|
||||
if NO_WEBINAR_MARKER not in content:
|
||||
logger.info('!!! WEBINAR FOUND !!!')
|
||||
|
||||
# Extract join URL from fourth column
|
||||
url_elem = row.locator('td').nth(3).locator('a[href*="/webinar/join/"]').first
|
||||
url = await url_elem.get_attribute('href')
|
||||
# Extract webinar details from table rows
|
||||
rows = page.locator('#meetings table tbody tr')
|
||||
count = await rows.count()
|
||||
|
||||
if name and url:
|
||||
current_webinars.append({'name': name, 'url': url, 'text': text.strip()})
|
||||
logger.info(f'Found webinar: {name} -> {url}')
|
||||
else:
|
||||
logger.info('No webinars found (expected message present)')
|
||||
for i in range(count):
|
||||
row = rows.nth(i)
|
||||
text = await row.inner_text()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'Error checking page: {e}. Saving content for debug.')
|
||||
# If page content is available, save it on error
|
||||
with contextlib.suppress(Exception):
|
||||
if page and not content:
|
||||
content = await page.content()
|
||||
if NO_WEBINAR_MARKER not in text:
|
||||
# Extract name (topic) from first column
|
||||
name_elem = row.locator('td').nth(0)
|
||||
name = await name_elem.inner_text()
|
||||
name = name.strip()
|
||||
|
||||
# Extract join URL from fourth column
|
||||
url_elem = row.locator('td').nth(3).locator('a[href*="/webinar/join/"]').first
|
||||
url = await url_elem.get_attribute('href')
|
||||
|
||||
if name and url:
|
||||
current_webinars.append({'name': name, 'url': url, 'text': text.strip()})
|
||||
logger.info(f'Found webinar: {name} -> {url}')
|
||||
else:
|
||||
logger.info('No webinars found (expected message present)')
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f'Error checking page: {e}. Saving content for debug.')
|
||||
# If page content is available, save it on error
|
||||
with contextlib.suppress(Exception):
|
||||
if page and not content:
|
||||
content = await page.content()
|
||||
|
||||
_metric_check_fail(time.time() - _t0)
|
||||
return None
|
||||
finally:
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(page.close(), timeout=5)
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(context_browser.close(), timeout=5)
|
||||
|
||||
return None
|
||||
finally:
|
||||
await page.close()
|
||||
await context_browser.close()
|
||||
|
||||
finally:
|
||||
await browser.close()
|
||||
with contextlib.suppress(Exception):
|
||||
await asyncio.wait_for(browser.close(), timeout=5)
|
||||
|
||||
except TimeoutError:
|
||||
logger.error('Webinar check timed out after 90s (playwright hang)')
|
||||
_metric_check_fail(time.time() - _t0)
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f'Playwright error: {e}')
|
||||
_metric_check_fail(time.time() - _t0)
|
||||
return None
|
||||
|
||||
# --- DEBUG LOGGING (Saving last response content) ---
|
||||
@@ -1637,6 +1767,7 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
|
||||
else:
|
||||
logger.info(f'Found {len(current_webinars)} webinar(s), but all are already known')
|
||||
|
||||
_metric_check_ok(time.time() - _t0)
|
||||
return len(current_webinars)
|
||||
|
||||
|
||||
@@ -1680,6 +1811,8 @@ def main():
|
||||
job_queue = app.job_queue
|
||||
job_queue.run_repeating(check_webinars_job, interval=WEBINAR_CHECK_INTERVAL, first=10)
|
||||
|
||||
start_metrics_server()
|
||||
|
||||
logger.info('Bot started polling...')
|
||||
app.run_polling()
|
||||
|
||||
|
||||
File renamed without changes.
@@ -27,7 +27,14 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: error-pages
|
||||
image: gcr.forust.xyz/forust/error-pages:latest
|
||||
image: gcr.forust.xyz/forust/error-pages:prod
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /404.html
|
||||
port: 80
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 2
|
||||
failureThreshold: 3
|
||||
---
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
server:
|
||||
image: docker.gitea.com/gitea:1.26
|
||||
image: docker.gitea.com/gitea:1.27.3
|
||||
container_name: gitea
|
||||
restart: always
|
||||
environment:
|
||||
@@ -61,7 +61,7 @@ services:
|
||||
depends_on:
|
||||
- db
|
||||
db:
|
||||
image: docker.io/library/postgres:14
|
||||
image: docker.io/library/postgres:14.24-alpine
|
||||
restart: always
|
||||
environment:
|
||||
- POSTGRES_USER=gitea
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: gitea-prod-tls
|
||||
namespace: gitea
|
||||
spec:
|
||||
secretName: gitea-prod-tls
|
||||
dnsNames:
|
||||
- gcr.forust.xyz
|
||||
- gitea.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: gitea
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -10,13 +10,13 @@ data:
|
||||
GITEA__server__SSH_PORT: "2221"
|
||||
|
||||
GITEA__database__DB_TYPE: "postgres"
|
||||
GITEA__database__HOST: "gitea-postgres-service:5432"
|
||||
GITEA__database__HOST: "postgres.database.svc.cluster.local:5432"
|
||||
GITEA__database__NAME: "gitea"
|
||||
GITEA__security__REVERSE_PROXY_LIMIT: "1"
|
||||
GITEA__security__REVERSE_PROXY_TRUSTED_PROXIES: "*"
|
||||
|
||||
GITEA__mailer__ENABLED: "false"
|
||||
|
||||
GITEA__log__logger__access__MODE: "console, file"
|
||||
GITEA__log__logger.access.MODE: "console, file"
|
||||
USER_UID: "1000"
|
||||
USER_GID: "1000"
|
||||
@@ -31,7 +31,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: gitea
|
||||
image: docker.gitea.com/gitea:1.26
|
||||
image: gitea/gitea:1.27.3
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: gitea-config
|
||||
|
||||
+17
-1
@@ -9,16 +9,29 @@ spec:
|
||||
routes:
|
||||
- match: Host(`gitea.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: gitea-service
|
||||
port: 3000
|
||||
# Registry route: NO crowdsec-bouncer.
|
||||
# The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on
|
||||
# *every* request. A deploy burst (runner Action API polls, `docker
|
||||
# manifest inspect` per own image, containerd pulls, smoke probes) fires
|
||||
# hundreds of parallel registry calls; LAPI saturation pushed the lookup
|
||||
# past the plugin timeout, and the bouncer fail-closed with 403 - which
|
||||
# containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod.
|
||||
# This route only serves authenticated OCI traffic (registry tokens,
|
||||
# basic-auth already handled by gitea) and scanners get nothing useful
|
||||
# from /v2, so there is no bruteforce surface to protect here.
|
||||
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
|
||||
kind: Rule
|
||||
services:
|
||||
- name: gitea-service
|
||||
port: 3000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: gitea-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -39,6 +52,9 @@ spec:
|
||||
services:
|
||||
- name: gitea-service
|
||||
port: 3000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRouteTCP
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: gitea-postgres-service
|
||||
namespace: gitea
|
||||
spec:
|
||||
clusterIP: None
|
||||
selector:
|
||||
app: gitea-postgres
|
||||
ports:
|
||||
- port: 5432
|
||||
targetPort: 5432
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: gitea-postgres-statefulset
|
||||
namespace: gitea
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
app: gitea-postgres
|
||||
serviceName: gitea-postgres-service
|
||||
replicas: 1
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: gitea-postgres
|
||||
spec:
|
||||
containers:
|
||||
- name: gitea-postgres
|
||||
image: postgres:14
|
||||
env:
|
||||
- name: POSTGRES_USER
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: gitea-secrets
|
||||
key: GITEA__database__USER
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: gitea-secrets
|
||||
key: GITEA__database__PASSWD
|
||||
- name: POSTGRES_DB
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: gitea-secrets
|
||||
key: GITEA__database__USER
|
||||
ports:
|
||||
- containerPort: 5432
|
||||
name: postgres
|
||||
volumeMounts:
|
||||
- name: postgres-data
|
||||
mountPath: /var/lib/postgresql/data
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
name: postgres-data
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
resources:
|
||||
requests:
|
||||
storage: 1Gi
|
||||
@@ -7,3 +7,4 @@ type: Opaque
|
||||
stringData:
|
||||
GITEA__database__USER: "gitea"
|
||||
GITEA__database__PASSWD: "gitea"
|
||||
GITEA__database__NAME: "gitea"
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
services:
|
||||
glance:
|
||||
container_name: glance
|
||||
image: glanceapp/glance
|
||||
image: glanceapp/glance:v0.8.6
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- ./config:/app/config:ro
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: glance-prod-tls
|
||||
namespace: glance
|
||||
spec:
|
||||
secretName: glance-prod-tls
|
||||
dnsNames:
|
||||
- forust.xyz
|
||||
- www.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: glance
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -27,7 +27,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: glance
|
||||
image: glanceapp/glance
|
||||
image: glanceapp/glance:v0.8.6
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: glance-secrets
|
||||
@@ -56,11 +56,11 @@ spec:
|
||||
readOnly: true
|
||||
resources:
|
||||
requests:
|
||||
memory: "30Mi"
|
||||
cpu: "20m"
|
||||
limits:
|
||||
memory: "100Mi"
|
||||
cpu: "50m"
|
||||
memory: "64Mi"
|
||||
limits:
|
||||
cpu: "200m"
|
||||
memory: "256Mi"
|
||||
volumes:
|
||||
- name: glance-config
|
||||
configMap:
|
||||
|
||||
@@ -16,7 +16,7 @@ spec:
|
||||
- name: glance-service
|
||||
port: 8080
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: glance-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -35,6 +35,9 @@ spec:
|
||||
services:
|
||||
- name: glance-service
|
||||
port: 8080
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: Middleware
|
||||
|
||||
File renamed without changes.
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
headscale:
|
||||
image: headscale/headscale:latest
|
||||
image: headscale/headscale:v0.29.4
|
||||
restart: unless-stopped
|
||||
container_name: headscale-server
|
||||
command: serve
|
||||
@@ -53,7 +53,7 @@ services:
|
||||
- "traefik.http.routers.headscale-metrics-dev.service=headscale-metrics"
|
||||
- "traefik.http.routers.headscale-metrics-dev.tls=true"
|
||||
headplane:
|
||||
image: ghcr.io/tale/headplane:latest
|
||||
image: ghcr.io/tale/headplane:0.7.1
|
||||
container_name: headplane
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
@@ -100,7 +100,7 @@ services:
|
||||
- "traefik.http.routers.headplane-dev.entrypoints=websecure"
|
||||
- "traefik.http.routers.headplane-dev.tls=true"
|
||||
web:
|
||||
image: goodieshq/headscale-admin:latest
|
||||
image: goodieshq/headscale-admin:0.28.0
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- 10080:80
|
||||
|
||||
Whitespace-only changes.
@@ -0,0 +1,41 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: headscale-prod-tls
|
||||
namespace: headscale
|
||||
spec:
|
||||
secretName: headscale-prod-tls
|
||||
dnsNames:
|
||||
- hs.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: headplane-prod-tls
|
||||
namespace: headscale
|
||||
spec:
|
||||
secretName: headplane-prod-tls
|
||||
dnsNames:
|
||||
- hp.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: headscale
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -23,16 +23,22 @@ spec:
|
||||
port: 8080
|
||||
- match: Host(`hs.forust.xyz`) && PathPrefix(`/admin`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: headscale-ui-external
|
||||
port: 80
|
||||
- match: Host(`hs.forust.xyz`) && PathPrefix(`/metrics`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: headscale-server-external
|
||||
port: 9090
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: headscale-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -47,6 +53,8 @@ spec:
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: headplane-prefix
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: headplane-external
|
||||
port: 3000
|
||||
@@ -56,7 +64,7 @@ spec:
|
||||
- name: headplane-external
|
||||
port: 3000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: headplane-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -82,6 +90,9 @@ spec:
|
||||
services:
|
||||
- name: headscale-server-external
|
||||
port: 9090
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -102,3 +113,5 @@ spec:
|
||||
services:
|
||||
- name: headplane-external
|
||||
port: 3000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -0,0 +1,42 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: forust-homepage-prod-tls
|
||||
namespace: homepages
|
||||
spec:
|
||||
secretName: forust-homepage-prod-tls
|
||||
dnsNames:
|
||||
- forust.xyz
|
||||
- www.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: xdfnx-homepage-prod-tls
|
||||
namespace: homepages
|
||||
spec:
|
||||
secretName: xdfnx-homepage-prod-tls
|
||||
dnsNames:
|
||||
- xdfnx.cfd
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: homepages
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
@@ -27,10 +27,16 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: forust-homepage
|
||||
image: gcr.forust.xyz/forust/forust-homepage:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/forust-homepage:prod
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /
|
||||
port: 80
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 2
|
||||
failureThreshold: 3
|
||||
resources:
|
||||
requests:
|
||||
memory: "10Mi"
|
||||
@@ -68,10 +74,16 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: xdfnx-homepage
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
|
||||
imagePullPolicy: Always
|
||||
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
|
||||
ports:
|
||||
- containerPort: 80
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /
|
||||
port: 80
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 2
|
||||
failureThreshold: 3
|
||||
resources:
|
||||
requests:
|
||||
memory: "10Mi"
|
||||
|
||||
@@ -9,12 +9,15 @@ spec:
|
||||
routes:
|
||||
- match: Host(`forust.xyz`) || Host(`www.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
priority: 10
|
||||
services:
|
||||
- name: forust-homepage-service
|
||||
port: 80
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: forust-homepage-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -31,6 +34,9 @@ spec:
|
||||
services:
|
||||
- name: forust-homepage-service
|
||||
port: 80
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -43,11 +49,14 @@ spec:
|
||||
routes:
|
||||
- match: Host(`xdfnx.cfd`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: xdfnx-homepage-service
|
||||
port: 80
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: xdfnx-homepage-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -63,3 +72,5 @@ spec:
|
||||
services:
|
||||
- name: xdfnx-homepage-service
|
||||
port: 80
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
kener:
|
||||
image: rajnandan1/kener:4.0.23
|
||||
image: rajnandan1/kener:4.1.5
|
||||
container_name: kener
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
@@ -36,7 +36,7 @@ services:
|
||||
- proxy
|
||||
- kener
|
||||
redis:
|
||||
image: redis:7-alpine
|
||||
image: redis:8.10.2-alpine
|
||||
container_name: kener-redis
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
|
||||
@@ -9,11 +9,14 @@ spec:
|
||||
routes:
|
||||
- match: Host(`status.forust.xyz`)
|
||||
kind: Rule
|
||||
middlewares:
|
||||
- name: crowdsec-bouncer
|
||||
namespace: crowdsec
|
||||
services:
|
||||
- name: kener-service
|
||||
port: 3000
|
||||
tls:
|
||||
certResolver: letsencrypt
|
||||
secretName: kener-prod-tls
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: IngressRoute
|
||||
@@ -29,3 +32,5 @@ spec:
|
||||
services:
|
||||
- name: kener-service
|
||||
port: 3000
|
||||
tls:
|
||||
secretName: internal-wildcard-tls
|
||||
@@ -27,7 +27,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: kener
|
||||
image: rajnandan1/kener:4.1.0
|
||||
image: rajnandan1/kener:4.1.5
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: kener-config
|
||||
|
||||
@@ -30,7 +30,7 @@ spec:
|
||||
spec:
|
||||
containers:
|
||||
- name: redis
|
||||
image: redis:7-alpine
|
||||
image: redis:8.10.2-alpine
|
||||
ports:
|
||||
- containerPort: 6379
|
||||
volumeMounts:
|
||||
|
||||
Whitespace-only changes.
@@ -0,0 +1,57 @@
|
||||
# Pinned chart: grafana/alloy 1.12.1 (app v1.19.2).
|
||||
# Install (deferred to deploy task, namespace prometheus):
|
||||
# helm upgrade --install alloy grafana/alloy --version 1.12.1 \
|
||||
# --namespace prometheus --values loki/k8s/alloy-values.yaml --wait
|
||||
# DaemonSet ships k8s pod logs (API-tailed, no hostPath mounts) to Loki.
|
||||
# Scope phase 1: k8s only, compose leftovers out.
|
||||
|
||||
controller:
|
||||
type: daemonset
|
||||
resources:
|
||||
requests:
|
||||
memory: "128Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "512Mi"
|
||||
cpu: "500m"
|
||||
|
||||
image:
|
||||
tag: "v1.19.2"
|
||||
|
||||
alloy:
|
||||
configMap:
|
||||
create: true
|
||||
content: |
|
||||
discovery.kubernetes "pods" {
|
||||
role = "pod"
|
||||
}
|
||||
|
||||
discovery.relabel "pods" {
|
||||
targets = discovery.kubernetes.pods.targets
|
||||
|
||||
rule {
|
||||
source_labels = ["__meta_kubernetes_namespace"]
|
||||
target_label = "namespace"
|
||||
}
|
||||
|
||||
rule {
|
||||
source_labels = ["__meta_kubernetes_pod_name"]
|
||||
target_label = "pod"
|
||||
}
|
||||
|
||||
rule {
|
||||
source_labels = ["__meta_kubernetes_pod_container_name"]
|
||||
target_label = "container"
|
||||
}
|
||||
}
|
||||
|
||||
loki.source.kubernetes "pods" {
|
||||
targets = discovery.relabel.pods.output
|
||||
forward_to = [loki.write.default.receiver]
|
||||
}
|
||||
|
||||
loki.write "default" {
|
||||
endpoint {
|
||||
url = "http://loki-gateway.prometheus.svc.cluster.local/loki/api/v1/push"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
# Pinned chart: grafana/loki 7.3.0 (app 3.6.12).
|
||||
# Install (deferred to deploy task, namespace prometheus):
|
||||
# helm upgrade --install loki grafana/loki --version 7.3.0 \
|
||||
# --namespace prometheus --values loki/k8s/loki-values.yaml --wait
|
||||
# SingleBinary, filesystem storage on local-path-retain, 14d retention.
|
||||
# No IngressRoute: Loki is cluster-internal, queried via Grafana datasource.
|
||||
|
||||
deploymentMode: SingleBinary
|
||||
|
||||
loki:
|
||||
# Multitenancy off: single-node homelab, gateway + Alloy + Grafana talk to one tenant.
|
||||
auth_enabled: false
|
||||
image:
|
||||
tag: "3.6.12"
|
||||
commonConfig:
|
||||
# Single replica: default RF=3 would require 3 ingesters and fail all writes.
|
||||
replication_factor: 1
|
||||
storage:
|
||||
type: filesystem
|
||||
schemaConfig:
|
||||
configs:
|
||||
- from: "2024-04-01"
|
||||
store: tsdb
|
||||
object_store: filesystem
|
||||
schema: v13
|
||||
index:
|
||||
prefix: index_
|
||||
period: 24h
|
||||
compactor:
|
||||
retention_enabled: true
|
||||
delete_request_store: filesystem
|
||||
limits_config:
|
||||
retention_period: 336h
|
||||
rulerConfig:
|
||||
wal:
|
||||
dir: /var/loki/ruler-wal
|
||||
storage:
|
||||
type: local
|
||||
local:
|
||||
directory: /var/loki/rules
|
||||
|
||||
singleBinary:
|
||||
replicas: 1
|
||||
persistence:
|
||||
enabled: true
|
||||
size: 20Gi
|
||||
storageClass: local-path-retain
|
||||
resources:
|
||||
requests:
|
||||
memory: "512Mi"
|
||||
cpu: "200m"
|
||||
limits:
|
||||
memory: "2Gi"
|
||||
cpu: "1000m"
|
||||
|
||||
# Zeroed: unused in SingleBinary mode (chart validation requires it).
|
||||
write:
|
||||
replicas: 0
|
||||
read:
|
||||
replicas: 0
|
||||
backend:
|
||||
replicas: 0
|
||||
|
||||
gateway:
|
||||
replicas: 1
|
||||
resources:
|
||||
requests:
|
||||
memory: "64Mi"
|
||||
cpu: "50m"
|
||||
limits:
|
||||
memory: "256Mi"
|
||||
cpu: "300m"
|
||||
|
||||
monitoring:
|
||||
serviceMonitor:
|
||||
enabled: true
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
interval: 15s
|
||||
rules:
|
||||
enabled: true
|
||||
namespace: prometheus
|
||||
labels:
|
||||
release: prometheus-stack
|
||||
|
||||
# Disabled: memcached caches don't fit a memory-tight single node.
|
||||
# SingleBinary works without them (slower repeated queries, fine at homelab scale).
|
||||
resultsCache:
|
||||
enabled: false
|
||||
chunksCache:
|
||||
enabled: false
|
||||
|
||||
# Disabled: synthetic canary traffic + helm test pod, noise on a single node.
|
||||
lokiCanary:
|
||||
enabled: false
|
||||
test:
|
||||
enabled: false
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
services:
|
||||
metube:
|
||||
image: ghcr.io/alexta69/metube
|
||||
image: ghcr.io/alexta69/metube:2026.09.27
|
||||
container_name: metube
|
||||
restart: unless-stopped
|
||||
# ports:
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: metube-prod-tls
|
||||
namespace: metube
|
||||
spec:
|
||||
secretName: metube-prod-tls
|
||||
dnsNames:
|
||||
- metube.forust.xyz
|
||||
issuerRef:
|
||||
name: letsencrypt-prod
|
||||
kind: ClusterIssuer
|
||||
---
|
||||
apiVersion: cert-manager.io/v1
|
||||
kind: Certificate
|
||||
metadata:
|
||||
name: internal-wildcard-tls
|
||||
namespace: metube
|
||||
spec:
|
||||
secretName: internal-wildcard-tls
|
||||
dnsNames:
|
||||
- "*.workstation.internal"
|
||||
- "*.gigaforust.internal"
|
||||
- workstation.internal
|
||||
- gigaforust.internal
|
||||
issuerRef:
|
||||
name: internal-ca
|
||||
kind: ClusterIssuer
|
||||
Loaded 100 of 223 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user