Compare commits

..
Author SHA1 Message Date
renovate-bot 659810afff chore(config): migrate config renovate.json
deploy / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
ci / lint-prettier (push) Failing after 3s
ci / lint-ruff (push) Successful in 1s
ci / lint-yaml (push) Successful in 2s
ci / lint-dockerfiles (push) Successful in 0s
ci / validate (push) Successful in 1s
ci / lint-prettier (pull_request) Failing after 3s
ci / lint-ruff (pull_request) Successful in 1s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 24s
ci / build (push) Skipped
ci / lint-yaml (pull_request) Successful in 3s
ci / lint-dockerfiles (pull_request) Successful in 1s
ci / validate (pull_request) Successful in 1s
2026-09-25 22:19:25 +00:00
53 changed files with 621 additions and 2589 deletions

No files matched your search

-10
View File
@@ -1,10 +0,0 @@
# actionlint configuration. Passed explicitly from the ci workflow:
# actionlint -config-file .gitea/actionlint.yaml .gitea/workflows/*.yaml
#
# The self-hosted act_runner registers custom labels that actionlint cannot know
# about, so declare them here instead of silencing the whole runner-label check.
self-hosted-runner:
labels:
- arch
- homelab
- prod
+13 -362
View File
@@ -7,12 +7,6 @@ on:
pull_request:
workflow_dispatch:
# Every job here is checkout plus local tools. The token needs to read the tree
# and nothing else, and saying so keeps a future step that reaches for the API
# from quietly holding a token that can write to the repository.
permissions:
contents: read
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.ref != 'refs/heads/main' }}
@@ -21,90 +15,8 @@ env:
REGISTRY: gcr.forust.xyz
jobs:
lint-compose:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# Structure check for every committed Compose file, active or not.
# Interpolation, env-file and bind-mount resolution are all switched off,
# because inactive stacks have no .env here and would only fail on their
# ${VAR:?} guards. Active stacks get the full check with interpolation in
# the deploy workflow, where the real .env files live.
- name: Validate Compose files
shell: bash
run: |
set -euo pipefail
source .gitea/workflows/compose-lint.sh
mapfile -t safe_flags < <(compose_safe_flags)
echo "docker compose config ${safe_flags[*]-}"
mapfile -t files < <(compose_files)
if [ "${#files[@]}" -eq 0 ]; then
echo "No Compose files found."
exit 0
fi
failed=0
for f in "${files[@]}"; do
if ! out="$(validate_compose_file "$f" ${safe_flags[@]+"${safe_flags[@]}"} 2>&1)"; then
failed=1
echo "::error file=${f}::$(printf '%s' "$out" | head -1)"
fi
done
if [ "$failed" -ne 0 ]; then
echo "Compose validation failed."
exit 1
fi
echo "checked ${#files[@]} Compose file(s)"
lint-actionlint:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint Gitea Actions workflows with actionlint
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint)"
export PATH="$tools_dir:$PATH"
actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml
lint-shellcheck:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint shell scripts with ShellCheck
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck)"
export PATH="$tools_dir:$PATH"
# userbot/ is a git subtree synced from forust/userbot, so its shell
# scripts are upstream's to maintain, not ours. Linting them would let a
# routine subtree pull turn the deploy gate red on code we do not own.
mapfile -t scripts < <(
git ls-files '*.sh' ':(glob)**/*.bash' ':!userbot/**'
)
if [ "${#scripts[@]}" -eq 0 ]; then
echo "No shell scripts found."
exit 0
fi
shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}"
lint-prettier:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -112,10 +24,6 @@ jobs:
- name: Check formatting with Prettier
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)"
export PATH="$tools_dir:$PATH"
mapfile -t prettier_files < <(
git ls-files \
| grep -E '\.(md|json|ya?ml|html|css)$' \
@@ -131,23 +39,17 @@ jobs:
lint-ruff:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Lint and format-check Python with Ruff
- name: Lint Python with Ruff
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff)"
export PATH="$tools_dir:$PATH"
ruff check .
ruff format --check .
lint-yaml:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -155,10 +57,6 @@ jobs:
- name: Lint YAML syntax
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)"
export PATH="$tools_dir:$PATH"
mapfile -t yaml_files < <(
git ls-files '*.yaml' '*.yml' \
':!node_modules/**' \
@@ -174,7 +72,6 @@ jobs:
lint-dockerfiles:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -182,10 +79,6 @@ jobs:
- name: Lint Dockerfiles
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)"
export PATH="$tools_dir:$PATH"
mapfile -t dockerfiles < <(
git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*'
)
@@ -197,169 +90,15 @@ jobs:
hadolint -c .hadolint.yaml "${dockerfiles[@]}"
# Known, accepted, and recorded. Each line is a real advisory against a
# package we build into the panel image, kept in this workflow rather than in
# the package manifest so that a subtree sync from forust/userbot cannot
# silently widen the exemption.
#
# starlette is the reason this job is not simply "fail on everything":
# fastapi 0.115.12 pins `starlette<0.47.0`, and the fixes for the last four
# below need 0.49.1 through 1.3.1, so clearing them means a jump from fastapi
# 0.115.12 to 0.141.x. That is upstream's call, not a drive-by in a lint
# commit. Of the seven, four are reachable here in principle: 1942 is a
# crafted Range header hitting FileResponse, and the panel serves its built
# SPA through exactly that; 249 is request.form() ignoring max_fields for
# x-www-form-urlencoded, which is the login form; 1941 is a large multipart
# body blocking the event loop; 161 and 248 are unvalidated Host and request
# path reaching request.url. 2280 needs HTTPEndpoint, which the panel does
# not use, and 2281 is Windows-only, and this deploys on Linux.
#
# The panel answers on userbot.workstation.internal and has no public
# forust.xyz route, which is what keeps the four reachable ones from being
# an internet-facing DoS. It still manages Telegram credentials.
#
# Deleting an entry here is how you accept a new advisory, so the diff says
# so out loud.
scan-deps:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Audit the Python dependencies that ship in the image
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh pip-audit)"
export PATH="$tools_dir:$PATH"
# requirements.txt, not requirements-dev.txt: this is what the image
# installs, and the test tooling is not a shipped attack surface.
pip-audit -r userbot/panel/backend/requirements.txt --strict \
--ignore-vuln CVE-2025-67720 \
--ignore-vuln PYSEC-2026-161 \
--ignore-vuln PYSEC-2026-1941 \
--ignore-vuln PYSEC-2026-1942 \
--ignore-vuln PYSEC-2026-2280 \
--ignore-vuln PYSEC-2026-2281 \
--ignore-vuln PYSEC-2026-248 \
--ignore-vuln PYSEC-2026-249
# devDependencies are excluded on purpose. `npm audit` on the full tree
# reports 7 findings, and every one of them is a build- or test-time
# package: the esbuild CORS advisory needs a vite dev server serving to
# the internet, and nanoid's infinite loop needs a custom generator
# called with size 0, which postcss does not do. None of them are in the
# 91 kB bundle the panel serves. The one production finding, devalue
# via svelte, is moderate, which is where --audit-level draws the line;
# this fails on the next high or critical one.
- name: Audit the production npm dependencies
shell: bash
run: |
set -euo pipefail
# The pinned node, not whatever the runner has. Its system node is a
# rolling Arch package: during this very push its npm was missing
# entirely, and an hour later it was npm 12 on node 26. Both are the
# wrong major anyway — the panel image is node:22-alpine.
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
export PATH="$tools_dir:$PATH"
cd userbot/panel/frontend
npm ci
npm audit --omit=dev --audit-level=high
test-backend:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# 25 tests over the panel's pydantic models, its auth flow, the SPA
# fallback and the Kubernetes client it shells out with. They existed and
# had never been executed by anything.
#
# Note that userbot/ is a subtree synced from forust/userbot, so a routine
# sync can turn this red on upstream's code. Unlike the shellcheck job,
# which skips that tree because style disagreements there are ours to
# lose, a failing test here is a real defect in a service we deploy.
- name: Run the panel backend test suite
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh uv)"
export PATH="$tools_dir:$PATH"
# A venv in a temp dir rather than a checked-out one: the runner is
# shared, and a leftover .venv would let a dependency the
# requirements no longer pin still satisfy an import.
#
# --python is not optional. uv otherwise takes whatever interpreter it
# finds first, and which one that is depends on the machine: this
# runner runs jobs on the host, where the only interpreter is 3.14,
# and pyrogram's sync.py calls the bare asyncio.get_event_loop() that
# 3.14 no longer auto-creates, so three tests fail at collection. The
# image is python:3.13-slim, so 3.13 is also the version worth
# testing: uv fetches a managed build of it when the host has none,
# which is what makes this job independent of the runner.
venv="$(mktemp -d)/venv"
uv venv --python 3.13 --quiet "$venv"
uv pip install --quiet --python "$venv/bin/python" \
-r userbot/panel/backend/requirements-dev.txt
# `python -m`, not bare `pytest`: the tests import `app.*` relative to
# the backend directory, which only works if the cwd is on sys.path,
# and only `python -m` puts it there.
cd userbot/panel/backend
"$venv/bin/python" -m pytest tests/ -q
test-frontend:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# One `npm ci` for both checks below: it is by far the slowest part of
# this job, and a second one would learn nothing the first did not.
#
# `npm ci`, not `npm install`, for the same reason the Dockerfile uses it:
# the lockfile is what makes the tree that gets checked the tree that
# gets shipped.
- name: Type-check and test the panel frontend
shell: bash
run: |
set -euo pipefail
# The pinned node, not whatever the runner has. Its system node is a
# rolling Arch package: during this very push its npm was missing
# entirely, and an hour later it was npm 12 on node 26. Both are the
# wrong major anyway — the panel image is node:22-alpine.
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
export PATH="$tools_dir:$PATH"
cd userbot/panel/frontend
npm ci
# svelte-check has been a devDependency all along with no script
# pointing at it, so the type errors it reports had nowhere to
# surface. It is clean today, which is the only reason it can be a
# gate: it stops at whatever upstream introduces rather than
# reporting a backlog we inherited.
npm run check
npm test
validate:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Validate Kubernetes manifests against JSON schemas
- name: Validate Kubernetes manifests
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
export PATH="$tools_dir:$PATH"
mapfile -t manifests < <(
git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
@@ -376,98 +115,10 @@ jobs:
-summary \
"${manifests[@]}"
# kubeconform has no schemas for CRDs, so every IngressRoute, Certificate,
# PrometheusRule, Middleware, ServersTransport and ServiceMonitor is silently
# skipped above. The live API server knows the real CRD schemas (and runs the
# cert-manager / Traefik admission webhooks), so validate there too.
#
# Only services marked with a k8s/active marker are checked: server-side
# dry-run needs the target namespace to exist, and inactive services are not
# deployed. Services being enabled for the first time are still covered by
# the JSON-schema pass above.
#
# Main pushes only. `--dry-run=server` persists nothing, but it does execute
# the admission webhooks of the production API server, so anyone able to open
# a pull request would be able to run arbitrary manifest content through
# cert-manager and Traefik. A pull request has nothing to gain from it either:
# only main is ever deployed, and this job runs to completion before the
# deploy workflow is allowed to start, so a bad CRD is still caught before
# anything reaches the cluster -- just on the push rather than on the PR.
- name: Note the server-side check is not running here
if: github.event_name == 'pull_request' || github.ref != 'refs/heads/main'
shell: bash
run: |
echo "::notice::Skipping the server-side dry-run. It executes the cert-manager and" \
"Traefik admission webhooks against the production API server, so it is limited" \
"to pushes to main. CRDs are still schema-checked by kubeconform above, and the" \
"server-side pass still runs on main before the deploy."
- name: Validate active manifests against the live API server
if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main'
shell: bash
run: |
set -euo pipefail
if ! kubectl get --raw='/readyz' --request-timeout=10s >/dev/null 2>&1; then
echo "::warning::Cluster unreachable — skipped server-side validation of CRDs (IngressRoute, Certificate, PrometheusRule). Review manifest changes manually."
exit 0
fi
mapfile -t k8s_dirs < <(
git ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
manifests=()
kustomize_apps=()
for dir in "${k8s_dirs[@]}"; do
if [ ! -f "${dir}/active" ]; then
echo "skip (no k8s/active): ${dir}"
continue
fi
if [ -f "${dir}/overlays/prod/kustomization.yaml" ]; then
kustomize_apps+=("${dir}/overlays/prod")
elif [ -f "${dir}/base/kustomization.yaml" ]; then
kustomize_apps+=("${dir}/base")
else
while IFS= read -r f; do
[ -n "$f" ] && manifests+=("$f")
done < <(
git ls-files "${dir}/*.yaml" "${dir}/*.yml" \
| grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$'
)
fi
done
echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps"
failed=0
for m in ${manifests[@]+"${manifests[@]}"}; do
if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then
failed=1
echo "::error file=${m}::$(printf '%s' "$out" | head -1)"
fi
done
for k in ${kustomize_apps[@]+"${kustomize_apps[@]}"}; do
if ! out="$(kubectl apply -k "$k" --dry-run=server 2>&1)"; then
failed=1
echo "::error file=${k}::$(printf '%s' "$out" | head -1)"
fi
done
if [ "$failed" -ne 0 ]; then
echo "Server-side validation failed. The API server (or an admission webhook) rejected these manifests."
exit 1
fi
echo "server-side dry-run: all active manifests accepted by the API server"
build:
needs:
[lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
needs: [lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 60
outputs:
services: ${{ steps.services.outputs.services }}
steps:
@@ -550,7 +201,7 @@ jobs:
case "$service" in
dtek_notif)
image="${REGISTRY}/forust/dtek-notif"
tags=()
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -573,7 +224,7 @@ jobs:
;;
errorpages)
image="${REGISTRY}/forust/error-pages"
tags=()
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -595,7 +246,7 @@ jobs:
done
;;
userbot)
tags=()
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -629,8 +280,8 @@ jobs:
done
;;
homepages)
for variant in forust xdfnx; do
case "$variant" in
for service in forust xdfnx; do
case "$service" in
forust)
image="${REGISTRY}/forust/forust-homepage"
;;
@@ -638,7 +289,7 @@ jobs:
image="${REGISTRY}/forust/xdfnx-homepage"
;;
esac
tags=()
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
@@ -654,15 +305,15 @@ jobs:
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" -f "homepages/Dockerfile.${variant}" homepages
"${build_args[@]}" -f "homepages/Dockerfile.${service}" homepages
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
done
;;
edu_master)
for variant in session-keeper webinar-checker; do
case "$variant" in
for service in session-keeper webinar-checker; do
case "$service" in
session-keeper)
context="edu_master/phpsessid-bot"
image="${REGISTRY}/forust/session-keeper"
@@ -672,7 +323,7 @@ jobs:
image="${REGISTRY}/forust/webinar-checker"
;;
esac
tags=()
tags=("latest")
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
-46
View File
@@ -1,46 +0,0 @@
#!/usr/bin/env bash
# Shared helpers for validating Compose files. Sourced both by steps in
# .gitea/workflows/ci.yaml and by deploy-lib.sh on the workstation.
#
# Two levels of checking, matching how the repo is structured:
#
# general every committed Compose file, active or not. Pure structure check:
# no ${VAR} interpolation, no .env lookup, no bind-mount path
# resolution. Disabled stacks deliberately have no .env in the repo
# and no values on the CI runner, so a full `config` run would fail on
# their `${VAR:?}` guards for reasons that have nothing to do with the
# change under review.
#
# full active stacks only, with interpolation and env-file resolution, so
# required variables and referenced files are actually resolved. Needs
# the gitignored .env files, so this only runs in the deploy workflow
# on the workstation.
#
# This file is meant to be sourced, not executed.
# All committed Compose files, including the ones deploy never starts.
compose_files() {
git ls-files \
'*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \
'*/docker-compose.yaml' '*/docker-compose.yml'
}
# Prints the flags that turn `docker compose config` into the general check.
# Probed rather than hardcoded so an older Compose without --no-env-resolution
# still gets the flags it does support.
compose_safe_flags() {
local help flag
help="$(docker compose config --help 2>/dev/null || true)"
for flag in --no-interpolate --no-env-resolution --no-path-resolution; do
if printf '%s' "$help" | grep -q -- "$flag"; then
printf '%s\n' "$flag"
fi
done
}
# validate_compose_file <file> [extra docker compose config flags...]
validate_compose_file() {
local file="$1"
shift
docker compose -f "$file" config --quiet "$@"
}
+34 -730
View File
@@ -3,34 +3,16 @@
# REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF'
# source "$REPO/.gitea/workflows/deploy-lib.sh"
# run_stage "$STAGE"
# EOF
# EOF
set -euo pipefail
: "${REPO:?REPO must be set}"
APPLY_PRUNE="${APPLY_PRUNE:-false}"
# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to
# the current origin/main.
DEPLOY_SHA="${DEPLOY_SHA:-}"
# Handoff point between the apply stage (writes) and the verify stage (reads).
# Under the deploy user's own XDG state directory rather than /var/backups: the
# deploy is unprivileged, /var/backups does not exist on a minimal Arch host, and
# creating it would need root — which is why the first real deploy died here with
# "is not writable" before touching a single workload. $HOME comes from sshd.
DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state}/homelab-deploy}"
# Per-workload rollout budget and how many workloads to watch at once. The whole
# apply job has its own timeout-minutes as a backstop.
ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}"
ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}"
WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps"
log() {
echo "== $* =="
}
warn() {
echo "WARNING: $*" >&2
}
collect_k8s() {
git -C "$REPO" ls-files -- "$1" \
| grep -E '\.ya?ml$' \
@@ -45,8 +27,6 @@ kustomize_overlay() {
echo "$1/overlays/prod"
elif [ -f "$1/base/kustomization.yaml" ]; then
echo "$1/base"
elif [ -f "$1/kustomization.yaml" ]; then
echo "$1"
fi
}
@@ -63,7 +43,7 @@ select_manifests() {
fi
overlay="$(kustomize_overlay "$kd" || true)"
if [ -n "${overlay:-}" ]; then
echo "kustomize app: ${overlay#"$REPO"/}"
echo "kustomize app: ${overlay#$REPO/}"
KUSTOMIZE_APPS+=("$overlay")
else
while IFS= read -r f; do
@@ -87,485 +67,32 @@ select_manifests() {
done < <(git -C "$REPO" ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort)
}
# --- post-apply verification and rollback -------------------------------------
#
# A green `kubectl apply` says nothing about the cluster being healthy. These
# helpers watch exactly the workloads whose spec changed during this apply, and
# on failure roll them back to the revision that was running before, so a bad
# push to main cannot leave a service crash-looping.
#
# Verification lives in its own workflow job, not at the end of the apply stage.
# Inside a single process it is worthless exactly when it is needed most: a job
# killed by timeout-minutes or cancelled mid-apply never reaches the rollback
# code, and leaves a half-applied cluster behind. Split out, the apply job can
# die in any way and the verify job still runs.
#
# That split needs a handoff point on the workstation, because the two stages are
# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written
# before anything is applied, read by the verify stage afterwards.
# Creates this run's snapshot directory and publishes it as the handoff point for
# the verify stage. Fails hard by design: a deploy that cannot record what it is
# about to change must not start, because then nothing can be rolled back for it
# automatically. Publishing happens before the first apply, so an apply killed
# mid-flight still leaves a usable baseline behind.
snapshot_dir() {
local stamp dir
stamp="$(date -u +%Y%m%dT%H%M%SZ)-${DEPLOY_SHA:-$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo unknown)}"
dir="$DEPLOY_SNAPSHOT_DIR/$stamp"
if ! mkdir -p "$DEPLOY_SNAPSHOT_DIR" 2>/dev/null || [ ! -w "$DEPLOY_SNAPSHOT_DIR" ]; then
echo "ERROR: $DEPLOY_SNAPSHOT_DIR is not writable." >&2
echo "The verify job needs it to learn which workloads this deploy touches." >&2
echo "Refusing to deploy without a way to roll back." >&2
return 1
fi
if ! mkdir -p "$dir" 2>/dev/null || [ ! -w "$dir" ]; then
echo "ERROR: cannot create snapshot dir $dir" >&2
return 1
fi
if ! printf '%s\n' "$dir" >"$DEPLOY_SNAPSHOT_DIR/current" 2>/dev/null; then
echo "ERROR: cannot publish the snapshot pointer at $DEPLOY_SNAPSHOT_DIR/current" >&2
return 1
fi
printf '%s\n' "$dir"
}
save_snapshot() {
local dir="$1"
log "Saving pre-apply snapshot to $dir"
workload_generations >"$dir/generations.before" 2>/dev/null \
|| warn "could not snapshot workload generations"
kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \
|| warn "could not snapshot workloads"
for release in prometheus-stack loki alloy; do
if helm status "$release" -n prometheus >/dev/null 2>&1; then
{
echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)"
helm get values "$release" -n prometheus --all 2>/dev/null
} >"$dir/helm-$release.txt"
fi
done
# The verify stage compares this against the commit it is deploying, to refuse
# rolling back against a baseline left by an earlier run. A snapshot we cannot
# attribute to a commit is unusable for that, so fail before anything is applied.
if ! git -C "$REPO" rev-parse HEAD >"$dir/commit" 2>/dev/null; then
echo "ERROR: cannot record the deploy commit in $dir/commit" >&2
return 1
fi
}
# Prints "<ns> <name> <kind> <generation>" for every workload in the cluster.
workload_generations() {
kubectl get "$WORKLOAD_KINDS" -A \
-o 'custom-columns=NS:.metadata.namespace,NAME:.metadata.name,KIND:.kind,GEN:.metadata.generation' \
--no-headers 2>/dev/null \
| awk 'NF >= 4 { printf "%s %s %s %s\n", $1, $2, tolower($3), $4 }'
}
# Prints "<kind> <ns> <name>" for every workload that is new or whose generation
# moved since the snapshot, i.e. the ones this apply actually touched.
changed_workloads() {
local before="$1"
local ns name kind gen old
while read -r ns name kind gen; do
[ -n "${gen:-}" ] || continue
old="$(awk -v want_ns="$ns" -v want_name="$name" \
'$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)"
if [ "$old" != "$gen" ]; then
printf '%s %s %s\n' "$kind" "$ns" "$name"
fi
done < <(workload_generations)
}
# Prints "<ns> <kind>/<name> <image>" for every workload this repository owns that
# runs an image from our own registry.
#
# The repository is the scope, deliberately. The cluster also holds workloads on
# our registry that no manifest here declares (they are applied out of band), and
# those are somebody else's to deploy. Walking the manifests rather than the
# cluster means those can never be restarted by this pipeline, now or later.
owned_registry_workloads() {
local kd_rel f
while IFS= read -r kd_rel; do
[ -f "$REPO/$kd_rel/active" ] || continue
while IFS= read -r f; do
[ -n "$f" ] || continue
# A file that does not mention the registry cannot declare a workload on it,
# and parsing costs ~2.5s per file against a millisecond for the grep. The
# filter keeps this at a handful of parses instead of one per manifest.
grep -q 'gcr\.forust\.xyz/forust/' "$REPO/$f" 2>/dev/null || continue
# kubectl prints a bare object for a single-document file and a List for a
# multi-document one, so normalise both shapes before filtering.
kubectl apply --dry-run=client -f "$REPO/$f" -o json 2>/dev/null \
| jq -r '
(if .items then .items[] else . end)
| select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$"))
| select(any((.spec.template.spec.containers // [])[]?;
(.image // "") | test("^gcr\\.forust\\.xyz/forust/")))
| (.metadata.namespace // "default") as $ns
| ([.spec.template.spec.containers[].image
| select(test("^gcr\\.forust\\.xyz/forust/"))][0]) as $img
| "\($ns) \(.kind | ascii_downcase)/\(.metadata.name) \($img)"
' 2>/dev/null || true
done < <(collect_k8s "$kd_rel" || true)
done < <(
git -C "$REPO" ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
)
}
# Prints the digest an image tag resolves to for this cluster's architecture, or
# nothing when it cannot be resolved.
#
# Only the manifest entry matching the node architecture counts. A multi-arch tag
# also carries `unknown/unknown` entries for the build attestation, and a pod's
# imageID is always the per-platform digest, so comparing the wrong entry would
# mark every workload stale forever and restart the whole cluster on every deploy.
registry_digest() {
local arch
arch="$(kubectl get nodes -o jsonpath='{.items[0].status.nodeInfo.architecture}' 2>/dev/null || true)"
[ -n "$arch" ] || arch=amd64
# The || true is load-bearing. Every caller runs under set -euo pipefail, and
# pipefail reports the rightmost non-zero stage, so a ref the registry does not
# have would abort the caller at the assignment instead of yielding an empty
# string. The callers check for empty themselves and report it by name.
docker manifest inspect "$1" 2>/dev/null \
| jq -r --arg arch "$arch" '
.manifests[]?
| select(.platform.os == "linux" and .platform.architecture == $arch)
| .digest
' 2>/dev/null \
| head -1 || true
}
# Rewrites our own images to immutable digests on the way into the cluster.
# Reads a manifest stream on stdin, writes the pinned stream to stdout.
#
# A digest is not knowable when a manifest is written, so it is resolved here, at
# apply time, and never committed: git keeps a readable `:prod` tag. That is what
# makes rollback mean something. `kubectl rollout undo` restores the previous
# ReplicaSet's pod template verbatim, and a template naming a digest restores the
# exact bytes that were serving before. A template naming a moving tag does not —
# the tag has already moved by the time the rollback runs, so the "rollback"
# re-pulls the very image that just failed and the cluster stays broken.
#
# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a
# reference that is not `:latest` defaults to IfNotPresent, which is what the
# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot
# change, so pulling again buys nothing.
#
# An image that cannot be resolved is fatal. Carrying on would quietly apply a
# mutable tag again, which is the exact failure this function exists to remove.
render_pinned() {
local src refs map ref digest missing=0
src="$(mktemp)"
refs="$(mktemp)"
map="$(mktemp)"
cat >"$src"
grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true
while read -r ref; do
[ -n "$ref" ] || continue
digest="$(registry_digest "$ref")"
if [ -z "$digest" ]; then
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
echo " The build job has to push that tag before the deploy resolves it." >&2
missing=$((missing + 1))
continue
fi
printf '%s\t%s\n' "$ref" "$digest" >>"$map"
done <"$refs"
if [ "$missing" -gt 0 ]; then
rm -f "$src" "$refs" "$map"
return 1
fi
awk -v mapfile="$map" '
BEGIN {
while ((getline line < mapfile) > 0) {
i = index(line, "\t")
d[substr(line, 1, i - 1)] = substr(line, i + 1)
}
}
{
if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) {
name = $0
sub(/^[[:space:]]*image:[[:space:]]*/, "", name)
sub(/[[:space:]]*$/, "", name)
if (name in d) {
pad = $0
sub(/image:.*/, "", pad)
# Drop the tag: the canonical form used in the docs is repo@sha256:...,
# and leaving :prod next to the digest reads like it still matters.
repo = name
sub(/:[A-Za-z0-9._-]+$/, "", repo)
print pad "image: " repo "@" d[name]
next
}
}
print
}
' "$src"
rm -f "$src" "$refs" "$map"
}
# Restarts every owned workload whose running image is not the one its tag
# resolves to now.
#
# This used to be how a rebuild reached the cluster at all: the manifests pinned
# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply`
# decided there was nothing to do, and the cluster served the previous build
# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves
# the pod template and rolls out on its own.
#
# What is left is the drift check: a hand-run `kubectl set image`, or anything
# else that edits a live workload behind the deploy's back, is the only way to end
# up serving a digest the tag has moved past. It stays idempotent, so a redeploy
# that changed no image still does not bounce healthy services.
#
# The container is matched on its repository rather than on the exact reference:
# once render_pinned has run, a pod's status reports `repo@sha256:...` while this
# still reads the repository's `:prod` tag out of the manifest.
restart_stale_images() {
local ns target image want selector running entry one
local unchecked=0
local -A digests=()
local -a stale=()
while read -r ns target image; do
[ -n "${target:-}" ] || continue
if [ -z "${digests[$image]:-}" ]; then
digests[$image]="$(registry_digest "$image")"
fi
want="${digests[$image]}"
if [ -z "$want" ]; then
warn "cannot resolve ${image##*/} in the registry, leaving $target alone"
unchecked=$((unchecked + 1))
continue
fi
selector="$(kubectl get "$target" -n "$ns" -o jsonpath='{.spec.selector.matchLabels}' 2>/dev/null \
| jq -r 'to_entries | map("\(.key)=\(.value)") | join(",")' 2>/dev/null)"
if [ -z "$selector" ]; then
warn "cannot read the pod selector of $target, skipping"
unchecked=$((unchecked + 1))
continue
fi
running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \
| jq -r --arg repo "${image%%:*}" '
.items[] | .status.containerStatuses[]?
| select(.image == $repo
or (.image | startswith($repo + ":"))
or (.image | startswith($repo + "@")))
| .imageID
' 2>/dev/null)"
if [ -z "$running" ]; then
# Scaled to zero. Nothing is serving stale code, and imagePullPolicy
# resolves the tag when it is scaled back up.
continue
fi
entry=""
while IFS= read -r one; do
[ -n "$one" ] || continue
entry="${one##*@}"
if [ "$entry" != "$want" ]; then
stale+=("$ns $target")
break
fi
done <<<"$running"
done < <(owned_registry_workloads)
if [ "${#stale[@]}" -eq 0 ]; then
if [ "$unchecked" -gt 0 ]; then
# Say so plainly. Reporting "everything is current" after checking nothing
# would tell the operator the deploy is fine when it may not be.
warn "No workload needed a restart, but $unchecked could not be checked"
else
log "All owned workloads already run the image their tag points at"
fi
return 0
fi
log "Restarting ${#stale[@]} workload(s) running an image their tag has moved past"
for ref in "${stale[@]}"; do
log " $ref"
done
local failed=()
for ref in "${stale[@]}"; do
ns="${ref%% *}"
target="${ref#* }"
if ! kubectl rollout restart "$target" -n "$ns" >/dev/null 2>&1; then
failed+=("$ref")
fi
done
if [ "${#failed[@]}" -gt 0 ]; then
warn "could not restart: ${failed[*]}"
return 1
fi
}
# verify_workloads <failed-file> <kind> <ns> <name> ...
# Watches every workload in parallel and records the ones that never became
# healthy. Returns non-zero if any of them failed.
verify_workloads() {
local failed_file="$1"
shift
[ "$#" -gt 0 ] || return 0
: >"$failed_file"
local running=0 pid kind ns name
local -a pids=()
for entry in "$@"; do
read -r kind ns name <<<"$entry"
(
if kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " ok: ${kind}/${ns}/${name}"
else
echo " FAILED: ${kind}/${ns}/${name}"
printf '%s %s %s\n' "$kind" "$ns" "$name" >>"$failed_file"
fi
) &
pids+=($!)
running=$((running + 1))
if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then
wait -n 2>/dev/null || true
running=$((running - 1))
fi
done
for pid in ${pids[@]+"${pids[@]}"}; do
wait "$pid" || true
done
# Non-zero when the file holds at least one failure, i.e. a workload never
# became healthy. `[ -s ]` alone is the opposite test and silently disabled
# every rollback this stage is meant to perform.
[ ! -s "$failed_file" ]
}
# rollback_workloads <failed-file>
# Restores the previous revision of every failed workload and waits for it to
# settle. Prints a report and returns non-zero if any workload is still unhealthy,
# so the operator knows manual recovery is required.
rollback_workloads() {
local failed_file="$1"
local kind ns name unrecovered=()
local -a recovered=()
while read -r kind ns name; do
[ -n "${kind:-}" ] || continue
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " rolled back: ${kind}/${ns}/${name}"
recovered+=("${kind}/${ns}/${name}")
else
echo " NOT RECOVERED: ${kind}/${ns}/${name}"
unrecovered+=("${kind}/${ns}/${name}")
fi
done <"$failed_file"
echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file"
echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file"
[ "${#unrecovered[@]}" -eq 0 ]
}
# Helm releases owned by this stage, one line each:
#
# release|chart|namespace|chart version|values file (rel. to $REPO)|active marker
#
# The chart version is the field Renovate keeps current. The helmv3 manager only
# understands Chart.yaml and the helm-values manager only values files, so a pin
# written straight into a `helm upgrade` command would never be updated: these
# have to be declared as custom.regex managers in renovate/renovate.json.
HELM_RELEASES=(
"prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active"
"loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active"
"alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active"
"reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active"
)
# "name url" for the Helm repository hosting a chart, empty if unknown.
helm_repo_for() {
case "$1" in
prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;;
grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;;
stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;;
esac
}
upgrade_helm_releases() {
local entry release chart namespace version values marker repo
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
IFS='|' read -r release chart namespace version values marker <<<"$entry"
if [ ! -f "$REPO/$marker" ]; then
echo "skip (no $marker): $release"
continue
fi
if [ ! -f "$REPO/$values" ]; then
echo "ERROR: $values is gitignored but missing on the workstation, restore it first."
return 1
fi
repo="$(helm_repo_for "$chart")"
if [ -z "$repo" ]; then
echo "ERROR: no Helm repository configured for chart $chart"
return 1
fi
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
helm repo update "${repo%% *}" >/dev/null 2>&1 || true
log "Upgrading $release ($chart $version)"
# --atomic rolls the release back when the upgrade times out or the workloads
# it touches never become ready, so a bad chart bump is not left half applied.
helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \
--version "$version" \
--values "$REPO/$values" \
--atomic --cleanup-on-fail --timeout 10m
done
}
stage_preflight() {
if [ ! -d "$REPO/.git" ]; then
echo "Repository not found at $REPO"
exit 1
fi
if [ -n "$DEPLOY_SHA" ]; then
log "Checking out the commit CI validated ($DEPLOY_SHA)"
git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \
|| git -C "$REPO" fetch origin main
else
git -C "$REPO" fetch origin main
fi
target="${DEPLOY_SHA:-origin/main}"
git -C "$REPO" fetch origin main
log "Workstation state"
echo " local: $(git -C "$REPO" rev-parse --short HEAD)"
echo " target: $(git -C "$REPO" rev-parse --short "$target")"
echo " remote: $(git -C "$REPO" rev-parse --short origin/main)"
if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then
echo "ERROR: workstation has local tracked modifications, refusing reset:"
git -C "$REPO" status --porcelain --untracked-files=no
git -C "$REPO" diff --stat
echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy."
exit 1
fi
git -C "$REPO" reset --hard "$target"
git -C "$REPO" reset --hard origin/main
}
stage_validate() {
cd "$REPO"
select_manifests
local m k cf
# Compose .env files and secret files are gitignored by design, so the
# workstation never has real values for the inactive stacks. This stage only
# runs the full check on active stacks; the general structure check for every
# committed Compose file (active or not) lives in the ci workflow, which has no
# .env at all.
#
# Active stacks are still validated with interpolation and env-file resolution
# off, so required-variable guards (:?) and missing local files do not fail the
# deploy. Normalization and consistency checks stay enabled.
# shellcheck source=compose-lint.sh
source "$REPO/.gitea/workflows/compose-lint.sh"
local compose_validate_flags=()
mapfile -t compose_validate_flags < <(compose_safe_flags)
log "Validate compose stacks"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " config: $cf"
validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"}
docker compose -f "$cf" config --quiet
done
log "Validate k8s manifests (kubectl dry-run=client)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
@@ -582,7 +109,7 @@ stage_validate() {
kubectl apply -k "$k" --dry-run=server >/dev/null
done
log "Checking referenced Secrets exist"
echo " (deploy never applies *secret*.yaml; create missing ones manually)"
echo " (deploy never applies *secret*.yaml; create missing ones from the laptop)"
local ref_secrets=() missing_secrets=() all_secrets s
if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then
while IFS= read -r s; do
@@ -625,13 +152,6 @@ stage_apply_k8s() {
if [ "$APPLY_PRUNE" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
# Record what is about to change, and publish it for the verify job, before
# the first apply. Both are fatal on failure: see snapshot_dir.
local snapshot
snapshot="$(snapshot_dir)" || return 1
save_snapshot "$snapshot" || return 1
if [ "${#ns_files[@]}" -gt 0 ]; then
log "Applying namespaces (${#ns_files[@]} files)"
for m in "${ns_files[@]}"; do
@@ -643,23 +163,37 @@ stage_apply_k8s() {
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
exit 1
fi
log "Upgrading kube-prometheus-stack"
helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \
--namespace prometheus \
--version 86.2.3 \
--values "$REPO/prometheus-stack/k8s/grafana-values.yaml" \
--wait --timeout 10m
fi
if [ -f "$REPO/loki/k8s/active" ]; then
log "Upgrading loki/alloy"
helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true
helm repo update grafana >/dev/null 2>&1 || true
helm upgrade --install loki grafana/loki \
--version 7.3.0 \
--namespace prometheus \
--values "$REPO/loki/k8s/loki-values.yaml" \
--wait --timeout 10m
helm upgrade --install alloy grafana/alloy \
--version 1.12.1 \
--namespace prometheus \
--values "$REPO/loki/k8s/alloy-values.yaml" \
--wait --timeout 10m
fi
upgrade_helm_releases
if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
log "Applying resources (${#other_files[@]} files)"
for m in "${other_files[@]}"; do
if ! render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -; then
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
exit 1
fi
kubectl apply "${prune_opts[@]}" -f "$m"
done
fi
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)"
if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then
echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2
exit 1
fi
log "Applying kustomize app: ${k#$REPO/}"
kubectl apply -k "$k"
done
if [ -f "$REPO/userbot/k8s/active" ]; then
log "userbot panel hook"
@@ -673,220 +207,9 @@ stage_apply_k8s() {
else
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
fi
kubectl rollout restart deployment/userbot-panel -n userbot
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
fi
restart_stale_images
# No verification here on purpose. This stage may be killed at any point by
# timeout-minutes, by the runner cancelling the job, or by a dropped SSH
# connection, and any code below that line would simply not run. stage_verify_k8s
# picks the work up from the snapshot instead.
log "Applied. Verification and rollback are the verify job's job, not this one's."
}
# Runs as its own workflow job, after apply-k8s (and apply-compose) are done —
# including when they failed, timed out or were cancelled. Reads the baseline the
# apply stage published and works out what it changed, watches those workloads,
# and rolls back the ones that never became healthy.
stage_verify_k8s() {
local pointer="$DEPLOY_SNAPSHOT_DIR/current"
local snapshot want have generations
local -a touched=()
if [ ! -s "$pointer" ]; then
echo "ERROR: no snapshot pointer at $pointer."
echo "The apply stage died before publishing any state, so there is no baseline to"
echo "tell which workloads it touched. Nothing can be rolled back automatically —"
echo "inspect the cluster by hand."
return 1
fi
snapshot="$(head -1 "$pointer")"
if [ ! -d "$snapshot" ]; then
echo "ERROR: snapshot pointer refers to a missing directory: $snapshot"
return 1
fi
# Never trust the pointer blindly. If the apply stage was killed before it
# published its own snapshot, `current` still points at the previous deploy's
# baseline. Verifying against that would watch the wrong workloads and the
# rollback would revert the wrong revisions, so refuse instead.
want="${DEPLOY_SHA:-}"
if [ -z "$want" ]; then
want="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
fi
have="$(cat "$snapshot/commit" 2>/dev/null || true)"
if [ -z "$want" ] || [ "$have" != "$want" ]; then
echo "ERROR: refusing to verify or roll back against a stale snapshot."
echo " snapshot: $snapshot"
echo " snapshot commit: ${have:-<missing>}"
echo " deploy commit: ${want:-<unknown>}"
return 1
fi
echo " snapshot: $snapshot (commit ${have:0:12})"
generations="$snapshot/generations.before"
if [ ! -s "$generations" ]; then
# Without a baseline we cannot tell which workloads the apply touched, so
# fall back to watching everything rather than silently skipping the check.
warn "no pre-apply baseline, verifying every workload in the cluster"
: >"$generations"
fi
while read -r kind ns name; do
[ -n "${kind:-}" ] && touched+=("$kind $ns $name")
done < <(changed_workloads "$generations")
log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)"
if [ "${#touched[@]}" -eq 0 ]; then
echo " nothing to verify"
return 0
fi
printf ' watching: %s\n' "${touched[@]/#/ }"
local failed_file="$snapshot/failed-workloads"
if ! verify_workloads "$failed_file" ${touched[@]+"${touched[@]}"}; then
echo
echo "ERROR: ${#touched[@]} workload(s) changed by this deploy, and these never became healthy:"
grep -v -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ - /'
echo
log "Rolling back to the previous revision"
if rollback_workloads "$failed_file"; then
echo
echo "Rolled back successfully. The cluster is back on the pre-deploy revision."
echo "Nothing else was reverted: Git holds desired state only, so config changes, PVCs and"
echo "externally created resources from this commit are still in place. Review the failed"
echo "workload, then re-run the deploy (Actions -> deploy -> Run workflow)."
else
echo
echo "Rollback did NOT fully recover the cluster. Manual intervention required:"
grep -E '^(ROLLED_BACK|UNRECOVERED)=' "$failed_file" | sed 's/^/ /'
echo "Pre-apply snapshot: $snapshot"
fi
return 1
fi
}
# verify_compose_stack <compose-file>
# `docker compose up -d` exits 0 as soon as containers are created, so a stack can
# come back broken with a green pipeline. Require every long-running service to
# actually be running.
verify_compose_stack() {
local cf="$1"
local expected running missing=()
expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)"
running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)"
[ -n "$expected" ] || return 0
while IFS= read -r svc; do
[ -n "$svc" ] || continue
# restart:"no" services are allowed to have exited.
if ! printf '%s\n' "$running" | grep -qx "$svc" \
&& ! docker compose -f "$cf" config 2>/dev/null \
| grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then
missing+=("$svc")
fi
done <<<"$expected"
if [ "${#missing[@]}" -gt 0 ]; then
echo " NOT RUNNING: ${missing[*]}"
docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true
return 1
fi
echo " all ${#expected} service(s) running"
return 0
}
# The public hostname of every active service, one per line.
#
# Comments are stripped first, and deliberately so: a route that someone
# disabled by commenting it out is not a service to probe, and naio and xui are
# both still in the tree that way. A `#` only starts a comment when it is at the
# start of a line or after whitespace, so `s/#.*//` alone would also cut a
# legitimate value in half.
#
# Only the public names. The *.internal names are the same Traefik and the same
# Services, reached by a different label, so probing both would double the run
# to learn the same thing. The public name is also the one a user types.
smoke_hosts() {
local m k
# The backticks below are literal. They are Traefik's Host() delimiter, and the
# single quotes are precisely what keeps the shell from reading them as a
# command substitution, so the warning is the opposite of a real problem.
# shellcheck disable=SC2016
{
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
[ -f "$m" ] && cat "$m"
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
kubectl kustomize "$k" 2>/dev/null || true
done
} | sed -E 's/(^|[[:space:]])#.*$//' \
| grep -oE 'Host\(`[^`]+`\)' \
| sed -E 's/^Host\(`//; s/`\)$//' \
| grep -E '(^|\.)forust\.xyz$' \
| grep -v '\${' \
| sort -u
}
# stage_verify_k8s watches the rollout, which reports that the pods converged.
# It cannot tell a converged pod from a serving one: a route pointing at the
# wrong port, a Service selector that matches nothing the app listens on, a 500
# from the app itself, an OOMKill loop that still counts as Available for long
# enough to pass. All of those are green at the rollout level.
#
# So ask the thing users ask. Any HTTP response proves Traefik matched the
# host, the Service resolved to a pod and the pod answered -- a 302 to a login
# or a 404 from a path the service does not serve still means the chain is
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
# the service is not serving, and only those fail the run.
stage_smoke() {
cd "$REPO"
select_manifests >/dev/null
local -a hosts=()
# Not named failed: an array of that name already exists in restart_stale_images
# above, and a scalar shadowing an array is a trap rather than a shadow.
local h code rc bad=0
while IFS= read -r h; do
[ -n "$h" ] && hosts+=("$h")
done < <(smoke_hosts)
if [ "${#hosts[@]}" -eq 0 ]; then
# Nothing to probe means the extraction broke, not that the cluster is empty.
echo "ERROR: no public hostnames found in active manifests, refusing to report success"
return 1
fi
log "Probing ${#hosts[@]} public route(s)"
for h in "${hosts[@]}"; do
code="$(curl -sS -o /dev/null --max-time 20 -w '%{http_code}' "https://$h/" 2>/dev/null)" && rc=0 || rc=$?
if [ "$rc" -ne 0 ]; then
echo " UNREACHABLE $h (curl exit $rc)"
bad=1
continue
fi
# A glob, not a string compare. `case` on the leading digit is the only one
# of these that survives a three-digit code, and the obvious expansion to
# try first -- ${code%%[0-9]*} -- is empty for every input, so it silently
# reports a 500 as healthy.
case "$code" in
5*)
echo " SERVER ERROR $h $code"
bad=1
;;
000)
# curl exited 0 and still no status, so nothing on the far end replied.
# Not a pass, whatever the transport thought.
echo " NO RESPONSE $h"
bad=1
;;
*)
echo " ok $h $code"
;;
esac
done
if [ "$bad" -ne 0 ]; then
echo "ERROR: at least one active service is not serving over its public route"
return 1
fi
echo "all ${#hosts[@]} route(s) answered"
}
stage_apply_compose() {
@@ -902,23 +225,6 @@ stage_apply_compose() {
fi
docker compose -f "$cf" up -d --pull always --remove-orphans
done
local -a broken=()
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " verifying: $cf"
if ! verify_compose_stack "$cf"; then
broken+=("$cf")
fi
done
if [ "${#broken[@]}" -gt 0 ]; then
echo
echo "ERROR: ${#broken[@]} compose stack(s) did not come up:"
printf ' - %s\n' "${broken[@]}"
echo "Compose stacks are not rolled back automatically: their images use mutable"
echo "':latest' tags, so there is no previous version to return to. Check the logs"
echo "above, then re-run the deploy once the cause is fixed."
return 1
fi
}
run_stage() {
@@ -926,8 +232,6 @@ run_stage() {
preflight) stage_preflight ;;
validate) stage_validate ;;
apply-k8s) stage_apply_k8s ;;
verify-k8s) stage_verify_k8s ;;
smoke) stage_smoke ;;
apply-compose) stage_apply_compose ;;
*)
echo "ERROR: unknown stage: $1"
+3 -83
View File
@@ -1,26 +1,13 @@
name: deploy
on:
# Deploy only what CI already validated. workflow_run is used instead of
# workflow_dispatch so a red lint/validate run can never reach the cluster.
workflow_run:
workflows: [ci]
types: [completed]
push:
branches:
- main
workflow_dispatch:
# The deploy jobs read the tree, then reach the cluster over SSH with the
# deploy key. The Actions token itself is not part of that path, so it gets
# read-only contents and no more.
permissions:
contents: read
concurrency:
group: deploy-main
# Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and
# takes the verify job down with it, so a superseded deploy would leave the
# cluster half-applied and unchecked — the exact failure the verify job exists
# to catch. kubectl apply and docker compose up are both idempotent, so letting
# the older run finish and then deploying the newer commit costs little.
cancel-in-progress: false
env:
@@ -30,21 +17,10 @@ env:
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
# workflow_run's own GITHUB_SHA points at the branch head, not at the commit the
# finished ci run checked. Pin the exact validated commit instead, so a push
# landing mid-deploy cannot make the workstation deploy something else. Also
# what the verify job checks the snapshot against. Empty for workflow_dispatch,
# which falls back to the current origin/main.
DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }}
jobs:
preflight:
if: >-
github.event_name != 'workflow_run' ||
(github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.head_branch == 'main')
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -58,7 +34,6 @@ jobs:
validate:
needs: [preflight]
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -72,10 +47,6 @@ jobs:
apply-k8s:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
# Apply only, no verification, so this is just the work itself: snapshot,
# then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
# apply loop. Verification has its own job and its own budget.
timeout-minutes: 45
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -89,7 +60,6 @@ jobs:
apply-compose:
needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
@@ -99,53 +69,3 @@ jobs:
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh apply-compose
# Watches the workloads this deploy changed and rolls back the ones that never
# became healthy. Runs even when the apply jobs failed, timed out or were
# cancelled — that is the whole point of splitting it out. `always()` is what
# lets it start after a failed dependency; the needs on apply-compose are a
# barrier, so verification begins only once both applies are done.
verify-k8s:
needs: [apply-k8s, apply-compose]
if: >-
always() &&
needs.apply-k8s.result != 'skipped' &&
needs.apply-compose.result != 'skipped'
runs-on: [self-hosted, linux, arch, homelab, prod]
# ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
timeout-minutes: 30
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Verify workloads and roll back on failure
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh verify-k8s
# Asks the public route of every active service whether it is actually
# serving, which the rollout check above structurally cannot: a pod can
# converge and still be crash-looping, or be listening on a port no Service
# points at, or answer 500.
#
# `always()` for the same reason verify-k8s has it, and it runs after that job
# specifically because a rollback is when a route most needs re-checking. The
# needs is a barrier, not a filter: whether verify-k8s passed, failed or was
# cancelled, the probes are what say whether the cluster is serving, and
# suppressing them on a rollback would hide the one run where the answer
# matters most.
smoke:
needs: [verify-k8s]
if: always() && needs.verify-k8s.result != 'skipped'
runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Probe the public route of every active service
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/ssh-run.sh smoke
-233
View File
@@ -1,233 +0,0 @@
#!/usr/bin/env bash
# Installs the pinned CI tools into "$TOOLS_DIR/bin" and echoes that directory
# on stdout, so callers can do:
#
# export PATH="$(bash .gitea/workflows/install-ci-tools.sh kubeconform shellcheck):$PATH"
#
# Versions come from tool-versions.env next to this script and are kept fresh by
# Renovate. Re-running is cheap: an already-installed tool at the pinned version
# is left alone.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=tool-versions.env
. "$here/tool-versions.env"
TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}"
BIN_DIR="$TOOLS_DIR/bin"
mkdir -p "$BIN_DIR"
arch="$(uname -m)"
# Upstream projects disagree on arch spelling: kubeconform and actionlint use
# Go names (amd64/arm64), shellcheck uses uname names (x86_64/aarch64), node
# uses neither (x64/arm64), and hadolint mixes the two in a single release
# (x86_64 but arm64).
case "$arch" in
x86_64 | amd64)
goarch=amd64
sharch=x86_64
nodearch=x64
hadolintarch=x86_64
;;
aarch64 | arm64)
goarch=arm64
sharch=aarch64
nodearch=arm64
hadolintarch=arm64
;;
*)
echo "install-ci-tools: unsupported architecture: $arch" >&2
exit 1
;;
esac
fetch() {
# fetch <url> <dest>
if command -v curl >/dev/null 2>&1; then
curl -sSLf --retry 3 -o "$2" "$1"
elif command -v wget >/dev/null 2>&1; then
wget -q -O "$2" "$1"
else
echo "install-ci-tools: neither curl nor wget is available" >&2
exit 1
fi
}
# installed_version <command>
# Prints the version of an already-installed tool, or nothing. Each tool spells
# its version flag differently, hence the case.
installed_version() {
local out
case "$1" in
kubeconform) out="$("$1" -v 2>/dev/null | head -1 || true)" ;;
*) out="$("$1" --version 2>/dev/null | head -1 || true)" ;;
esac
printf '%s' "$out"
}
# at_version <command> <expected>
at_version() {
case "$(installed_version "$1")" in
*"$2"*) return 0 ;;
*) return 1 ;;
esac
}
install_kubeconform() {
if at_version kubeconform "v${KUBECONFORM_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
fetch "https://github.com/yannh/kubeconform/releases/download/v${KUBECONFORM_VERSION}/kubeconform-linux-${goarch}.tar.gz" \
"$tmp/kubeconform.tar.gz"
tar -xzf "$tmp/kubeconform.tar.gz" -C "$tmp" kubeconform
install -m 0755 "$tmp/kubeconform" "$BIN_DIR/kubeconform"
rm -rf "$tmp"
}
install_shellcheck() {
if at_version shellcheck "${SHELLCHECK_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
fetch "https://github.com/koalaman/shellcheck/releases/download/v${SHELLCHECK_VERSION}/shellcheck-v${SHELLCHECK_VERSION}.linux.${sharch}.tar.xz" \
"$tmp/shellcheck.tar.xz"
tar -xJf "$tmp/shellcheck.tar.xz" -C "$tmp" --strip-components=1 "shellcheck-v${SHELLCHECK_VERSION}/shellcheck"
install -m 0755 "$tmp/shellcheck" "$BIN_DIR/shellcheck"
rm -rf "$tmp"
}
install_uv() {
if at_version uv "${UV_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
# uv release tags carry no leading v, unlike every other tool installed here.
fetch "https://github.com/astral-sh/uv/releases/download/${UV_VERSION}/uv-${sharch}-unknown-linux-gnu.tar.gz" \
"$tmp/uv.tar.gz"
tar -xzf "$tmp/uv.tar.gz" -C "$tmp" --strip-components=1 "uv-${sharch}-unknown-linux-gnu/uv"
install -m 0755 "$tmp/uv" "$BIN_DIR/uv"
rm -rf "$tmp"
}
install_hadolint() {
if at_version hadolint "${HADOLINT_VERSION}"; then
return 0
fi
# A bare binary, no archive: hadolint ships one file per platform.
fetch "https://github.com/hadolint/hadolint/releases/download/v${HADOLINT_VERSION}/hadolint-linux-${hadolintarch}" \
"$BIN_DIR/hadolint"
chmod 0755 "$BIN_DIR/hadolint"
}
# ruff and yamllint both come from PyPI as wheels, which uv unpacks for us.
install_uv_tool() {
# <package> <pinned version>
if at_version "$1" "$2"; then
return 0
fi
install_uv
UV_TOOL_BIN_DIR="$BIN_DIR" uv tool install --force "$1==$2" >/dev/null
}
install_ruff() {
install_uv_tool ruff "${RUFF_VERSION}"
}
install_yamllint() {
install_uv_tool yamllint "${YAMLLINT_VERSION}"
}
install_pip_audit() {
install_uv_tool pip-audit "${PIP_AUDIT_VERSION}"
}
install_prettier() {
if at_version prettier "${PRETTIER_VERSION}"; then
return 0
fi
# Not a standalone binary: prettier's entry point requires ../package.json
# relative to its own real path, so the package directory has to survive
# next to it. Hence a versioned directory plus a relative symlink, rather
# than copying the one file out as the other installers do.
local dir="$BIN_DIR/prettier-${PRETTIER_VERSION}"
if [ ! -f "$dir/package/package.json" ]; then
rm -rf "$dir"
mkdir -p "$dir"
fetch "https://registry.npmjs.org/prettier/-/prettier-${PRETTIER_VERSION}.tgz" "$dir/prettier.tgz"
tar -xzf "$dir/prettier.tgz" -C "$dir"
rm -f "$dir/prettier.tgz"
# npm strips the exec bit from bin/ on the way into the tarball.
chmod 0755 "$dir/package/bin/prettier.cjs"
fi
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
ln -sfn "prettier-${PRETTIER_VERSION}/package/bin/prettier.cjs" "$BIN_DIR/prettier"
}
install_node() {
# npm gets checked by running it, not by looking it up: what matters is that
# it answers, so a stub, a half-removed Arch package or a name that resolves
# to something broken all have to read as "not installed". The runner's npm
# is a symlink into /usr/lib/node_modules/npm, which is exactly the kind of
# thing that disappears between runs.
if at_version node "v${NODE_VERSION}" && [ -n "$(installed_version npm)" ]; then
return 0
fi
# Same shape as prettier above: the tarball's bin/npm and bin/npx are links
# into lib/node_modules, so the whole tree has to survive next to them.
local dir="$BIN_DIR/node-${NODE_VERSION}"
if [ ! -x "$dir/bin/node" ]; then
rm -rf "$dir"
mkdir -p "$dir"
fetch "https://nodejs.org/dist/v${NODE_VERSION}/node-v${NODE_VERSION}-linux-${nodearch}.tar.xz" \
"$dir/node.tar.xz"
tar -xJf "$dir/node.tar.xz" -C "$dir" --strip-components=1 "node-v${NODE_VERSION}-linux-${nodearch}"
rm -f "$dir/node.tar.xz"
fi
# Relative, so the whole tree stays valid if TOOLS_DIR is relocated.
for bin in node npm npx; do
ln -sfn "node-${NODE_VERSION}/bin/${bin}" "$BIN_DIR/${bin}"
done
}
install_actionlint() {
if at_version actionlint "${ACTIONLINT_VERSION}"; then
return 0
fi
local tmp
tmp="$(mktemp -d)"
fetch "https://github.com/rhysd/actionlint/releases/download/v${ACTIONLINT_VERSION}/actionlint_${ACTIONLINT_VERSION}_linux_${goarch}.tar.gz" \
"$tmp/actionlint.tar.gz"
tar -xzf "$tmp/actionlint.tar.gz" -C "$tmp" actionlint
install -m 0755 "$tmp/actionlint" "$BIN_DIR/actionlint"
rm -rf "$tmp"
}
wanted=("$@")
if [ "${#wanted[@]}" -eq 0 ]; then
wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint)
fi
for tool in "${wanted[@]}"; do
case "$tool" in
kubeconform) install_kubeconform ;;
shellcheck) install_shellcheck ;;
actionlint) install_actionlint ;;
prettier) install_prettier ;;
ruff) install_ruff ;;
yamllint) install_yamllint ;;
pip-audit) install_pip_audit ;;
hadolint) install_hadolint ;;
node) install_node ;;
uv) install_uv ;;
*)
echo "install-ci-tools: unknown tool: $tool" >&2
exit 1
;;
esac
done
printf '%s\n' "$BIN_DIR"
+18 -42
View File
@@ -7,58 +7,33 @@ on:
- main
workflow_dispatch:
permissions:
contents: read
jobs:
validate-renovate:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 20
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag,
# so the same version that runs in the cluster is the one validated here.
- name: Resolve the deployed Renovate image
id: image
- name: Validate Renovate Compose draft
shell: bash
run: |
set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
exit 1
fi
echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT"
trap 'rm -f renovate/.env' EXIT
printf '%s\n' \
'RENOVATE_ENDPOINT=https://gitea.example/api/v1' \
'RENOVATE_TOKEN=test-token' \
'RENOVATE_REPOSITORIES=forust/homelab' \
> renovate/.env
docker compose -f renovate/renovate-compose.yaml config --quiet
- name: Validate Renovate repository config
- name: Validate Kubernetes manifests
shell: bash
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate:/opt/renovate:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \
renovate-config-validator /opt/renovate/renovate.json
# The CronJob cannot read the repository, so renovate/k8s/configmap.yaml
# carries an inlined copy of the config. Fail if it no longer matches.
- name: Check the generated Renovate ConfigMap
shell: bash
run: |
set -euo pipefail
./.gitea/workflows/sync-renovate-configmap.sh --check
- name: Validate Renovate Kubernetes manifests
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)"
export PATH="$tools_dir:$PATH"
kubeconform \
-v "$PWD:/work" \
-w /work \
ghcr.io/yannh/kubeconform:latest \
-strict \
-ignore-missing-schemas \
-summary \
@@ -66,11 +41,12 @@ jobs:
renovate/k8s/configmap.yaml \
renovate/k8s/cronjob.yaml
- name: Validate Renovate Compose file
- name: Validate Renovate repository config
shell: bash
run: |
set -euo pipefail
source .gitea/workflows/compose-lint.sh
mapfile -t safe_flags < <(compose_safe_flags)
validate_compose_file renovate/renovate-compose.yaml \
${safe_flags[@]+"${safe_flags[@]}"}
docker run --rm \
-v "$PWD:/work" \
-w /work \
renovate/renovate:44.103.0 \
renovate-config-validator renovate.json
+6 -29
View File
@@ -21,11 +21,6 @@ on:
default: false
type: boolean
# Renovate writes through its own bot PAT, passed in as RENOVATE_TOKEN, so the
# Actions token is only ever used to read the checkout.
permissions:
contents: read
concurrency:
group: renovate-run
cancel-in-progress: false
@@ -33,36 +28,18 @@ concurrency:
jobs:
run-renovate:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 60
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# renovate/k8s/cronjob.yaml is the single source of truth for the image tag.
# Reading it here means this workflow validates and runs the exact version
# that is deployed, instead of a copy that silently goes stale.
- name: Resolve the deployed Renovate image
id: image
shell: bash
run: |
set -euo pipefail
image="$(sed -n 's|.*image:[[:space:]]*\(renovate/renovate:[^[:space:]]*\).*|\1|p' \
renovate/k8s/cronjob.yaml | head -1)"
if [ -z "$image" ]; then
echo "::error::no renovate/renovate image found in renovate/k8s/cronjob.yaml"
exit 1
fi
echo "using $image"
echo "image=$image" >> "$GITHUB_OUTPUT"
- name: Validate Renovate config
shell: bash
run: |
set -euo pipefail
docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
"${{ steps.image.outputs.image }}" \
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
renovate/renovate:44.103.0 \
renovate-config-validator
- name: Run Renovate
@@ -79,14 +56,14 @@ jobs:
: "${RENOVATE_TOKEN:?missing RENOVATE_TOKEN secret — add a renovate-bot PAT in repo/org Actions secrets}"
docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-v "$PWD/renovate/config.js:/opt/renovate/config.js:ro" \
-e RENOVATE_PLATFORM=gitea \
-e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
-e RENOVATE_DRY_RUN="${RENOVATE_DRY_RUN:-}" \
-e RENOVATE_CONFIG_FILE=/opt/renovate/renovate.json \
-e RENOVATE_CONFIG_FILE=/opt/renovate/config.js \
-e RENOVATE_BASE_DIR=/tmp/renovate \
-e LOG_LEVEL="${LOG_LEVEL:-info}" \
"${{ steps.image.outputs.image }}"
renovate/renovate:44.103.0
+3 -8
View File
@@ -11,20 +11,15 @@ deploy_port="${DEPLOY_PORT:-22}"
deploy_path="${DEPLOY_PATH:-/srv/homelab}"
deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)"
# The private key is written to a per-run directory that is removed on exit, so a
# failed or cancelled job cannot leave deploy credentials in the runner's temp
# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel.
key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")"
trap 'rm -rf "$key_dir"' EXIT INT TERM
ssh_key="$key_dir/deploy_key"
ssh_key="$RUNNER_TEMP/deploy_key"
mkdir -p "$RUNNER_TEMP"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key"
ssh -i "$ssh_key" -p "$deploy_port" \
-o BatchMode=yes -o StrictHostKeyChecking=accept-new \
"${DEPLOY_USER}@${DEPLOY_HOST}" \
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-} STAGE=$1 bash -se" <<'EOF'
"REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} STAGE=$1 bash -se" <<'EOF'
source "$REPO/.gitea/workflows/deploy-lib.sh"
run_stage "$STAGE"
EOF
@@ -1,55 +0,0 @@
#!/usr/bin/env bash
# Regenerates renovate/k8s/configmap.yaml from renovate/renovate.json.
#
# renovate/renovate.json is the single source of truth: the CronJob, the Compose
# file and the renovate-run workflow all mount that exact file. A ConfigMap cannot
# read a file from the repository, so the same bytes are inlined here as a literal
# block. This script keeps the copy honest:
#
# .gitea/workflows/sync-renovate-configmap.sh # rewrite in place
# .gitea/workflows/sync-renovate-configmap.sh --check # fail if out of date
#
# renovate-ci runs the --check form on every PR and push, so a config change that
# forgets to regenerate the ConfigMap cannot be merged.
set -euo pipefail
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
repo="$(git -C "$here" rev-parse --show-toplevel)"
src="$repo/renovate/renovate.json"
dst="$repo/renovate/k8s/configmap.yaml"
[ -f "$src" ] || {
echo "missing $src" >&2
exit 1
}
render() {
cat <<'HEADER'
# GENERATED FILE - do not edit by hand.
# Source: renovate/renovate.json
# Regenerate: .gitea/workflows/sync-renovate-configmap.sh
# Verify: .gitea/workflows/sync-renovate-configmap.sh --check
apiVersion: v1
kind: ConfigMap
metadata:
name: renovate-config
namespace: renovate
data:
renovate.json: |
HEADER
sed 's/^/ /' "$src"
}
if [ "${1:-}" = "--check" ]; then
if ! diff -u "$dst" <(render) >/dev/null 2>&1; then
echo "ERROR: $dst is out of sync with renovate/renovate.json"
echo "Run: .gitea/workflows/sync-renovate-configmap.sh"
diff -u "$dst" <(render) || true
exit 1
fi
echo "renovate/k8s/configmap.yaml is in sync with renovate/renovate.json"
exit 0
fi
render >"$dst"
echo "wrote $dst"
-33
View File
@@ -1,33 +0,0 @@
# Pinned versions of the CI tools installed by install-ci-tools.sh.
# Renovate keeps these up to date (see customManagers in renovate/renovate.json).
#
# Every version here except NODE_VERSION matches what was already installed on
# the runner, so pinning them changes what CI does not at all. It changes what
# CI does when the runner is rebuilt with something else: today
# install-ci-tools.sh finds the pinned version already on PATH and installs
# nothing, and a runner that drifts gets the pinned one installed over it.
#
# The renovate image version is NOT pinned here: renovate/k8s/cronjob.yaml is the
# single source of truth and the workflows read the tag from it, so there is
# nothing to drift.
ACTIONLINT_VERSION="1.7.7"
SHELLCHECK_VERSION="0.11.0"
KUBECONFORM_VERSION="0.8.0"
PRETTIER_VERSION="3.8.1"
RUFF_VERSION="0.16.8"
YAMLLINT_VERSION="1.38.0"
HADOLINT_VERSION="2.14.0"
# pip-audit reads the advisory database over the network, so a floating version
# would make the same commit report different things on different days. Pin it
# like the rest: the advisories themselves are the moving part, not the tool.
PIP_AUDIT_VERSION="2.10.1"
# uv builds the throwaway venv the pytest job runs in, and unpacks the PyPI
# wheels for ruff, yamllint and pip-audit.
UV_VERSION="0.12.17"
# node runs `npm ci` for the frontend tests and the npm audit, and it is the one
# pin here that does NOT come from the runner: the runner's system node is a
# rolling Arch package (it was node 26 with no npm at all when this was pinned),
# and the panel image is node:22-alpine. Pinned to the image's major on purpose,
# so the tree that gets tested is the tree that gets built. Renovate keeps this
# in step with the Dockerfile's node: tag via the "node runtime" group.
NODE_VERSION="22.23.3"
-2
View File
@@ -96,8 +96,6 @@ replacements.txt
# Temp files
edu_master/temp/
temp/*
# Local-only tooling scratch space (pinned CI tools, verification scripts)
tmp/
# Environment
.env
+1 -1
View File
@@ -54,7 +54,7 @@ services:
- "traefik.http.routers.bentopdf.tls.certresolver=letsencrypt"
- "traefik.http.routers.bentopdf.tls=true"
# Local router
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.workstation.internal`)"
- "traefik.http.routers.bentopdf-local.rule=Host(`pdf.wokstation.internal`)"
- "traefik.http.routers.bentopdf-local.entrypoints=websecure"
- "traefik.http.routers.bentopdf-local.tls=true"
# Dev router
-10
View File
@@ -12,13 +12,3 @@ spec:
CrowdsecLapiScheme: http
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
# LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on
# `GET /v1/decisions?ip=...&banned=true` before the request reaches
# the backend, and fails CLOSED (403) if the lookup exceeds the
# timeout. Unset, the fork defaults to 10s, which is an eternity for
# a request path: a single slow LAPI (idle 1.3-7.4s here) turned
# every request into a 10s hang and then a self-inflicted 403.
# 2s keeps the fail-closed path fast and bounded; with the LAPI
# resourced properly (see crowdsec-values.yaml) the lookup is
# sub-100ms and this budget is never hit.
CrowdsecLapiTimeout: "2s"
+4 -28
View File
@@ -68,23 +68,6 @@ config:
reason: "Home dynamic IP"
expression:
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
# The hairpin-NAT address of the router (192.168.88.1) is what the
# Gitea Actions runner presents to Traefik - it is NOT the home
# dynamic IP, so the whitelist above did not cover it. During a
# deploy the runner POSTs to the Actions API many times a second;
# a single 403 storm was enough to earn it a 4h ban and break every
# later job. Whitelisting the whole LAN also covers phones and
# tablets browsing over 192.168.88.0/24.
lan.yaml: |
name: forust/lan
description: "Whitelist local network"
whitelist:
reason: "Local network"
cidr:
- "127.0.0.0/8"
- "10.0.0.0/8"
- "172.16.0.0/12"
- "192.168.0.0/16"
lapi:
env:
@@ -107,20 +90,13 @@ lapi:
enabled: true
size: 1Gi
storageClassName: local-path-retain
# LAPI answers a blocking /v1/decisions lookup for EVERY bouncer-protected
# request (whole Traefik front door), so it is the hot path of the proxy.
# At 400m/500Mi it went CPU-throttled and idle lookups measured 1.3-7.4s,
# which pushed requests into the bouncer's fail-closed 403.
# Single replica on purpose: LAPI is stateful (BoltDB on the `data` PVC,
# credentials on the `config` PVC) - two replicas sharing those RWO
# volumes would corrupt the decision store. Scale up CPU, not replicas.
resources:
limits:
cpu: 1500m
memory: 1Gi
requests:
cpu: 250m
cpu: 400m
memory: 500Mi
requests:
cpu: 50m
memory: 150Mi
service:
type: ClusterIP
storeLAPICscliCredentialsInSecret: true
+1 -20
View File
@@ -30,10 +30,7 @@
# 3. ensure the static machine exists, recreating it with the
# Secret password if missing (agent retry loops reconnect
# on their own - same name + same password);
# 4. prune bouncer entries idle for 30d;
# 5. delete decisions from LePresidente/http-generic-403-bf, a hub
# scenario that bans an IP for 4h after 5 POST-403s in 10s and
# therefore bans us for our own bouncer's fail-closed 403s.
# 4. prune bouncer entries idle for 30d.
#
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
@@ -196,19 +193,3 @@ spec:
fi
echo "== 4. prune stale bouncers (no pull for 30d) =="
$LAPI_EXEC cscli bouncers prune -d 720h --force
echo "== 5. drop http-403-bf decisions (4h self-bans) =="
# `LePresidente/http-generic-403-bf` (hub item
# crowdsecurity/http-generic-bf v0.9) bans any source IP
# after 5 POSTs answered 403 within 10s, for 4h. That
# includes 403s this homelab generates ITSELF (any
# bouncer fail-closed, any app CSRF/rate-limit 403), and a
# 4h ban on the runner/home IP silently breaks deploys and
# browsing. The scenario cannot be removed per-scenario -
# it is baked into a hub item, and disabling the whole
# base-http-scenarios collection would drop ~40 useful
# detections. Instead we keep the detection and drop its
# decisions hourly; the LAN/home whitelists in
# crowdsec-values.yaml handle the legit sources, so this
# only ever hits real scanners (who are re-banned anyway).
$LAPI_EXEC cscli decisions delete \
--scenario LePresidente/http-generic-403-bf --all || true
+1 -20
View File
@@ -11,14 +11,9 @@ spec:
rules:
# No successful webinar check for 5m (~2-3 missed 2-min checks).
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
# The last_success > 0 guard is mandatory: checker.py initialises
# last_success to 0, so without it `time() - 0` equals the current epoch
# and humanizeDuration renders ~20722d on every pod restart. Keep the
# duration expression on the left so $value stays the real gap.
- alert: WebinarCheckerNoSuccessfulCheck
expr: |
((time() - webinar_check_last_success_timestamp_seconds) > 300)
and (webinar_check_last_success_timestamp_seconds > 0)
(time() - webinar_check_last_success_timestamp_seconds > 300)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 2m
labels:
@@ -27,20 +22,6 @@ spec:
summary: "Webinar checker has no successful check for 5m"
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
# Checks are running but none has ever succeeded since pod start.
# Split out from the rule above so a zeroed gauge never feeds
# humanizeDuration.
- alert: WebinarCheckerNeverSucceeded
expr: |
(webinar_check_last_success_timestamp_seconds == 0)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker has never completed a successful check"
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
- alert: WebinarCheckerConsecutiveFailures
expr: |
+2 -1
View File
@@ -31,7 +31,8 @@ spec:
echo "redis is ready"
containers:
- name: session-keeper
image: gcr.forust.xyz/forust/session-keeper:prod
image: gcr.forust.xyz/forust/session-keeper:latest
imagePullPolicy: Always
envFrom:
- secretRef:
name: edu-master-secrets
+2 -9
View File
@@ -45,19 +45,12 @@ spec:
echo "playwright ok"
containers:
- name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:prod
image: gcr.forust.xyz/forust/webinar-checker:latest
imagePullPolicy: Always
ports:
- name: metrics
containerPort: 8000
protocol: TCP
readinessProbe:
httpGet:
path: /health
port: metrics
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 12
initialDelaySeconds: 10
envFrom:
- secretRef:
name: edu-master-secrets
+1 -8
View File
@@ -27,14 +27,7 @@ spec:
spec:
containers:
- name: error-pages
image: gcr.forust.xyz/forust/error-pages:prod
image: gcr.forust.xyz/forust/error-pages:latest
ports:
- containerPort: 80
readinessProbe:
httpGet:
path: /404.html
port: 80
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
---
+3 -10
View File
@@ -15,18 +15,11 @@ spec:
services:
- name: gitea-service
port: 3000
# Registry route: NO crowdsec-bouncer.
# The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on
# *every* request. A deploy burst (runner Action API polls, `docker
# manifest inspect` per own image, containerd pulls, smoke probes) fires
# hundreds of parallel registry calls; LAPI saturation pushed the lookup
# past the plugin timeout, and the bouncer fail-closed with 403 - which
# containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod.
# This route only serves authenticated OCI traffic (registry tokens,
# basic-auth already handled by gitea) and scanners get nothing useful
# from /v2, so there is no bruteforce surface to protect here.
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services:
- name: gitea-service
port: 3000
+4 -16
View File
@@ -27,16 +27,10 @@ spec:
spec:
containers:
- name: forust-homepage
image: gcr.forust.xyz/forust/forust-homepage:prod
image: gcr.forust.xyz/forust/forust-homepage:latest
imagePullPolicy: Always
ports:
- containerPort: 80
readinessProbe:
httpGet:
path: /
port: 80
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
resources:
requests:
memory: "10Mi"
@@ -74,16 +68,10 @@ spec:
spec:
containers:
- name: xdfnx-homepage
image: gcr.forust.xyz/forust/xdfnx-homepage:prod
image: gcr.forust.xyz/forust/xdfnx-homepage:latest
imagePullPolicy: Always
ports:
- containerPort: 80
readinessProbe:
httpGet:
path: /
port: 80
periodSeconds: 10
timeoutSeconds: 2
failureThreshold: 3
resources:
requests:
memory: "10Mi"
+1 -1
View File
@@ -1,6 +1,6 @@
services:
metube:
image: ghcr.io/alexta69/metube:2026.09.27
image: ghcr.io/alexta69/metube:2026.09.25
container_name: metube
restart: unless-stopped
# ports:
+1 -1
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: metube
image: ghcr.io/alexta69/metube:2026.09.27
image: ghcr.io/alexta69/metube:2026.09.25
envFrom:
- configMapRef:
name: metube-config
File renamed without changes.
+1 -1
View File
@@ -87,7 +87,7 @@ services:
- proxy
dashboard:
image: netbirdio/dashboard:v2.93.0
image: netbirdio/dashboard:v2.90.10
container_name: netbird-dashboard
restart: unless-stopped
environment:
+1 -1
View File
@@ -137,7 +137,7 @@ spec:
spec:
containers:
- name: dashboard
image: netbirdio/dashboard:v2.93.0
image: netbirdio/dashboard:v2.90.10
envFrom:
- configMapRef:
name: netbird-config
View File
File renamed without changes.
+31 -31
View File
@@ -1,48 +1,48 @@
import os
def _csv(name, default=''):
return [item.strip() for item in os.environ.get(name, default).split(',') if item.strip()]
def _csv(name, default=""):
return [item.strip() for item in os.environ.get(name, default).split(",") if item.strip()]
ALLOWED_HOSTS = _csv('ALLOWED_HOSTS', 'localhost,127.0.0.1,[::1]')
CSRF_TRUSTED_ORIGINS = _csv('CSRF_TRUSTED_ORIGINS')
ALLOWED_HOSTS = _csv("ALLOWED_HOSTS", "localhost,127.0.0.1,[::1]")
CSRF_TRUSTED_ORIGINS = _csv("CSRF_TRUSTED_ORIGINS")
USE_X_FORWARDED_HOST = True
SECURE_PROXY_SSL_HEADER = ('HTTP_X_FORWARDED_PROTO', 'https')
SECURE_PROXY_SSL_HEADER = ("HTTP_X_FORWARDED_PROTO", "https")
DATABASES = {
'default': {
'NAME': os.environ['DB_NAME'],
'USER': os.environ['DB_USER'],
'PASSWORD': os.environ['DB_PASSWORD'],
'HOST': os.environ['DB_HOST'],
'PORT': os.environ.get('DB_PORT', '5432'),
'OPTIONS': {'sslmode': os.environ.get('DB_SSLMODE', 'disable')},
'CONN_MAX_AGE': int(os.environ.get('DB_CONN_MAX_AGE', '300')),
"default": {
"NAME": os.environ["DB_NAME"],
"USER": os.environ["DB_USER"],
"PASSWORD": os.environ["DB_PASSWORD"],
"HOST": os.environ["DB_HOST"],
"PORT": os.environ.get("DB_PORT", "5432"),
"OPTIONS": {"sslmode": os.environ.get("DB_SSLMODE", "disable")},
"CONN_MAX_AGE": int(os.environ.get("DB_CONN_MAX_AGE", "300")),
}
}
REDIS = {
'tasks': {
'HOST': os.environ['REDIS_HOST'],
'PORT': int(os.environ.get('REDIS_PORT', '6379')),
'PASSWORD': os.environ['REDIS_PASSWORD'],
'DATABASE': int(os.environ.get('REDIS_DATABASE', '0')),
'SSL': False,
"tasks": {
"HOST": os.environ["REDIS_HOST"],
"PORT": int(os.environ.get("REDIS_PORT", "6379")),
"PASSWORD": os.environ["REDIS_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_DATABASE", "0")),
"SSL": False,
},
'caching': {
'HOST': os.environ['REDIS_CACHE_HOST'],
'PORT': int(os.environ.get('REDIS_CACHE_PORT', '6379')),
'PASSWORD': os.environ['REDIS_CACHE_PASSWORD'],
'DATABASE': int(os.environ.get('REDIS_CACHE_DATABASE', '1')),
'SSL': False,
"caching": {
"HOST": os.environ["REDIS_CACHE_HOST"],
"PORT": int(os.environ.get("REDIS_CACHE_PORT", "6379")),
"PASSWORD": os.environ["REDIS_CACHE_PASSWORD"],
"DATABASE": int(os.environ.get("REDIS_CACHE_DATABASE", "1")),
"SSL": False,
},
}
SECRET_KEY = os.environ['SECRET_KEY']
API_TOKEN_PEPPERS = {1: os.environ['API_TOKEN_PEPPER_1']}
TIME_ZONE = os.environ.get('TIME_ZONE', 'UTC')
MEDIA_ROOT = '/opt/netbox/netbox/media'
REPORTS_ROOT = '/opt/netbox/netbox/reports'
SCRIPTS_ROOT = '/opt/netbox/netbox/scripts'
SECRET_KEY = os.environ["SECRET_KEY"]
API_TOKEN_PEPPERS = {1: os.environ["API_TOKEN_PEPPER_1"]}
TIME_ZONE = os.environ.get("TIME_ZONE", "UTC")
MEDIA_ROOT = "/opt/netbox/netbox/media"
REPORTS_ROOT = "/opt/netbox/netbox/reports"
SCRIPTS_ROOT = "/opt/netbox/netbox/scripts"
CENSUS_REPORTING_ENABLED = False
+18 -27
View File
@@ -56,42 +56,33 @@ spec:
startupProbe:
exec:
command:
- /usr/bin/curl
- --fail
- --silent
- --show-error
- --max-time
- "4"
- --header
- "Host: netbox.forust.xyz"
- http://127.0.0.1:8080/login/
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
failureThreshold: 90
periodSeconds: 10
readinessProbe:
exec:
command:
- /usr/bin/curl
- --fail
- --silent
- --show-error
- --max-time
- "4"
- --header
- "Host: netbox.forust.xyz"
- http://127.0.0.1:8080/login/
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
periodSeconds: 10
livenessProbe:
exec:
command:
- /usr/bin/curl
- --fail
- --silent
- --show-error
- --max-time
- "4"
- --header
- "Host: netbox.forust.xyz"
- http://127.0.0.1:8080/login/
- /opt/netbox/venv/bin/python
- -c
- >-
exec /usr/bin/curl --fail --silent --show-error --max-time 4
--header 'Host: netbox.forust.xyz'
http://127.0.0.1:8080/login/ >/dev/null
initialDelaySeconds: 30
periodSeconds: 30
resources:
-15
View File
@@ -69,18 +69,3 @@ alertmanager:
defaultRules:
disabled:
CPUThrottlingHigh: true
KubeControllerManagerDown: true
KubeSchedulerDown: true
KubeEtcdDown: true
KubeEtcdHighCommitDurations: true
# k0s runs controller-manager/scheduler/etcd internally, not as pods with
# component=kube-controller-manager/kube-scheduler/k8s-app=kube-etcd labels.
# Their Services get no endpoints, so the targets are permanently down.
# kube-proxy and kubelet have endpoints on k0s, keep them enabled.
kubeControllerManager:
enabled: false
kubeScheduler:
enabled: false
kubeEtcd:
enabled: false
-20
View File
@@ -1,20 +0,0 @@
# Pinned chart: stakater/reloader 2.2.17 (app v1.4.22).
# Deployed by the deploy workflow, namespace reloader.
# Restarts pods when a ConfigMap or Secret they consume changes. Opt-in per workload
# via the reloader.stakater.com/auto: "true" pod annotation; watchGlobally because
# the workloads that need it are spread across a few dozen namespaces.
reloader:
watchGlobally: true
deployment:
replicas: 1
# The chart defaults to no requests or limits, so the pod is evictable under node
# pressure and the restarts go with it.
resources:
requests:
cpu: "10m"
memory: "64Mi"
limits:
cpu: "100m"
memory: "128Mi"
+69
View File
@@ -0,0 +1,69 @@
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended"],
"enabledManagers": [
"dockerfile",
"docker-compose",
"kubernetes",
"helm-values",
"custom.regex"
],
"helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
},
"kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": [
"/^edu_master/k8s/playwright\\.yaml$/",
"/^edu_master/compose\\.yaml$/"
],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["/^edu_master/PLAYWRIGHT_VERSION$/"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
}
],
"packageRules": [
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
},
{
"description": "Require approval for major upgrades",
"matchUpdateTypes": ["major"],
"dependencyDashboardApproval": true,
"automerge": false
},
{
"description": "Group container patch updates",
"matchDatasources": ["docker"],
"matchUpdateTypes": ["patch"],
"groupName": "container patch updates"
}
]
}
+3 -32
View File
@@ -26,17 +26,15 @@ from Git and must be applied separately after every new cluster.
Run it immediately instead of waiting for the six-hour schedule.
Two options, both use the same `renovate/renovate.json`:
Two options, both use the same `renovate/config.js`:
```sh
kubectl create job --from=cronjob/renovate renovate-manual-$(date +%s) -n renovate
```
or the `renovate-run` Actions workflow (Actions tab → `renovate-run` →
Run workflow). It runs the same image as the CronJob on the self-hosted runner
via Docker — the tag is read out of `renovate/k8s/cronjob.yaml` at run time
rather than hardcoded, so the two cannot drift apart. Required Actions secrets
(repo or org settings):
Run workflow). It runs `renovate/renovate:44.103.0` on the self-hosted
runner via Docker. Required Actions secrets (repo or org settings):
- `RENOVATE_TOKEN` — renovate-bot PAT (repository + issue read/write).
- `RENOVATE_GITHUB_COM_TOKEN` — optional, for changelogs and GitHub rate limits.
@@ -66,33 +64,6 @@ docker compose -f renovate-compose.yaml run --rm renovate
The Compose file is intentionally named `renovate-compose.yaml`, so the
repository's automatic deployment discovery does not start it accidentally.
## Configuration
`renovate/renovate.json` is the single source of truth. The Compose file and the
`renovate-run` workflow mount that file directly.
A ConfigMap cannot read from the repository, so the CronJob needs the config
inlined. `renovate/k8s/configmap.yaml` is therefore a **generated** copy:
```sh
.gitea/workflows/sync-renovate-configmap.sh # regenerate after editing
.gitea/workflows/sync-renovate-configmap.sh --check # fail if out of date
```
The `renovate-ci` workflow runs the `--check` form on every PR and push, so a
config edit that forgets to regenerate the ConfigMap cannot be merged.
Beyond images, `customManagers` in the config track:
- Helm chart versions pinned in `.gitea/workflows/deploy-lib.sh`. The built-in
`helmv3` manager only reads `Chart.yaml` and `helm-values` only reads values
files, so neither sees a version written into a `helm upgrade` command —
these are declared as `custom.regex` managers against the `helm` datasource.
- CI linter versions in `.gitea/workflows/tool-versions.env`.
The Renovate image tag is deliberately _not_ in `tool-versions.env`:
`renovate/k8s/cronjob.yaml` owns it, and the workflows read it from there.
## How updates flow
Renovate scans both `compose.yaml` files and Kubernetes manifests, opens a
+44
View File
@@ -0,0 +1,44 @@
module.exports = {
platform: 'gitea',
endpoint: process.env.RENOVATE_ENDPOINT || 'https://gitea.forust.xyz/api/v1',
enabledManagers: ['docker-compose', 'kubernetes', 'helm-values'],
'helm-values': {
managerFilePatterns: ['/k8s/.+values\\.ya?ml$/'],
},
kubernetes: {
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
},
repositories: (process.env.RENOVATE_REPOSITORIES || '')
.split(',')
.map((repository) => repository.trim())
.filter(Boolean),
onboarding: false,
requireConfig: 'optional',
autodiscover: false,
dependencyDashboard: true,
prCreation: 'immediate',
labels: ['dependencies', 'automated'],
extends: [
'config:recommended',
':dependencyDashboard',
],
packageRules: [
{
description: 'Do not update private homelab images',
matchDatasources: ['docker'],
matchPackageNames: ['/gcr\\.forust\\.xyz\\/forust\\/.+/'],
enabled: false,
},
{
description: 'Keep major upgrades manual',
matchUpdateTypes: ['major'],
dependencyDashboardApproval: true,
automerge: false,
},
{
description: 'Group patch updates',
matchUpdateTypes: ['patch'],
groupName: 'container patch updates',
},
],
};
+36 -196
View File
@@ -1,211 +1,51 @@
# GENERATED FILE - do not edit by hand.
# Source: renovate/renovate.json
# Regenerate: .gitea/workflows/sync-renovate-configmap.sh
# Verify: .gitea/workflows/sync-renovate-configmap.sh --check
apiVersion: v1
kind: ConfigMap
metadata:
name: renovate-config
namespace: renovate
data:
renovate.json: |
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended", ":dependencyDashboard"],
"enabledManagers": ["dockerfile", "docker-compose", "kubernetes", "helm-values", "custom.regex"],
"onboarding": false,
"requireConfig": "optional",
"autodiscover": false,
"dependencyDashboard": true,
"prCreation": "immediate",
"labels": ["dependencies", "automated"],
"helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
config.js: |
module.exports = {
platform: 'gitea',
endpoint: process.env.RENOVATE_ENDPOINT || 'https://gitea.forust.xyz/api/v1',
enabledManagers: ['docker-compose', 'kubernetes', 'helm-values'],
'helm-values': {
managerFilePatterns: ['/k8s/.+values\\.ya?ml$/'],
},
"kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
kubernetes: {
managerFilePatterns: ['/k8s/.+\\.ya?ml$/'],
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["^edu_master/PLAYWRIGHT_VERSION$"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "kube-prometheus-stack chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|prometheus-community/kube-prometheus-stack\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "kube-prometheus-stack",
"registryUrlTemplate": "https://prometheus-community.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/loki chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/loki\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "loki",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/alloy chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/alloy\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "alloy",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "actionlint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)ACTIONLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "rhysd/actionlint"
},
{
"customType": "regex",
"description": "shellcheck version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)SHELLCHECK_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "koalaman/shellcheck"
},
{
"customType": "regex",
"description": "kubeconform version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)KUBECONFORM_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "yannh/kubeconform"
},
{
"customType": "regex",
"description": "uv version used to build the pytest venv",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)UV_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "astral-sh/uv"
},
{
"customType": "regex",
"description": "prettier version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PRETTIER_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "npm",
"depNameTemplate": "prettier"
},
{
"customType": "regex",
"description": "ruff version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)RUFF_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "ruff"
},
{
"customType": "regex",
"description": "pip-audit version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PIP_AUDIT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "pip-audit"
},
{
"customType": "regex",
"description": "yamllint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)YAMLLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "yamllint"
},
{
"customType": "regex",
"description": "hadolint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)HADOLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "hadolint/hadolint"
},
{
"customType": "regex",
"description": "node version the ci workflow runs npm with",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)NODE_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "node",
"depNameTemplate": "node"
},
{
"customType": "regex",
"description": "stakater/reloader chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|stakater/reloader\\|reloader\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "reloader",
"registryUrlTemplate": "https://stakater.github.io/stakater-charts"
}
repositories: (process.env.RENOVATE_REPOSITORIES || '')
.split(',')
.map((repository) => repository.trim())
.filter(Boolean),
onboarding: false,
requireConfig: 'optional',
autodiscover: false,
dependencyDashboard: true,
prCreation: 'immediate',
labels: ['dependencies', 'automated'],
extends: [
'config:recommended',
':dependencyDashboard',
],
"packageRules": [
packageRules: [
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
description: 'Do not update private homelab images',
matchDatasources: ['docker'],
matchPackageNames: ['/gcr\\.forust\\.xyz\\/forust\\/.+/'],
enabled: false,
},
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
description: 'Keep major upgrades manual',
matchUpdateTypes: ['major'],
dependencyDashboardApproval: true,
automerge: false,
},
{
"description": "playwright must not automerge - version skew breaks the WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
description: 'Group patch updates',
matchUpdateTypes: ['patch'],
groupName: 'container patch updates',
},
{
"description": "Renovate updates itself in lockstep across the CronJob and the Compose file",
"matchPackageNames": ["renovate/renovate"],
"groupName": "renovate self-update",
"automerge": false
},
{
"description": "CI runs npm on the node the panel image is built from - the NODE_VERSION pin in tool-versions.env and node:22-alpine in the Dockerfile are the same dependency and move as one",
"matchPackageNames": ["node"],
"groupName": "node runtime",
"groupSlug": "node",
"automerge": false
},
{
"description": "Helm chart bumps change PVC fields and admission behaviour, keep them reviewable",
"matchDatasources": ["helm"],
"automerge": false
},
{
"description": "Require approval for major upgrades",
"matchUpdateTypes": ["major"],
"dependencyDashboardApproval": true,
"automerge": false
},
{
"description": "Group container patch updates",
"matchDatasources": ["docker"],
"matchUpdateTypes": ["patch"],
"groupName": "container patch updates"
}
]
}
],
};
+4 -4
View File
@@ -16,7 +16,7 @@ spec:
restartPolicy: Never
containers:
- name: renovate
image: renovate/renovate:44.115.13
image: renovate/renovate:44.115.9
env:
- name: RENOVATE_PLATFORM
value: gitea
@@ -36,7 +36,7 @@ spec:
name: renovate-secrets
key: RENOVATE_REPOSITORIES
- name: RENOVATE_CONFIG_FILE
value: /opt/renovate/renovate.json
value: /opt/renovate/config.js
- name: RENOVATE_BASE_DIR
value: /tmp/renovate
- name: RENOVATE_GITHUB_COM_TOKEN
@@ -49,8 +49,8 @@ spec:
value: info
volumeMounts:
- name: config
mountPath: /opt/renovate/renovate.json
subPath: renovate.json
mountPath: /opt/renovate/config.js
subPath: config.js
readOnly: true
volumes:
- name: config
+3 -5
View File
@@ -1,8 +1,6 @@
services:
renovate:
# Kept in step with renovate/k8s/cronjob.yaml by the "renovate self-update"
# package rule in renovate/renovate.json.
image: renovate/renovate:44.115.9
image: renovate/renovate:44.103.0
container_name: renovate
restart: "no"
env_file:
@@ -12,8 +10,8 @@ services:
RENOVATE_ENDPOINT: ${RENOVATE_ENDPOINT:?set RENOVATE_ENDPOINT}
RENOVATE_TOKEN: ${RENOVATE_TOKEN:?set RENOVATE_TOKEN}
RENOVATE_REPOSITORIES: ${RENOVATE_REPOSITORIES:?set RENOVATE_REPOSITORIES}
RENOVATE_CONFIG_FILE: /opt/renovate/renovate.json
RENOVATE_CONFIG_FILE: /opt/renovate/config.js
RENOVATE_BASE_DIR: /tmp/renovate
LOG_LEVEL: ${LOG_LEVEL:-info}
volumes:
- ./renovate.json:/opt/renovate/renovate.json:ro
- ./config.js:/opt/renovate/config.js:ro
-200
View File
@@ -1,200 +0,0 @@
{
"$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended", ":dependencyDashboard"],
"enabledManagers": ["dockerfile", "docker-compose", "kubernetes", "helm-values", "custom.regex"],
"onboarding": false,
"requireConfig": "optional",
"autodiscover": false,
"dependencyDashboard": true,
"prCreation": "immediate",
"labels": ["dependencies", "automated"],
"helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
},
"kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
},
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"managerFilePatterns": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"managerFilePatterns": ["^edu_master/PLAYWRIGHT_VERSION$"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "kube-prometheus-stack chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|prometheus-community/kube-prometheus-stack\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "kube-prometheus-stack",
"registryUrlTemplate": "https://prometheus-community.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/loki chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/loki\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "loki",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "grafana/alloy chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|grafana/alloy\\|prometheus\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "alloy",
"registryUrlTemplate": "https://grafana.github.io/helm-charts"
},
{
"customType": "regex",
"description": "actionlint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)ACTIONLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "rhysd/actionlint"
},
{
"customType": "regex",
"description": "shellcheck version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)SHELLCHECK_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "koalaman/shellcheck"
},
{
"customType": "regex",
"description": "kubeconform version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)KUBECONFORM_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "yannh/kubeconform"
},
{
"customType": "regex",
"description": "uv version used to build the pytest venv",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)UV_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "astral-sh/uv"
},
{
"customType": "regex",
"description": "prettier version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PRETTIER_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "npm",
"depNameTemplate": "prettier"
},
{
"customType": "regex",
"description": "ruff version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)RUFF_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "ruff"
},
{
"customType": "regex",
"description": "pip-audit version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)PIP_AUDIT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "pip-audit"
},
{
"customType": "regex",
"description": "yamllint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)YAMLLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "pypi",
"depNameTemplate": "yamllint"
},
{
"customType": "regex",
"description": "hadolint version used by the ci workflow",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)HADOLINT_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "github-tags",
"depNameTemplate": "hadolint/hadolint"
},
{
"customType": "regex",
"description": "node version the ci workflow runs npm with",
"managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"],
"matchStrings": ["(?:^|\\n)NODE_VERSION=\"(?<currentValue>[0-9.]+)\""],
"datasourceTemplate": "node",
"depNameTemplate": "node"
},
{
"customType": "regex",
"description": "stakater/reloader chart version pinned in the deploy workflow",
"managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"],
"matchStrings": ["\\|stakater/reloader\\|reloader\\|(?<currentValue>[0-9.]+)\\|"],
"datasourceTemplate": "helm",
"depNameTemplate": "reloader",
"registryUrlTemplate": "https://stakater.github.io/stakater-charts"
}
],
"packageRules": [
{
"description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"],
"matchPackageNames": ["/gcr\\.forust\\.xyz\\/forust\\/.+/"],
"enabled": false
},
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks the WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{
"description": "Renovate updates itself in lockstep across the CronJob and the Compose file",
"matchPackageNames": ["renovate/renovate"],
"groupName": "renovate self-update",
"automerge": false
},
{
"description": "CI runs npm on the node the panel image is built from - the NODE_VERSION pin in tool-versions.env and node:22-alpine in the Dockerfile are the same dependency and move as one",
"matchPackageNames": ["node"],
"groupName": "node runtime",
"groupSlug": "node",
"automerge": false
},
{
"description": "Helm chart bumps change PVC fields and admission behaviour, keep them reviewable",
"matchDatasources": ["helm"],
"automerge": false
},
{
"description": "Require approval for major upgrades",
"matchUpdateTypes": ["major"],
"dependencyDashboardApproval": true,
"automerge": false
},
{
"description": "Group container patch updates",
"matchDatasources": ["docker"],
"matchUpdateTypes": ["patch"],
"groupName": "container patch updates"
}
]
}
+1 -1
View File
@@ -3,7 +3,7 @@
services:
core:
container_name: searxng-core
image: docker.io/searxng/searxng:${SEARXNG_VERSION:-2026.9.25-12f8b6515}
image: docker.io/searxng/searxng:${SEARXNG_VERSION:-2026.09.13-d4ce87c23}
restart: unless-stopped
# ports:
# - ${SEARXNG_PORT:-8080}
+1 -1
View File
@@ -27,7 +27,7 @@ spec:
spec:
containers:
- name: searxng
image: docker.io/searxng/searxng:2026.9.25-12f8b6515
image: docker.io/searxng/searxng:2026.09.13-d4ce87c23
envFrom:
- configMapRef:
name: searxng-config
+3 -4
View File
@@ -198,7 +198,8 @@ spec:
serviceAccountName: userbot-panel
containers:
- name: userbot-panel
image: gcr.forust.xyz/forust/userbot-panel:prod
image: gcr.forust.xyz/forust/userbot-panel:latest
imagePullPolicy: Always
ports:
- name: http
containerPort: 8080
@@ -208,9 +209,7 @@ spec:
- name: USERBOT_LEGACY_NAMESPACES
value: default
- name: USERBOT_IMAGE
# The build pushes main/prod only. The deploy resolves every gcr ref in
# this file, so a :latest here aborts the whole apply as unresolvable.
value: gcr.forust.xyz/forust/userbot:prod
value: gcr.forust.xyz/forust/userbot:latest
- name: USERBOT_STORAGE_CLASS
value: local-path-retain
- name: USERBOT_DOWNLOADS_HOST_PATH
+4 -2
View File
@@ -24,7 +24,8 @@ spec:
spec:
containers:
- name: forust-userbot
image: gcr.forust.xyz/forust/userbot:prod
image: gcr.forust.xyz/forust/userbot:latest
imagePullPolicy: Always
resources:
limits:
memory: "1.5Gi"
@@ -95,7 +96,8 @@ spec:
spec:
containers:
- name: anna-userbot
image: gcr.forust.xyz/forust/userbot:prod
image: gcr.forust.xyz/forust/userbot:latest
imagePullPolicy: Always
resources:
limits:
memory: "1.5Gi"
+18 -14
View File
@@ -51,11 +51,11 @@ class TelegramAuthService:
if len(self.flows) >= self.max_flows:
raise PanelError(
429,
'Too many pending authorization flows; try again later',
"Too many pending authorization flows; try again later",
)
flow_id = secrets.token_urlsafe(24)
telegram = Client(
f'auth-{flow_id}',
f"auth-{flow_id}",
api_id=account.api_id,
api_hash=account.api_hash,
in_memory=True,
@@ -117,7 +117,7 @@ class TelegramAuthService:
account: StringSessionStart,
) -> AuthorizedAccount:
telegram = Client(
f'validate-{secrets.token_urlsafe(12)}',
f"validate-{secrets.token_urlsafe(12)}",
api_id=account.api_id,
api_hash=account.api_hash,
session_string=account.session_string,
@@ -148,7 +148,7 @@ class TelegramAuthService:
async with self._lock:
flow = self.flows.get(flow_id)
if flow is None:
raise PanelError(410, 'Authorization flow expired; start again')
raise PanelError(410, "Authorization flow expired; start again")
return flow
async def _finish(self, flow: AuthFlow) -> AuthorizedAccount:
@@ -171,7 +171,11 @@ class TelegramAuthService:
async def _cleanup_expired(self) -> None:
now = datetime.now(UTC)
async with self._lock:
expired = [self.flows.pop(flow_id) for flow_id, flow in list(self.flows.items()) if flow.expires_at <= now]
expired = [
self.flows.pop(flow_id)
for flow_id, flow in list(self.flows.items())
if flow.expires_at <= now
]
if expired:
await asyncio.gather(
*(self._disconnect(flow.client) for flow in expired),
@@ -187,19 +191,19 @@ class TelegramAuthService:
@staticmethod
def _translate(exc: Exception, *, session: bool = False) -> PanelError:
if isinstance(exc, FloodWait):
return PanelError(429, f'Telegram rate limit; retry in {exc.value} seconds')
return PanelError(429, f"Telegram rate limit; retry in {exc.value} seconds")
if isinstance(exc, ApiIdInvalid):
return PanelError(422, 'Telegram API ID or API Hash is invalid')
return PanelError(422, "Telegram API ID or API Hash is invalid")
if isinstance(exc, PhoneNumberInvalid):
return PanelError(422, 'Phone number is invalid')
return PanelError(422, "Phone number is invalid")
if isinstance(exc, PhoneCodeInvalid):
return PanelError(422, 'Telegram code is invalid')
return PanelError(422, "Telegram code is invalid")
if isinstance(exc, PhoneCodeExpired):
return PanelError(410, 'Telegram code expired; start again')
return PanelError(410, "Telegram code expired; start again")
if isinstance(exc, PasswordHashInvalid):
return PanelError(422, '2FA password is invalid')
return PanelError(422, "2FA password is invalid")
if session and isinstance(exc, (Unauthorized, RPCError)):
return PanelError(422, 'StringSession is invalid or expired')
return PanelError(422, "StringSession is invalid or expired")
if isinstance(exc, RPCError):
return PanelError(422, 'Telegram rejected the authorization request')
return PanelError(503, 'Telegram authorization is unavailable')
return PanelError(422, "Telegram rejected the authorization request")
return PanelError(503, "Telegram authorization is unavailable")
+20 -18
View File
@@ -7,37 +7,39 @@ from pathlib import Path
@dataclass(frozen=True)
class Settings:
namespace: str = os.environ.get('USERBOT_NAMESPACE', 'userbot')
namespace: str = os.environ.get("USERBOT_NAMESPACE", "userbot")
legacy_namespaces: tuple[str, ...] = tuple(
value.strip() for value in os.environ.get('USERBOT_LEGACY_NAMESPACES', 'default').split(',') if value.strip()
value.strip()
for value in os.environ.get("USERBOT_LEGACY_NAMESPACES", "default").split(",")
if value.strip()
)
image: str = os.environ.get(
'USERBOT_IMAGE',
'gcr.forust.xyz/forust/userbot:latest',
"USERBOT_IMAGE",
"gcr.forust.xyz/forust/userbot:latest",
)
common_secret: str = os.environ.get(
'USERBOT_COMMON_SECRET',
'userbot-common-secrets',
"USERBOT_COMMON_SECRET",
"userbot-common-secrets",
)
common_config: str = os.environ.get(
'USERBOT_COMMON_CONFIG',
'userbot-common-config',
"USERBOT_COMMON_CONFIG",
"userbot-common-config",
)
storage_class: str = os.environ.get(
'USERBOT_STORAGE_CLASS',
'local-path-retain',
"USERBOT_STORAGE_CLASS",
"local-path-retain",
)
downloads_host_path: str = os.environ.get(
'USERBOT_DOWNLOADS_HOST_PATH',
'/srv/homelab/userbot/Downloads',
"USERBOT_DOWNLOADS_HOST_PATH",
"/srv/homelab/userbot/Downloads",
)
static_dir: Path = Path(os.environ.get('PANEL_STATIC_DIR', '/app/static'))
auth_ttl_seconds: int = int(os.environ.get('PANEL_AUTH_TTL_SECONDS', '600'))
default_storage: str = os.environ.get('USERBOT_DEFAULT_STORAGE', '1Gi')
default_cpu_limit: str = os.environ.get('USERBOT_DEFAULT_CPU_LIMIT', '300m')
static_dir: Path = Path(os.environ.get("PANEL_STATIC_DIR", "/app/static"))
auth_ttl_seconds: int = int(os.environ.get("PANEL_AUTH_TTL_SECONDS", "600"))
default_storage: str = os.environ.get("USERBOT_DEFAULT_STORAGE", "1Gi")
default_cpu_limit: str = os.environ.get("USERBOT_DEFAULT_CPU_LIMIT", "300m")
default_memory_limit: str = os.environ.get(
'USERBOT_DEFAULT_MEMORY_LIMIT',
'1536Mi',
"USERBOT_DEFAULT_MEMORY_LIMIT",
"1536Mi",
)
+117 -109
View File
@@ -14,18 +14,18 @@ from .models import AccountBase, InstanceSummary
logger = logging.getLogger(__name__)
MANAGED_LABEL = 'app.kubernetes.io/name=userbot'
INSTANCE_LABEL = 'app.kubernetes.io/instance'
MANAGED_BY_LABEL = 'app.kubernetes.io/managed-by'
DISPLAY_ANNOTATION = 'userbot.forust.xyz/display-name'
LEGACY_ANNOTATION = 'userbot.forust.xyz/legacy'
CREDENTIALS_ANNOTATION = 'userbot.forust.xyz/credentials-secret'
PVC_ANNOTATION = 'userbot.forust.xyz/pvc'
RESTART_ANNOTATION = 'userbot.forust.xyz/restarted-at'
MANAGED_LABEL = "app.kubernetes.io/name=userbot"
INSTANCE_LABEL = "app.kubernetes.io/instance"
MANAGED_BY_LABEL = "app.kubernetes.io/managed-by"
DISPLAY_ANNOTATION = "userbot.forust.xyz/display-name"
LEGACY_ANNOTATION = "userbot.forust.xyz/legacy"
CREDENTIALS_ANNOTATION = "userbot.forust.xyz/credentials-secret"
PVC_ANNOTATION = "userbot.forust.xyz/pvc"
RESTART_ANNOTATION = "userbot.forust.xyz/restarted-at"
def _selector(labels: dict[str, str] | None) -> str:
return ','.join(f'{key}={value}' for key, value in (labels or {}).items())
return ",".join(f"{key}={value}" for key, value in (labels or {}).items())
def _as_datetime(value: Any) -> datetime | None:
@@ -33,7 +33,7 @@ def _as_datetime(value: Any) -> datetime | None:
return None
if isinstance(value, datetime):
return value
return getattr(value, 'replace', lambda **_: None)(tzinfo=UTC)
return getattr(value, "replace", lambda **_: None)(tzinfo=UTC)
class KubernetesService:
@@ -55,7 +55,7 @@ class KubernetesService:
except Exception as exc:
raise PanelError(
503,
'No in-cluster or kubeconfig configuration is available',
"No in-cluster or kubeconfig configuration is available",
) from exc
self.core = core or client.CoreV1Api()
self.apps = apps or client.AppsV1Api()
@@ -76,14 +76,14 @@ class KubernetesService:
if exc.status == 404:
raise PanelError(
503,
'Userbot common Secret or ConfigMap is missing in the userbot namespace',
"Userbot common Secret or ConfigMap is missing in the userbot namespace",
) from exc
raise self._api_error(exc, 'Could not verify userbot prerequisites') from exc
raise self._api_error(exc, "Could not verify userbot prerequisites") from exc
def list_instances(
self,
query: str = '',
status: str = '',
query: str = "",
status: str = "",
) -> list[InstanceSummary]:
instances: list[InstanceSummary] = []
for namespace in (self.settings.namespace, *self.settings.legacy_namespaces):
@@ -93,7 +93,7 @@ class KubernetesService:
label_selector=MANAGED_LABEL,
).items
except ApiException as exc:
raise self._api_error(exc, f'Could not list Deployments in {namespace}') from exc
raise self._api_error(exc, f"Could not list Deployments in {namespace}") from exc
instances.extend(self._summarize(namespace, deployment) for deployment in deployments)
query = query.strip().lower()
@@ -103,7 +103,7 @@ class KubernetesService:
for item in instances
if query in item.instance_id.lower()
or query in item.display_name.lower()
or query in (item.pod or '').lower()
or query in (item.pod or "").lower()
]
if status:
instances = [item for item in instances if item.status == status]
@@ -116,9 +116,9 @@ class KubernetesService:
def assert_available(self, instance_id: str) -> None:
names = self._resource_names(instance_id)
checks = (
(self.apps.read_namespaced_deployment, names['deployment'], 'Deployment'),
(self.core.read_namespaced_secret, names['secret'], 'Secret'),
(self.core.read_namespaced_persistent_volume_claim, names['pvc'], 'PVC'),
(self.apps.read_namespaced_deployment, names["deployment"], "Deployment"),
(self.core.read_namespaced_secret, names["secret"], "Secret"),
(self.core.read_namespaced_persistent_volume_claim, names["pvc"], "PVC"),
)
for read, name, kind in checks:
try:
@@ -126,8 +126,8 @@ class KubernetesService:
except ApiException as exc:
if exc.status == 404:
continue
raise self._api_error(exc, f'Could not check {kind} {name}') from exc
raise PanelError(409, f'{kind} {name} already exists')
raise self._api_error(exc, f"Could not check {kind} {name}") from exc
raise PanelError(409, f"{kind} {name} already exists")
def provision(self, account: AccountBase, session_string: str) -> InstanceSummary:
with self._provision_lock:
@@ -139,25 +139,25 @@ class KubernetesService:
self.settings.namespace,
self._secret(account, session_string, names),
)
created.append(('secret', names['secret']))
created.append(("secret", names["secret"]))
self.core.create_namespaced_persistent_volume_claim(
self.settings.namespace,
self._pvc(account, names),
)
created.append(('pvc', names['pvc']))
created.append(("pvc", names["pvc"]))
self.apps.create_namespaced_deployment(
self.settings.namespace,
self._deployment(account, names),
)
created.append(('deployment', names['deployment']))
created.append(("deployment", names["deployment"]))
except ApiException as exc:
self._rollback(created)
if exc.status == 409:
raise PanelError(
409,
f'Instance {account.instance_id} already exists',
f"Instance {account.instance_id} already exists",
) from exc
raise self._api_error(exc, 'Could not create userbot instance') from exc
raise self._api_error(exc, "Could not create userbot instance") from exc
return self.get_instance(account.instance_id)
def scale(self, instance_id: str, replicas: int) -> InstanceSummary:
@@ -166,40 +166,40 @@ class KubernetesService:
self.apps.patch_namespaced_deployment_scale(
deployment.metadata.name,
namespace,
{'spec': {'replicas': replicas}},
{"spec": {"replicas": replicas}},
)
except ApiException as exc:
raise self._api_error(exc, 'Could not scale userbot instance') from exc
raise self._api_error(exc, "Could not scale userbot instance") from exc
return self.get_instance(instance_id)
def restart(self, instance_id: str) -> InstanceSummary:
namespace, deployment = self._find_deployment(instance_id)
if (deployment.spec.replicas or 0) == 0:
raise PanelError(409, 'Stopped instance cannot be restarted')
raise PanelError(409, "Stopped instance cannot be restarted")
timestamp = datetime.now(UTC).isoformat()
try:
self.apps.patch_namespaced_deployment(
deployment.metadata.name,
namespace,
{
'spec': {
'template': {
'metadata': {
'annotations': {RESTART_ANNOTATION: timestamp},
"spec": {
"template": {
"metadata": {
"annotations": {RESTART_ANNOTATION: timestamp},
}
}
}
},
)
except ApiException as exc:
raise self._api_error(exc, 'Could not restart userbot instance') from exc
raise self._api_error(exc, "Could not restart userbot instance") from exc
return self.get_instance(instance_id)
def logs(self, instance_id: str, tail: int = 250) -> str:
namespace, deployment = self._find_deployment(instance_id)
pods = self._pods_for_deployment(namespace, deployment)
if not pods:
raise PanelError(409, 'Userbot Pod is not running')
raise PanelError(409, "Userbot Pod is not running")
pod = pods[0]
container = deployment.spec.template.spec.containers[0].name
try:
@@ -211,22 +211,22 @@ class KubernetesService:
timestamps=True,
)
except ApiException as exc:
raise self._api_error(exc, 'Could not read userbot logs') from exc
raise self._api_error(exc, "Could not read userbot logs") from exc
def delete(self, instance_id: str, *, delete_data: bool) -> None:
namespace, deployment = self._find_deployment(instance_id)
annotations = deployment.metadata.annotations or {}
if annotations.get(LEGACY_ANNOTATION) == 'true' or namespace != self.settings.namespace:
raise PanelError(409, 'Legacy instances cannot be deleted from the panel')
if annotations.get(LEGACY_ANNOTATION) == "true" or namespace != self.settings.namespace:
raise PanelError(409, "Legacy instances cannot be deleted from the panel")
names = self._resource_names(instance_id)
secret_name = annotations.get(CREDENTIALS_ANNOTATION, names['secret'])
pvc_name = annotations.get(PVC_ANNOTATION, names['pvc'])
secret_name = annotations.get(CREDENTIALS_ANNOTATION, names["secret"])
pvc_name = annotations.get(PVC_ANNOTATION, names["pvc"])
operations = [
(
self.apps.delete_namespaced_deployment,
(deployment.metadata.name, namespace),
{'propagation_policy': 'Foreground'},
{"propagation_policy": "Foreground"},
),
(self.core.delete_namespaced_secret, (secret_name, namespace), {}),
]
@@ -243,10 +243,10 @@ class KubernetesService:
delete_resource(*args, **kwargs)
except ApiException as exc:
if exc.status != 404:
raise self._api_error(exc, 'Could not delete userbot instance') from exc
raise self._api_error(exc, "Could not delete userbot instance") from exc
def _find_deployment(self, instance_id: str) -> tuple[str, Any]:
selector = f'{MANAGED_LABEL},{INSTANCE_LABEL}={instance_id}'
selector = f"{MANAGED_LABEL},{INSTANCE_LABEL}={instance_id}"
for namespace in (self.settings.namespace, *self.settings.legacy_namespaces):
try:
items = self.apps.list_namespaced_deployment(
@@ -254,16 +254,16 @@ class KubernetesService:
label_selector=selector,
).items
except ApiException as exc:
raise self._api_error(exc, 'Could not find userbot instance') from exc
raise self._api_error(exc, "Could not find userbot instance") from exc
if items:
return namespace, items[0]
raise PanelError(404, f'Instance {instance_id} does not exist')
raise PanelError(404, f"Instance {instance_id} does not exist")
def _summarize(self, namespace: str, deployment: Any) -> InstanceSummary:
labels = deployment.metadata.labels or {}
annotations = deployment.metadata.annotations or {}
instance_id = labels.get(INSTANCE_LABEL, deployment.metadata.name)
legacy = annotations.get(LEGACY_ANNOTATION) == 'true'
legacy = annotations.get(LEGACY_ANNOTATION) == "true"
pods = self._pods_for_deployment(namespace, deployment)
pod = pods[0] if pods else None
desired = deployment.spec.replicas or 0
@@ -287,21 +287,21 @@ class KubernetesService:
updated_at = pod.status.start_time or pod.metadata.creation_timestamp
if desired == 0:
status = 'stopped'
status = "stopped"
elif reason in {
'CrashLoopBackOff',
'Error',
'ImagePullBackOff',
'ErrImagePull',
'CreateContainerConfigError',
'RunContainerError',
} or (pod is not None and pod.status.phase == 'Failed'):
status = 'error'
"CrashLoopBackOff",
"Error",
"ImagePullBackOff",
"ErrImagePull",
"CreateContainerConfigError",
"RunContainerError",
} or (pod is not None and pod.status.phase == "Failed"):
status = "error"
elif ready and (deployment.status.available_replicas or 0) > 0:
status = 'running'
status = "running"
else:
status = 'pending'
reason = reason or (pod.status.phase if pod is not None else 'Scheduling')
status = "pending"
reason = reason or (pod.status.phase if pod is not None else "Scheduling")
container_spec = deployment.spec.template.spec.containers[0]
limits = (container_spec.resources.limits or {}) if container_spec.resources else {}
@@ -321,8 +321,8 @@ class KubernetesService:
pvc=pvc_name,
storage=storage,
image=container_spec.image,
cpu_limit=limits.get('cpu'),
memory_limit=limits.get('memory'),
cpu_limit=limits.get("cpu"),
memory_limit=limits.get("memory"),
cpu_usage=cpu_usage,
memory_usage=memory_usage,
updated_at=_as_datetime(updated_at),
@@ -338,7 +338,7 @@ class KubernetesService:
label_selector=selector,
).items
except ApiException as exc:
raise self._api_error(exc, 'Could not list userbot Pods') from exc
raise self._api_error(exc, "Could not list userbot Pods") from exc
return sorted(
pods,
key=lambda pod: pod.metadata.creation_timestamp or datetime.min.replace(tzinfo=UTC),
@@ -350,17 +350,17 @@ class KubernetesService:
return None, None
try:
metrics = self.custom.get_namespaced_custom_object(
'metrics.k8s.io',
'v1beta1',
"metrics.k8s.io",
"v1beta1",
namespace,
'pods',
"pods",
pod_name,
)
except (ApiException, AttributeError):
return None, None
containers = metrics.get('containers', [])
cpu = containers[0].get('usage', {}).get('cpu') if containers else None
memory = containers[0].get('usage', {}).get('memory') if containers else None
containers = metrics.get("containers", [])
cpu = containers[0].get("usage", {}).get("cpu") if containers else None
memory = containers[0].get("usage", {}).get("memory") if containers else None
return cpu, memory
def _pvc_storage(self, namespace: str, pvc_name: str | None) -> str | None:
@@ -371,7 +371,7 @@ class KubernetesService:
except ApiException:
return None
requests = pvc.spec.resources.requests or {}
return requests.get('storage')
return requests.get("storage")
@staticmethod
def _deployment_pvc(deployment: Any) -> str | None:
@@ -383,9 +383,9 @@ class KubernetesService:
@staticmethod
def _resource_names(instance_id: str) -> dict[str, str]:
return {
'deployment': f'userbot-{instance_id}',
'secret': f'userbot-{instance_id}-credentials',
'pvc': f'userbot-{instance_id}-data',
"deployment": f"userbot-{instance_id}",
"secret": f"userbot-{instance_id}-credentials",
"pvc": f"userbot-{instance_id}-data",
}
def _metadata(
@@ -398,14 +398,14 @@ class KubernetesService:
name=resource_name,
namespace=self.settings.namespace,
labels={
'app.kubernetes.io/name': 'userbot',
"app.kubernetes.io/name": "userbot",
INSTANCE_LABEL: account.instance_id,
MANAGED_BY_LABEL: 'userbot-panel',
MANAGED_BY_LABEL: "userbot-panel",
},
annotations={
DISPLAY_ANNOTATION: account.display_name,
CREDENTIALS_ANNOTATION: names['secret'],
PVC_ANNOTATION: names['pvc'],
CREDENTIALS_ANNOTATION: names["secret"],
PVC_ANNOTATION: names["pvc"],
},
)
@@ -416,12 +416,12 @@ class KubernetesService:
names: dict[str, str],
) -> client.V1Secret:
return client.V1Secret(
metadata=self._metadata(account, names, names['secret']),
type='Opaque',
metadata=self._metadata(account, names, names["secret"]),
type="Opaque",
string_data={
'API_ID': str(account.api_id),
'API_HASH': account.api_hash,
'STRINGSESSION': session_string,
"API_ID": str(account.api_id),
"API_HASH": account.api_hash,
"STRINGSESSION": session_string,
},
)
@@ -431,12 +431,12 @@ class KubernetesService:
names: dict[str, str],
) -> client.V1PersistentVolumeClaim:
return client.V1PersistentVolumeClaim(
metadata=self._metadata(account, names, names['pvc']),
metadata=self._metadata(account, names, names["pvc"]),
spec=client.V1PersistentVolumeClaimSpec(
access_modes=['ReadWriteOnce'],
access_modes=["ReadWriteOnce"],
storage_class_name=self.settings.storage_class,
resources=client.V1VolumeResourceRequirements(
requests={'storage': account.resources.storage},
requests={"storage": account.resources.storage},
),
),
)
@@ -447,55 +447,63 @@ class KubernetesService:
names: dict[str, str],
) -> client.V1Deployment:
pod_labels = {
'app.kubernetes.io/name': 'userbot',
"app.kubernetes.io/name": "userbot",
INSTANCE_LABEL: account.instance_id,
MANAGED_BY_LABEL: 'userbot-panel',
MANAGED_BY_LABEL: "userbot-panel",
}
container = client.V1Container(
name='userbot',
name="userbot",
image=self.settings.image,
image_pull_policy='Always',
image_pull_policy="Always",
env_from=[
client.V1EnvFromSource(secret_ref=client.V1SecretEnvSource(name=self.settings.common_secret)),
client.V1EnvFromSource(config_map_ref=client.V1ConfigMapEnvSource(name=self.settings.common_config)),
client.V1EnvFromSource(secret_ref=client.V1SecretEnvSource(name=names['secret'])),
client.V1EnvFromSource(
secret_ref=client.V1SecretEnvSource(name=self.settings.common_secret)
),
client.V1EnvFromSource(
config_map_ref=client.V1ConfigMapEnvSource(name=self.settings.common_config)
),
client.V1EnvFromSource(
secret_ref=client.V1SecretEnvSource(name=names["secret"])
),
],
resources=client.V1ResourceRequirements(
requests={'cpu': '80m', 'memory': '512Mi'},
requests={"cpu": "80m", "memory": "512Mi"},
limits={
'cpu': account.resources.cpu_limit,
'memory': account.resources.memory_limit,
"cpu": account.resources.cpu_limit,
"memory": account.resources.memory_limit,
},
),
volume_mounts=[
client.V1VolumeMount(name='data', mount_path='/app/data'),
client.V1VolumeMount(name='downloads', mount_path='/app/downloads'),
client.V1VolumeMount(name="data", mount_path="/app/data"),
client.V1VolumeMount(name="downloads", mount_path="/app/downloads"),
],
)
pod_spec = client.V1PodSpec(
service_account_name='userbot-runtime',
service_account_name="userbot-runtime",
automount_service_account_token=False,
containers=[container],
termination_grace_period_seconds=30,
volumes=[
client.V1Volume(
name='data',
persistent_volume_claim=client.V1PersistentVolumeClaimVolumeSource(claim_name=names['pvc']),
name="data",
persistent_volume_claim=client.V1PersistentVolumeClaimVolumeSource(
claim_name=names["pvc"]
),
),
client.V1Volume(
name='downloads',
name="downloads",
host_path=client.V1HostPathVolumeSource(
path=self.settings.downloads_host_path,
type='DirectoryOrCreate',
type="DirectoryOrCreate",
),
),
],
)
return client.V1Deployment(
metadata=self._metadata(account, names, names['deployment']),
metadata=self._metadata(account, names, names["deployment"]),
spec=client.V1DeploymentSpec(
replicas=1,
strategy=client.V1DeploymentStrategy(type='Recreate'),
strategy=client.V1DeploymentStrategy(type="Recreate"),
selector=client.V1LabelSelector(match_labels=pod_labels),
template=client.V1PodTemplateSpec(
metadata=client.V1ObjectMeta(labels=pod_labels),
@@ -507,9 +515,9 @@ class KubernetesService:
def _rollback(self, created: list[tuple[str, str]]) -> None:
for kind, name in reversed(created):
try:
if kind == 'deployment':
if kind == "deployment":
self.apps.delete_namespaced_deployment(name, self.settings.namespace)
elif kind == 'pvc':
elif kind == "pvc":
self.core.delete_namespaced_persistent_volume_claim(
name,
self.settings.namespace,
@@ -518,7 +526,7 @@ class KubernetesService:
self.core.delete_namespaced_secret(name, self.settings.namespace)
except ApiException as exc:
logger.warning(
'Rollback of %s %s in %s failed: %s',
"Rollback of %s %s in %s failed: %s",
kind,
name,
self.settings.namespace,
@@ -528,9 +536,9 @@ class KubernetesService:
@staticmethod
def _api_error(exc: ApiException, detail: str) -> PanelError:
if exc.status == 403:
return PanelError(503, f'{detail}: Kubernetes RBAC denied the operation')
return PanelError(503, f"{detail}: Kubernetes RBAC denied the operation")
if exc.status == 409:
return PanelError(409, f'{detail}: resource conflict')
return PanelError(409, f"{detail}: resource conflict")
if exc.status == 404:
return PanelError(404, f'{detail}: resource not found')
return PanelError(404, f"{detail}: resource not found")
return PanelError(503, detail)
+38 -38
View File
@@ -30,21 +30,21 @@ async def lifespan(app: FastAPI):
try:
app.state.kubernetes.ensure_prerequisites()
except PanelError as exc:
print(f'WARNING: userbot prerequisites check failed at startup: {exc.detail}')
print(f"WARNING: userbot prerequisites check failed at startup: {exc.detail}")
yield
await app.state.telegram.close()
app = FastAPI(
title='Userbot Kubernetes Control',
version='1.0.0',
title="Userbot Kubernetes Control",
version="1.0.0",
lifespan=lifespan,
)
@app.exception_handler(PanelError)
async def panel_error_handler(_request: Request, exc: PanelError) -> JSONResponse:
return JSONResponse(status_code=exc.status_code, content={'detail': exc.detail})
return JSONResponse(status_code=exc.status_code, content={"detail": exc.detail})
@app.exception_handler(RequestValidationError)
@@ -54,13 +54,13 @@ async def validation_error_handler(
) -> JSONResponse:
errors = [
{
'loc': error.get('loc', ()),
'msg': error.get('msg', 'Invalid value'),
'type': error.get('type', 'value_error'),
"loc": error.get("loc", ()),
"msg": error.get("msg", "Invalid value"),
"type": error.get("type", "value_error"),
}
for error in exc.errors()
]
return JSONResponse(status_code=422, content={'detail': errors})
return JSONResponse(status_code=422, content={"detail": errors})
def kube(request: Request) -> KubernetesService:
@@ -71,7 +71,7 @@ def telegram(request: Request) -> TelegramAuthService:
return request.app.state.telegram
@app.get('/api/health')
@app.get("/api/health")
def health(request: Request) -> dict[str, object]:
service = kube(request)
try:
@@ -80,61 +80,61 @@ def health(request: Request) -> dict[str, object]:
limit=1,
)
except Exception:
return {'ok': False, 'kubernetes': False}
return {'ok': True, 'kubernetes': True}
return {"ok": False, "kubernetes": False}
return {"ok": True, "kubernetes": True}
@app.get('/api/instances', response_model=list[InstanceSummary])
@app.get("/api/instances", response_model=list[InstanceSummary])
def list_instances(
request: Request,
query: str = '',
status: str = '',
query: str = "",
status: str = "",
) -> list[InstanceSummary]:
return kube(request).list_instances(query=query, status=status)
@app.get('/api/instances/{instance_id}', response_model=InstanceSummary)
@app.get("/api/instances/{instance_id}", response_model=InstanceSummary)
def get_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).get_instance(instance_id)
@app.get('/api/instances/{instance_id}/logs')
@app.get("/api/instances/{instance_id}/logs")
def get_logs(
instance_id: str,
request: Request,
tail: Annotated[int, Query(ge=1, le=1000)] = 250,
) -> dict[str, str]:
return {'logs': kube(request).logs(instance_id, tail)}
return {"logs": kube(request).logs(instance_id, tail)}
@app.post('/api/instances/{instance_id}/start', response_model=InstanceSummary)
@app.post("/api/instances/{instance_id}/start", response_model=InstanceSummary)
def start_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).scale(instance_id, 1)
@app.post('/api/instances/{instance_id}/stop', response_model=InstanceSummary)
@app.post("/api/instances/{instance_id}/stop", response_model=InstanceSummary)
def stop_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).scale(instance_id, 0)
@app.post('/api/instances/{instance_id}/restart', response_model=InstanceSummary)
@app.post("/api/instances/{instance_id}/restart", response_model=InstanceSummary)
def restart_instance(instance_id: str, request: Request) -> InstanceSummary:
return kube(request).restart(instance_id)
@app.post('/api/instances/{instance_id}/delete', status_code=204)
@app.post("/api/instances/{instance_id}/delete", status_code=204)
def delete_instance(
instance_id: str,
payload: DeleteRequest,
request: Request,
) -> Response:
if payload.confirmation != instance_id:
raise PanelError(422, 'Type the instance id exactly to confirm deletion')
raise PanelError(422, "Type the instance id exactly to confirm deletion")
kube(request).delete(instance_id, delete_data=payload.delete_data)
return Response(status_code=204)
@app.post('/api/auth/phone/start', response_model=AuthResult)
@app.post("/api/auth/phone/start", response_model=AuthResult)
async def auth_phone_start(
payload: PhoneStart,
request: Request,
@@ -142,10 +142,10 @@ async def auth_phone_start(
service = kube(request)
service.assert_available(payload.instance_id)
flow_id = await telegram(request).start_phone(payload)
return AuthResult(status='code_required', flow_id=flow_id)
return AuthResult(status="code_required", flow_id=flow_id)
@app.post('/api/auth/phone/{flow_id}/code', response_model=AuthResult)
@app.post("/api/auth/phone/{flow_id}/code", response_model=AuthResult)
async def auth_phone_code(
flow_id: str,
payload: CodeSubmit,
@@ -153,12 +153,12 @@ async def auth_phone_code(
) -> AuthResult:
authorized = await telegram(request).submit_code(flow_id, payload.code)
if authorized is None:
return AuthResult(status='password_required', flow_id=flow_id)
return AuthResult(status="password_required", flow_id=flow_id)
instance = kube(request).provision(authorized.account, authorized.session_string)
return AuthResult(status='ready', instance=instance)
return AuthResult(status="ready", instance=instance)
@app.post('/api/auth/phone/{flow_id}/password', response_model=AuthResult)
@app.post("/api/auth/phone/{flow_id}/password", response_model=AuthResult)
async def auth_phone_password(
flow_id: str,
payload: PasswordSubmit,
@@ -166,10 +166,10 @@ async def auth_phone_password(
) -> AuthResult:
authorized = await telegram(request).submit_password(flow_id, payload.password)
instance = kube(request).provision(authorized.account, authorized.session_string)
return AuthResult(status='ready', instance=instance)
return AuthResult(status="ready", instance=instance)
@app.post('/api/auth/string-session', response_model=AuthResult)
@app.post("/api/auth/string-session", response_model=AuthResult)
async def auth_string_session(
payload: StringSessionStart,
request: Request,
@@ -178,28 +178,28 @@ async def auth_string_session(
service.assert_available(payload.instance_id)
authorized = await telegram(request).validate_string_session(payload)
instance = service.provision(authorized.account, authorized.session_string)
return AuthResult(status='ready', instance=instance)
return AuthResult(status="ready", instance=instance)
static_dir = settings.static_dir
assets_dir = static_dir / 'assets'
assets_dir = static_dir / "assets"
if assets_dir.exists():
app.mount('/assets', StaticFiles(directory=assets_dir), name='assets')
app.mount("/assets", StaticFiles(directory=assets_dir), name="assets")
@app.get('/', include_in_schema=False)
@app.get("/", include_in_schema=False)
def index() -> FileResponse:
return FileResponse(static_dir / 'index.html')
return FileResponse(static_dir / "index.html")
@app.get('/{path:path}', include_in_schema=False)
@app.get("/{path:path}", include_in_schema=False)
def spa_fallback(path: str) -> FileResponse:
root = static_dir.resolve()
candidate = (static_dir / path).resolve()
try:
candidate.relative_to(root)
except ValueError:
return FileResponse(static_dir / 'index.html')
return FileResponse(static_dir / "index.html")
if candidate.is_file():
return FileResponse(candidate)
return FileResponse(static_dir / 'index.html')
return FileResponse(static_dir / "index.html")
+12 -12
View File
@@ -5,13 +5,13 @@ from typing import Literal
from pydantic import BaseModel, Field, field_validator
INSTANCE_PATTERN = r'^[a-z0-9](?:[a-z0-9-]{0,38}[a-z0-9])?$'
INSTANCE_PATTERN = r"^[a-z0-9](?:[a-z0-9-]{0,38}[a-z0-9])?$"
class Resources(BaseModel):
storage: str = '1Gi'
cpu_limit: str = '300m'
memory_limit: str = '1536Mi'
storage: str = "1Gi"
cpu_limit: str = "300m"
memory_limit: str = "1536Mi"
class AccountBase(BaseModel):
@@ -21,7 +21,7 @@ class AccountBase(BaseModel):
api_hash: str = Field(min_length=16, max_length=128)
resources: Resources = Field(default_factory=Resources)
@field_validator('display_name', 'api_hash')
@field_validator("display_name", "api_hash")
@classmethod
def strip_text(cls, value: str) -> str:
return value.strip()
@@ -30,12 +30,12 @@ class AccountBase(BaseModel):
class PhoneStart(AccountBase):
phone: str = Field(min_length=7, max_length=32)
@field_validator('phone')
@field_validator("phone")
@classmethod
def normalize_phone(cls, value: str) -> str:
value = value.strip()
if not value.startswith('+'):
raise ValueError('phone must use international format')
if not value.startswith("+"):
raise ValueError("phone must use international format")
return value
@@ -46,10 +46,10 @@ class StringSessionStart(AccountBase):
class CodeSubmit(BaseModel):
code: str = Field(min_length=3, max_length=12)
@field_validator('code')
@field_validator("code")
@classmethod
def normalize_code(cls, value: str) -> str:
return ''.join(value.split())
return "".join(value.split())
class PasswordSubmit(BaseModel):
@@ -67,7 +67,7 @@ class InstanceSummary(BaseModel):
namespace: str
deployment: str
pod: str | None = None
status: Literal['running', 'stopped', 'pending', 'error']
status: Literal["running", "stopped", "pending", "error"]
reason: str | None = None
ready: bool = False
restarts: int = 0
@@ -84,6 +84,6 @@ class InstanceSummary(BaseModel):
class AuthResult(BaseModel):
status: Literal['code_required', 'password_required', 'ready']
status: Literal["code_required", "password_required", "ready"]
flow_id: str | None = None
instance: InstanceSummary | None = None
@@ -23,7 +23,7 @@ class FakeClient:
self.disconnected = True
async def send_code(self, _phone):
return SimpleNamespace(phone_code_hash='hash')
return SimpleNamespace(phone_code_hash="hash")
async def sign_in(self, *_args):
if self.requires_password:
@@ -38,49 +38,49 @@ class FakeClient:
return SimpleNamespace(id=1)
async def export_session_string(self):
return 'exported-session'
return "exported-session"
def phone_payload() -> PhoneStart:
return PhoneStart(
instance_id='test',
display_name='Test',
instance_id="test",
display_name="Test",
api_id=123,
api_hash='0123456789abcdef',
phone='+421900000000',
api_hash="0123456789abcdef",
phone="+421900000000",
)
@pytest.mark.asyncio
async def test_phone_code_success_closes_and_forgets_client(monkeypatch) -> None:
monkeypatch.setattr('app.auth_service.Client', FakeClient)
monkeypatch.setattr("app.auth_service.Client", FakeClient)
auth = TelegramAuthService()
flow_id = await auth.start_phone(phone_payload())
result = await auth.submit_code(flow_id, '12345')
result = await auth.submit_code(flow_id, "12345")
assert result.session_string == 'exported-session'
assert result.session_string == "exported-session"
assert flow_id not in auth.flows
assert FakeClient.instances[-1].disconnected is True
assert FakeClient.instances[-1].kwargs['in_memory'] is True
assert FakeClient.instances[-1].kwargs['no_updates'] is True
assert FakeClient.instances[-1].kwargs["in_memory"] is True
assert FakeClient.instances[-1].kwargs["no_updates"] is True
@pytest.mark.asyncio
async def test_string_session_is_validated_and_closed(monkeypatch) -> None:
monkeypatch.setattr('app.auth_service.Client', FakeClient)
monkeypatch.setattr("app.auth_service.Client", FakeClient)
auth = TelegramAuthService()
payload = StringSessionStart(
instance_id='test',
display_name='Test',
instance_id="test",
display_name="Test",
api_id=123,
api_hash='0123456789abcdef',
session_string='x' * 64,
api_hash="0123456789abcdef",
session_string="x" * 64,
)
result = await auth.validate_string_session(payload)
assert result.session_string == 'exported-session'
assert result.session_string == "exported-session"
assert FakeClient.instances[-1].disconnected is True
@@ -88,7 +88,7 @@ async def test_string_session_is_validated_and_closed(monkeypatch) -> None:
async def test_start_phone_rejects_overflow(monkeypatch) -> None:
from app.errors import PanelError
monkeypatch.setattr('app.auth_service.Client', FakeClient)
monkeypatch.setattr("app.auth_service.Client", FakeClient)
auth = TelegramAuthService(ttl_seconds=3600, max_flows=2)
await auth.start_phone(phone_payload())
@@ -98,4 +98,4 @@ async def test_start_phone_rejects_overflow(monkeypatch) -> None:
await auth.start_phone(phone_payload())
assert error.value.status_code == 429
assert 'Too many pending' in error.value.detail
assert "Too many pending" in error.value.detail
@@ -20,35 +20,35 @@ def service() -> KubernetesService:
def account() -> AccountBase:
return AccountBase(
instance_id='test-account',
display_name='Test Account',
instance_id="test-account",
display_name="Test Account",
api_id=12345,
api_hash='0123456789abcdef0123456789abcdef',
api_hash="0123456789abcdef0123456789abcdef",
)
def test_renders_managed_resources_without_leaking_credentials() -> None:
kube = service()
names = kube._resource_names('test-account')
secret = kube._secret(account(), 'SESSION', names)
names = kube._resource_names("test-account")
secret = kube._secret(account(), "SESSION", names)
pvc = kube._pvc(account(), names)
deployment = kube._deployment(account(), names)
assert secret.string_data == {
'API_ID': '12345',
'API_HASH': '0123456789abcdef0123456789abcdef',
'STRINGSESSION': 'SESSION',
"API_ID": "12345",
"API_HASH": "0123456789abcdef0123456789abcdef",
"STRINGSESSION": "SESSION",
}
assert pvc.spec.storage_class_name == 'local-path-retain'
assert pvc.spec.resources.requests['storage'] == '1Gi'
assert deployment.spec.strategy.type == 'Recreate'
assert pvc.spec.storage_class_name == "local-path-retain"
assert pvc.spec.resources.requests["storage"] == "1Gi"
assert deployment.spec.strategy.type == "Recreate"
assert deployment.spec.replicas == 1
assert deployment.spec.template.spec.automount_service_account_token is False
assert deployment.spec.template.spec.service_account_name == 'userbot-runtime'
assert deployment.spec.template.spec.volumes[1].host_path.path.endswith('/Downloads')
assert deployment.spec.template.spec.service_account_name == "userbot-runtime"
assert deployment.spec.template.spec.volumes[1].host_path.path.endswith("/Downloads")
assert deployment.spec.template.spec.containers[0].resources.requests == {
'cpu': '80m',
'memory': '512Mi',
"cpu": "80m",
"memory": "512Mi",
}
@@ -57,10 +57,10 @@ def test_collision_is_reported_before_auth() -> None:
kube.apps.read_namespaced_deployment.return_value = object()
with pytest.raises(PanelError) as error:
kube.assert_available('test-account')
kube.assert_available("test-account")
assert error.value.status_code == 409
assert 'Deployment' in error.value.detail
assert "Deployment" in error.value.detail
def test_partial_provision_rolls_back_only_created_resources() -> None:
@@ -70,11 +70,11 @@ def test_partial_provision_rolls_back_only_created_resources() -> None:
kube.core.create_namespaced_persistent_volume_claim.side_effect = ApiException(status=500)
with pytest.raises(PanelError):
kube.provision(account(), 'SESSION')
kube.provision(account(), "SESSION")
kube.core.delete_namespaced_secret.assert_called_once_with(
'userbot-test-account-credentials',
'userbot',
"userbot-test-account-credentials",
"userbot",
)
kube.core.delete_namespaced_persistent_volume_claim.assert_not_called()
kube.apps.delete_namespaced_deployment.assert_not_called()
@@ -86,18 +86,18 @@ def test_provision_conflict_reports_existing_instance() -> None:
kube.apps.create_namespaced_deployment.side_effect = ApiException(status=409)
with pytest.raises(PanelError) as error:
kube.provision(account(), 'SESSION')
kube.provision(account(), "SESSION")
assert error.value.status_code == 409
assert 'already exists' in error.value.detail
assert "already exists" in error.value.detail
# Partial resources created before the 409 must be rolled back.
kube.core.delete_namespaced_secret.assert_called_once_with(
'userbot-test-account-credentials',
'userbot',
"userbot-test-account-credentials",
"userbot",
)
kube.core.delete_namespaced_persistent_volume_claim.assert_called_once_with(
'userbot-test-account-data',
'userbot',
"userbot-test-account-data",
"userbot",
)
@@ -105,25 +105,25 @@ def test_delete_retains_pvc_unless_explicitly_requested() -> None:
kube = service()
deployment = SimpleNamespace(
metadata=SimpleNamespace(
name='userbot-test-account',
name="userbot-test-account",
annotations={
'userbot.forust.xyz/credentials-secret': 'credentials',
'userbot.forust.xyz/pvc': 'data',
"userbot.forust.xyz/credentials-secret": "credentials",
"userbot.forust.xyz/pvc": "data",
},
)
)
kube._find_deployment = Mock(return_value=('userbot', deployment))
kube._find_deployment = Mock(return_value=("userbot", deployment))
kube.delete('test-account', delete_data=False)
kube.delete("test-account", delete_data=False)
kube.apps.delete_namespaced_deployment.assert_called_once()
kube.core.delete_namespaced_secret.assert_called_once_with('credentials', 'userbot')
kube.core.delete_namespaced_secret.assert_called_once_with("credentials", "userbot")
kube.core.delete_namespaced_persistent_volume_claim.assert_not_called()
kube.delete('test-account', delete_data=True)
kube.delete("test-account", delete_data=True)
kube.core.delete_namespaced_persistent_volume_claim.assert_called_once_with(
'data',
'userbot',
"data",
"userbot",
)
@@ -131,19 +131,19 @@ def test_legacy_delete_is_blocked() -> None:
kube = service()
deployment = SimpleNamespace(
metadata=SimpleNamespace(
name='forust-userbot-deployment',
annotations={'userbot.forust.xyz/legacy': 'true'},
name="forust-userbot-deployment",
annotations={"userbot.forust.xyz/legacy": "true"},
)
)
kube._find_deployment = Mock(return_value=('default', deployment))
kube._find_deployment = Mock(return_value=("default", deployment))
with pytest.raises(PanelError) as error:
kube.delete('forust', delete_data=False)
kube.delete("forust", delete_data=False)
assert error.value.status_code == 409
def pod(*, ready: bool, phase: str = 'Running', reason: str | None = None):
def pod(*, ready: bool, phase: str = "Running", reason: str | None = None):
waiting = SimpleNamespace(reason=reason) if reason else None
state = SimpleNamespace(waiting=waiting, terminated=None)
status = SimpleNamespace(
@@ -154,25 +154,25 @@ def pod(*, ready: bool, phase: str = 'Running', reason: str | None = None):
start_time=datetime.now(UTC),
)
return SimpleNamespace(
metadata=SimpleNamespace(name='pod-1', creation_timestamp=datetime.now(UTC)),
metadata=SimpleNamespace(name="pod-1", creation_timestamp=datetime.now(UTC)),
status=status,
)
def deployment(replicas: int = 1):
resources = SimpleNamespace(limits={'cpu': '300m', 'memory': '1536Mi'})
container = SimpleNamespace(image='userbot:latest', resources=resources)
resources = SimpleNamespace(limits={"cpu": "300m", "memory": "1536Mi"})
container = SimpleNamespace(image="userbot:latest", resources=resources)
template_spec = SimpleNamespace(containers=[container], volumes=[])
return SimpleNamespace(
metadata=SimpleNamespace(
name='userbot-test-account',
labels={'app.kubernetes.io/instance': 'test-account'},
name="userbot-test-account",
labels={"app.kubernetes.io/instance": "test-account"},
annotations={},
creation_timestamp=datetime.now(UTC),
),
spec=SimpleNamespace(
replicas=replicas,
selector=SimpleNamespace(match_labels={'app': 'test'}),
selector=SimpleNamespace(match_labels={"app": "test"}),
template=SimpleNamespace(spec=template_spec),
),
status=SimpleNamespace(available_replicas=1 if replicas else 0),
@@ -180,13 +180,13 @@ def deployment(replicas: int = 1):
@pytest.mark.parametrize(
('replicas', 'pod_value', 'expected'),
("replicas", "pod_value", "expected"),
[
(0, None, 'stopped'),
(1, pod(ready=True), 'running'),
(1, pod(ready=False, phase='Pending'), 'pending'),
(1, pod(ready=False, reason='CrashLoopBackOff'), 'error'),
(1, pod(ready=False, phase='Failed'), 'error'),
(0, None, "stopped"),
(1, pod(ready=True), "running"),
(1, pod(ready=False, phase="Pending"), "pending"),
(1, pod(ready=False, reason="CrashLoopBackOff"), "error"),
(1, pod(ready=False, phase="Failed"), "error"),
],
)
def test_status_classification(replicas, pod_value, expected) -> None:
@@ -195,6 +195,6 @@ def test_status_classification(replicas, pod_value, expected) -> None:
kube._pvc_storage = Mock(return_value=None)
kube._pod_metrics = Mock(return_value=(None, None))
result = kube._summarize('userbot', deployment(replicas))
result = kube._summarize("userbot", deployment(replicas))
assert result.status == expected
+10 -10
View File
@@ -6,34 +6,34 @@ from pydantic import ValidationError
@pytest.mark.parametrize(
'instance_id',
['Upper', 'has_space', '-leading', 'trailing-', 'x' * 41],
"instance_id",
["Upper", "has_space", "-leading", "trailing-", "x" * 41],
)
def test_invalid_instance_ids(instance_id: str) -> None:
with pytest.raises(ValidationError):
AccountBase(
instance_id=instance_id,
display_name='Test',
display_name="Test",
api_id=1,
api_hash='0123456789abcdef',
api_hash="0123456789abcdef",
)
def test_delete_data_defaults_to_false() -> None:
request = DeleteRequest(confirmation='test')
request = DeleteRequest(confirmation="test")
assert request.delete_data is False
@pytest.mark.asyncio
async def test_validation_response_does_not_echo_secret_input() -> None:
sensitive_value = '-'.join(('very', 'private', 'string', 'session'))
sensitive_value = "-".join(("very", "private", "string", "session"))
error = RequestValidationError(
[
{
'type': 'string_too_short',
'loc': ('body', 'session_string'),
'msg': 'String should have at least 32 characters',
'input': sensitive_value,
"type": "string_too_short",
"loc": ("body", "session_string"),
"msg": "String should have at least 32 characters",
"input": sensitive_value,
}
]
)
+16 -16
View File
@@ -6,43 +6,43 @@ from app.main import spa_fallback
@pytest.fixture
def static_dir(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
static = tmp_path / 'static'
static = tmp_path / "static"
static.mkdir()
(static / 'index.html').write_text('<html>index</html>', encoding='utf-8')
(static / 'app.js').write_text("console.log('app')", encoding='utf-8')
secret = tmp_path / 'secret.txt'
secret.write_text('TOP SECRET', encoding='utf-8')
monkeypatch.setattr('app.main.static_dir', static)
(static / "index.html").write_text("<html>index</html>", encoding="utf-8")
(static / "app.js").write_text("console.log('app')", encoding="utf-8")
secret = tmp_path / "secret.txt"
secret.write_text("TOP SECRET", encoding="utf-8")
monkeypatch.setattr("app.main.static_dir", static)
return static
def static_index(static_dir: Path) -> Path:
return static_dir / 'index.html'
return static_dir / "index.html"
def test_returns_existing_file(static_dir: Path) -> None:
response = spa_fallback('app.js')
assert response.path == static_dir / 'app.js'
response = spa_fallback("app.js")
assert response.path == static_dir / "app.js"
def test_unknown_path_falls_back_to_index(static_dir: Path) -> None:
response = spa_fallback('does/not/exist.js')
response = spa_fallback("does/not/exist.js")
assert response.path == static_index(static_dir)
def test_traversal_does_not_leak_outside_static(static_dir: Path) -> None:
response = spa_fallback('../secret.txt')
response = spa_fallback("../secret.txt")
assert response.path == static_index(static_dir)
response = spa_fallback('%2e%2e/secret.txt')
response = spa_fallback("%2e%2e/secret.txt")
assert response.path == static_index(static_dir)
def test_symlink_outside_static_is_blocked(static_dir: Path, tmp_path: Path) -> None:
target = tmp_path / 'outside.txt'
target.write_text('secret', encoding='utf-8')
link = static_dir / 'leak.txt'
target = tmp_path / "outside.txt"
target.write_text("secret", encoding="utf-8")
link = static_dir / "leak.txt"
link.symlink_to(target)
response = spa_fallback('leak.txt')
response = spa_fallback("leak.txt")
assert response.path == static_index(static_dir)
+1 -2
View File
@@ -6,8 +6,7 @@
"scripts": {
"build": "vite build",
"dev": "vite --host 0.0.0.0",
"test": "vitest run",
"check": "svelte-check --tsconfig ./tsconfig.json"
"test": "vitest run"
},
"dependencies": {
"lucide-svelte": "^0.468.0",