Compare commits

..
Author SHA1 Message Date
renovate-bot f0e0fba125 chore(deps): update container patch updates
renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 7s
ci / lint-shellcheck (push) Successful in 8s
ci / lint-actionlint (push) Successful in 6s
ci / lint-prettier (push) Successful in 2m38s
ci / lint-ruff (push) Successful in 3s
ci / lint-yaml (push) Successful in 3s
ci / lint-dockerfiles (push) Successful in 5s
ci / scan-deps (push) Successful in 17s
ci / test-backend (push) Successful in 8s
ci / test-frontend (push) Successful in 13s
ci / validate (push) Successful in 6s
ci / build (push) Skipped
ci / lint-compose (pull_request) Successful in 3s
ci / lint-actionlint (pull_request) Successful in 2s
ci / lint-shellcheck (pull_request) Successful in 2s
ci / lint-prettier (pull_request) Successful in 3s
ci / lint-ruff (pull_request) Successful in 2s
ci / lint-yaml (pull_request) Successful in 3s
ci / lint-dockerfiles (pull_request) Successful in 2s
ci / scan-deps (pull_request) Successful in 14s
ci / test-backend (pull_request) Successful in 7s
ci / test-frontend (pull_request) Successful in 12s
ci / validate (pull_request) Successful in 4s
ci / build (pull_request) Skipped
renovate-ci / validate-renovate (pull_request) Successful in 5m2s
2026-09-27 22:22:41 +00:00
311 changed files with 25659 additions and 2910 deletions

No files matched your search

-80
View File
@@ -1,80 +0,0 @@
#!/usr/bin/env bash
# Local regressions only: kubectl is mocked and Docker is used for config parsing.
set -euo pipefail
repo="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
scratch="$(mktemp -d)"
trap 'rm -rf "$scratch"' EXIT
mkdir -p "$scratch/repo/app" "$scratch/repo/postgres" "$scratch/repo/netbird" "$scratch/repo/renovate"
git -C "$scratch/repo" init -q
for file in app/compose.yaml postgres/shared-compose.yaml netbird/client.compose.yaml renovate/renovate-compose.yaml; do
touch "$scratch/repo/$file"
done
git -C "$scratch/repo" add .
# shellcheck source=../workflows/compose-lint.sh
source "$repo/.gitea/workflows/compose-lint.sh"
actual="$(cd "$scratch/repo" && compose_files)"
expected=$'app/compose.yaml\nnetbird/client.compose.yaml\npostgres/shared-compose.yaml\nrenovate/renovate-compose.yaml'
[ "$actual" = "$expected" ] || { echo 'Compose discovery missed a file' >&2; exit 1; }
cat >"$scratch/compose.yaml" <<'YAML'
services:
example:
image: busybox:1.37.0
environment:
REQUIRED: ${HOMELAB_TEST_REQUIRED:?required for this regression}
YAML
unset HOMELAB_TEST_REQUIRED
if validate_compose_file "$scratch/compose.yaml" >"$scratch/config.log" 2>&1; then
echo 'Full Compose validation accepted a missing variable' >&2
exit 1
fi
grep -q 'required for this regression' "$scratch/config.log"
HOMELAB_TEST_REQUIRED=present validate_compose_file "$scratch/compose.yaml"
cat >"$scratch/resources.json" <<'JSON'
{"kind":"List","items":[
{"kind":"Deployment","metadata":{"namespace":"app"},"spec":{"template":{"spec":{
"containers":[{"envFrom":[{"secretRef":{"name":"credentials"}},{"secretRef":{"name":"optional","optional":true}}],"env":[{"valueFrom":{"secretKeyRef":{"name":"credentials","key":"password"}}}]}],
"initContainers":[{"envFrom":[{"secretRef":{"name":"init"}}]}],
"imagePullSecrets":[{"name":"registry"}],
"volumes":[{"secret":{"secretName":"mounted"}},{"projected":{"sources":[{"secret":{"name":"projected"}},{"secret":{"name":"optional-projected","optional":true}}]}}]
}}}},
{"kind":"CronJob","metadata":{},"spec":{"jobTemplate":{"spec":{"template":{"spec":{"containers":[{"envFrom":[{"secretRef":{"name":"cron"}}]}]}}}}}},
{"kind":"IngressRoute","metadata":{"namespace":"app"},"spec":{"tls":{"secretName":"controller-issued-tls"}}}
]}
JSON
actual="$(jq -r -f "$repo/.gitea/workflows/secret-references.jq" "$scratch/resources.json" | sort)"
expected=$'app credentials\napp init\napp mounted\napp projected\napp registry\ndefault cron'
[ "$actual" = "$expected" ] || { echo "Unexpected Secret references: $actual" >&2; exit 1; }
REPO="$repo"
# shellcheck source=../workflows/deploy-lib.sh
source "$repo/.gitea/workflows/deploy-lib.sh"
K8S_MANIFESTS=("$scratch/resources.json")
KUSTOMIZE_APPS=()
# No live cluster access. Reject credentials in app even if they exist elsewhere.
kubectl() {
case "$1" in
create) cat "$scratch/resources.json" ;;
get)
if [ "$3" = credentials ] && [ "$5" = app ]; then
return 1
fi
return 0
;;
*) echo "Unexpected kubectl invocation: $*" >&2; return 1 ;;
esac
}
if check_referenced_secrets >"$scratch/secrets.log"; then
echo 'Namespace-scoped Secret check accepted a missing Secret' >&2
exit 1
fi
grep -q 'MISSING OR UNREADABLE: app/credentials' "$scratch/secrets.log"
# API/rendering errors must not produce an empty reference list and pass.
kubectl() { return 1; }
if check_referenced_secrets >"$scratch/secrets.log"; then
echo 'Secret check accepted a failed manifest render' >&2
exit 1
fi
printf '%s\n' 'Deploy validation regressions passed.'
+250 -102
View File
@@ -3,7 +3,7 @@ name: ci
on: on:
push: push:
branches: branches:
- main - "**"
pull_request: pull_request:
workflow_dispatch: workflow_dispatch:
@@ -88,17 +88,19 @@ jobs:
shell: bash shell: bash
run: | run: |
set -euo pipefail set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck)"
export PATH="$tools_dir:$PATH" export PATH="$tools_dir:$PATH"
# userbot/ is a git subtree synced from forust/userbot, so its shell
# scripts are upstream's to maintain, not ours. Linting them would let a
# routine subtree pull turn the deploy gate red on code we do not own.
mapfile -t scripts < <( mapfile -t scripts < <(
git ls-files '*.sh' ':(glob)**/*.bash' git ls-files '*.sh' ':(glob)**/*.bash' ':!userbot/**'
) )
if [ "${#scripts[@]}" -eq 0 ]; then if [ "${#scripts[@]}" -eq 0 ]; then
echo "No shell scripts found." echo "No shell scripts found."
exit 0 exit 0
fi fi
shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}"
bash .gitea/tests/deploy-validation.sh
lint-prettier: lint-prettier:
runs-on: [self-hosted, linux, arch, homelab] runs-on: [self-hosted, linux, arch, homelab]
@@ -195,6 +197,155 @@ jobs:
hadolint -c .hadolint.yaml "${dockerfiles[@]}" hadolint -c .hadolint.yaml "${dockerfiles[@]}"
# Known, accepted, and recorded. Each line is a real advisory against a
# package we build into the panel image, kept in this workflow rather than in
# the package manifest so that a subtree sync from forust/userbot cannot
# silently widen the exemption.
#
# starlette is the reason this job is not simply "fail on everything":
# fastapi 0.115.12 pins `starlette<0.47.0`, and the fixes for the last four
# below need 0.49.1 through 1.3.1, so clearing them means a jump from fastapi
# 0.115.12 to 0.141.x. That is upstream's call, not a drive-by in a lint
# commit. Of the seven, four are reachable here in principle: 1942 is a
# crafted Range header hitting FileResponse, and the panel serves its built
# SPA through exactly that; 249 is request.form() ignoring max_fields for
# x-www-form-urlencoded, which is the login form; 1941 is a large multipart
# body blocking the event loop; 161 and 248 are unvalidated Host and request
# path reaching request.url. 2280 needs HTTPEndpoint, which the panel does
# not use, and 2281 is Windows-only, and this deploys on Linux.
#
# The panel answers on userbot.workstation.internal and has no public
# forust.xyz route, which is what keeps the four reachable ones from being
# an internet-facing DoS. It still manages Telegram credentials.
#
# Deleting an entry here is how you accept a new advisory, so the diff says
# so out loud.
scan-deps:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Audit the Python dependencies that ship in the image
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh pip-audit)"
export PATH="$tools_dir:$PATH"
# requirements.txt, not requirements-dev.txt: this is what the image
# installs, and the test tooling is not a shipped attack surface.
pip-audit -r userbot/panel/backend/requirements.txt --strict \
--ignore-vuln CVE-2025-67720 \
--ignore-vuln PYSEC-2026-161 \
--ignore-vuln PYSEC-2026-1941 \
--ignore-vuln PYSEC-2026-1942 \
--ignore-vuln PYSEC-2026-2280 \
--ignore-vuln PYSEC-2026-2281 \
--ignore-vuln PYSEC-2026-248 \
--ignore-vuln PYSEC-2026-249
# devDependencies are excluded on purpose. `npm audit` on the full tree
# reports 7 findings, and every one of them is a build- or test-time
# package: the esbuild CORS advisory needs a vite dev server serving to
# the internet, and nanoid's infinite loop needs a custom generator
# called with size 0, which postcss does not do. None of them are in the
# 91 kB bundle the panel serves. The one production finding, devalue
# via svelte, is moderate, which is where --audit-level draws the line;
# this fails on the next high or critical one.
- name: Audit the production npm dependencies
shell: bash
run: |
set -euo pipefail
# The pinned node, not whatever the runner has. Its system node is a
# rolling Arch package: during this very push its npm was missing
# entirely, and an hour later it was npm 12 on node 26. Both are the
# wrong major anyway — the panel image is node:22-alpine.
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
export PATH="$tools_dir:$PATH"
cd userbot/panel/frontend
npm ci
npm audit --omit=dev --audit-level=high
test-backend:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# 25 tests over the panel's pydantic models, its auth flow, the SPA
# fallback and the Kubernetes client it shells out with. They existed and
# had never been executed by anything.
#
# Note that userbot/ is a subtree synced from forust/userbot, so a routine
# sync can turn this red on upstream's code. Unlike the shellcheck job,
# which skips that tree because style disagreements there are ours to
# lose, a failing test here is a real defect in a service we deploy.
- name: Run the panel backend test suite
shell: bash
run: |
set -euo pipefail
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh uv)"
export PATH="$tools_dir:$PATH"
# A venv in a temp dir rather than a checked-out one: the runner is
# shared, and a leftover .venv would let a dependency the
# requirements no longer pin still satisfy an import.
#
# --python is not optional. uv otherwise takes whatever interpreter it
# finds first, and which one that is depends on the machine: this
# runner runs jobs on the host, where the only interpreter is 3.14,
# and pyrogram's sync.py calls the bare asyncio.get_event_loop() that
# 3.14 no longer auto-creates, so three tests fail at collection. The
# image is python:3.13-slim, so 3.13 is also the version worth
# testing: uv fetches a managed build of it when the host has none,
# which is what makes this job independent of the runner.
venv="$(mktemp -d)/venv"
uv venv --python 3.13 --quiet "$venv"
uv pip install --quiet --python "$venv/bin/python" \
-r userbot/panel/backend/requirements-dev.txt
# `python -m`, not bare `pytest`: the tests import `app.*` relative to
# the backend directory, which only works if the cwd is on sys.path,
# and only `python -m` puts it there.
cd userbot/panel/backend
"$venv/bin/python" -m pytest tests/ -q
test-frontend:
runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 15
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
# One `npm ci` for both checks below: it is by far the slowest part of
# this job, and a second one would learn nothing the first did not.
#
# `npm ci`, not `npm install`, for the same reason the Dockerfile uses it:
# the lockfile is what makes the tree that gets checked the tree that
# gets shipped.
- name: Type-check and test the panel frontend
shell: bash
run: |
set -euo pipefail
# The pinned node, not whatever the runner has. Its system node is a
# rolling Arch package: during this very push its npm was missing
# entirely, and an hour later it was npm 12 on node 26. Both are the
# wrong major anyway — the panel image is node:22-alpine.
tools_dir="$(bash .gitea/workflows/install-ci-tools.sh node)"
export PATH="$tools_dir:$PATH"
cd userbot/panel/frontend
npm ci
# svelte-check has been a devDependency all along with no script
# pointing at it, so the type errors it reports had nowhere to
# surface. It is clean today, which is the only reason it can be a
# gate: it stops at whatever upstream introduces rather than
# reporting a backlog we inherited.
npm run check
npm test
validate: validate:
runs-on: [self-hosted, linux, arch, homelab] runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 20 timeout-minutes: 20
@@ -313,12 +464,8 @@ jobs:
build: build:
needs: needs:
# The panel's scan-deps/test-backend/test-frontend jobs gated here until
# userbot moved to its own repo; upstream's code is upstream's gate now.
# The rule is unchanged: publishing and passing the checks are the same
# gate, so a commit that fails any of these still cannot move :prod.
[lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate] [lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate]
if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev') && !startsWith(github.ref_name, 'renovate/') if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev')
runs-on: [self-hosted, linux, arch, homelab] runs-on: [self-hosted, linux, arch, homelab]
timeout-minutes: 60 timeout-minutes: 60
outputs: outputs:
@@ -333,20 +480,12 @@ jobs:
id: services id: services
shell: bash shell: bash
run: | run: |
set -euo pipefail
base="${{ github.event.before }}" base="${{ github.event.before }}"
if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then
base="$(git rev-list --max-parents=0 HEAD)" base="$(git rev-list --max-parents=0 HEAD)"
fi fi
# A failed diff used to leave changed_files empty, which reads exactly mapfile -t changed_files < <(git diff --name-only "$base" "${GITHUB_SHA}")
# like "nothing to build": the job went green having built nothing and
# the tag never moved. The status is checked, not assumed.
if ! changed="$(git diff --name-only "$base" "${GITHUB_SHA}")"; then
echo "::error::cannot diff ${base}..${GITHUB_SHA}"
exit 1
fi
mapfile -t changed_files <<<"$changed"
services=() services=()
@@ -366,9 +505,15 @@ jobs:
for file in "${changed_files[@]}"; do for file in "${changed_files[@]}"; do
case "$file" in case "$file" in
dtek_notif/*)
add_service dtek_notif
;;
errorpages/*) errorpages/*)
add_service errorpages add_service errorpages
;; ;;
userbot/*)
add_service userbot
;;
homepages/*) homepages/*)
add_service homepages add_service homepages
;; ;;
@@ -388,63 +533,55 @@ jobs:
echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT" echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT"
- name: Log in to registry - name: Log in to registry
# The pin step below also writes (manifest PUTs), and it runs on every if: steps.services.outputs.services != ''
# main push — including manifest-only ones where services is empty. A
# stale persistent login on the old runner used to mask this; a clean
# runner pushes anonymously and gets 401.
if: steps.services.outputs.services != '' || github.ref_name == 'main'
shell: bash shell: bash
# Through env, not by substitution into the script. A secret written
# into a run: block is pasted into the shell source before bash parses
# it, so a password containing a quote, a backtick or $(...) becomes
# code that runs. Masking the value in the log does not prevent that.
env:
REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }}
REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }}
run: | run: |
set -euo pipefail echo "${{ secrets.REGISTRY_PASSWORD }}" | docker login "${REGISTRY}" \
printf '%s' "$REGISTRY_PASSWORD" | docker login "${REGISTRY}" \ -u "${{ secrets.REGISTRY_USERNAME }}" \
-u "$REGISTRY_USERNAME" \
--password-stdin --password-stdin
- name: Build and push changed images - name: Build and push changed images
if: steps.services.outputs.services != '' if: steps.services.outputs.services != ''
shell: bash shell: bash
run: | run: |
# This step was the one run: block in the workflow without it, and it
# is the one that cannot afford it: a docker push that failed partway
# through the loop used to be followed by more pushes, the loop's exit
# status came from the last one, and the job went green with half the
# images missing from the registry.
set -euo pipefail
IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}" IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}"
# Tags for this push. The commit-pinned name is the point of this
# step: the deploy resolves it in preference to :prod, so a deploy
# that sat in the queue behind a later push still gets the build of
# the commit CI validated, instead of whatever :prod points at by the
# time it runs. See render_pinned in deploy-lib.sh.
commit_tag=""
if [ "${GITHUB_REF_NAME}" = "main" ]; then
commit_tag="sha-${GITHUB_SHA:0:12}"
fi
set_tags() {
tags=()
case "${GITHUB_REF_NAME}" in
main) tags+=("main" "prod") ;;
dev) tags+=("dev") ;;
esac
if [ -n "$commit_tag" ]; then
tags+=("$commit_tag")
fi
}
for service in "${services[@]}"; do for service in "${services[@]}"; do
case "$service" in case "$service" in
dtek_notif)
image="${REGISTRY}/forust/dtek-notif"
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" dtek_notif
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
;;
errorpages) errorpages)
image="${REGISTRY}/forust/error-pages" image="${REGISTRY}/forust/error-pages"
set_tags tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=() build_args=()
for tag in "${tags[@]}"; do for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}") build_args+=(-t "${image}:${tag}")
@@ -457,6 +594,40 @@ jobs:
docker push "${image}:${tag}" docker push "${image}:${tag}"
done done
;; ;;
userbot)
tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
for target in runtime panel; do
case "$target" in
runtime)
context="userbot"
image="${REGISTRY}/forust/userbot"
;;
panel)
context="userbot/panel"
image="${REGISTRY}/forust/userbot-panel"
;;
esac
build_args=()
for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}")
done
docker build \
--cache-from "type=registry,ref=${image}:buildcache" \
--cache-to "type=registry,ref=${image}:buildcache,mode=max" \
"${build_args[@]}" "$context"
for tag in "${tags[@]}"; do
docker push "${image}:${tag}"
done
done
;;
homepages) homepages)
for variant in forust xdfnx; do for variant in forust xdfnx; do
case "$variant" in case "$variant" in
@@ -467,7 +638,15 @@ jobs:
image="${REGISTRY}/forust/xdfnx-homepage" image="${REGISTRY}/forust/xdfnx-homepage"
;; ;;
esac esac
set_tags tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=() build_args=()
for tag in "${tags[@]}"; do for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}") build_args+=(-t "${image}:${tag}")
@@ -493,7 +672,15 @@ jobs:
image="${REGISTRY}/forust/webinar-checker" image="${REGISTRY}/forust/webinar-checker"
;; ;;
esac esac
set_tags tags=()
case "${GITHUB_REF_NAME}" in
main)
tags+=("main" "prod")
;;
dev)
tags+=("dev")
;;
esac
build_args=() build_args=()
for tag in "${tags[@]}"; do for tag in "${tags[@]}"; do
build_args+=(-t "${image}:${tag}") build_args+=(-t "${image}:${tag}")
@@ -509,42 +696,3 @@ jobs:
;; ;;
esac esac
done done
# Every image the tree names has to carry the commit-pinned name, not only
# the ones this push rebuilt. A push that touches nothing but manifests
# builds nothing, and its deploy would then find no commit-pinned tag to
# resolve and quietly fall back to the moving :prod - which is the whole
# failure the commit-pinned name exists to remove.
#
# Re-tagging copies the manifest list and transfers no layers, so pinning
# six images that already exist costs six registry writes.
#
# The list is derived from the tree rather than written out here, so an
# image added to a manifest is covered without a second place to update.
- name: Pin the commit name on the images this push did not rebuild
if: github.ref_name == 'main'
shell: bash
run: |
set -euo pipefail
commit_tag="sha-${GITHUB_SHA:0:12}"
mapfile -t repos < <(
git grep -hoE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+' -- '*.yaml' '*.yml' \
| sort -u
)
if [ "${#repos[@]}" -eq 0 ]; then
echo "No own images referenced by the tree."
exit 0
fi
echo "pinning ${#repos[@]} image(s) to $commit_tag"
for repo in "${repos[@]}"; do
if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then
echo " already built by this push: ${repo##*/}"
continue
fi
if ! docker buildx imagetools inspect "$repo:prod" >/dev/null 2>&1; then
echo " WARNING: ${repo##*/} has no :prod to pin and no build produced it"
continue
fi
docker buildx imagetools create --tag "$repo:$commit_tag" "$repo:prod"
echo " pinned ${repo##*/}"
done
+2 -1
View File
@@ -21,7 +21,8 @@
# All committed Compose files, including the ones deploy never starts. # All committed Compose files, including the ones deploy never starts.
compose_files() { compose_files() {
git ls-files \ git ls-files \
'*compose.yaml' '*compose.yml' '*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \
'*/docker-compose.yaml' '*/docker-compose.yml'
} }
# Prints the flags that turn `docker compose config` into the general check. # Prints the flags that turn `docker compose config` into the general check.
+73 -283
View File
@@ -31,16 +31,6 @@ warn() {
echo "WARNING: $*" >&2 echo "WARNING: $*" >&2
} }
# Prune needs the complete desired set in one invocation. Per-file pruning
# treats resources from the other files as absent and can delete them.
check_prune_mode() {
if [ "$APPLY_PRUNE" = "true" ]; then
echo "ERROR: APPLY_PRUNE=true is unsupported by the per-file deploy loop." >&2
echo "Disable it; remove obsolete resources explicitly after review." >&2
return 1
fi
}
collect_k8s() { collect_k8s() {
git -C "$REPO" ls-files -- "$1" \ git -C "$REPO" ls-files -- "$1" \
| grep -E '\.ya?ml$' \ | grep -E '\.ya?ml$' \
@@ -243,73 +233,20 @@ registry_digest() {
# pipefail reports the rightmost non-zero stage, so a ref the registry does not # pipefail reports the rightmost non-zero stage, so a ref the registry does not
# have would abort the caller at the assignment instead of yielding an empty # have would abort the caller at the assignment instead of yielding an empty
# string. The callers check for empty themselves and report it by name. # string. The callers check for empty themselves and report it by name.
# docker manifest inspect "$1" 2>/dev/null \
# Retried with a hard timeout because the registry has a known hang mode (and | jq -r --arg arch "$arch" '
# a known blink mode: a single failed lookup aborts the whole apply file in .manifests[]?
# render_pinned). A short sleep between attempts lets a restarting registry | select(.platform.os == "linux" and .platform.architecture == $arch)
# come back instead of failing the deploy on one bad second. | .digest
local attempt=0 digest="" ' 2>/dev/null \
while [ "$attempt" -lt 3 ]; do | head -1 || true
digest="$(timeout 25s docker manifest inspect "$1" 2>/dev/null \
| jq -r --arg arch "$arch" '
.manifests[]?
| select(.platform.os == "linux" and .platform.architecture == $arch)
| .digest
' 2>/dev/null \
| head -1 || true)"
[ -n "$digest" ] && break
attempt=$((attempt + 1))
if [ "$attempt" -lt 3 ]; then
echo "WARNING: registry lookup of $1 failed (attempt $attempt/3), retrying in 5s" >&2
sleep 5
fi
done
printf '%s' "$digest"
}
# The commit this deploy is for: what CI validated, or - on a manual dispatch,
# whatever stage_preflight just checked out.
deploy_commit() {
local c="${DEPLOY_SHA:-}"
[ -n "$c" ] || c="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)"
printf '%.12s' "${c:-}"
}
# Resolves one of our image refs to the digest THIS commit's build produced.
#
# A manifest naming `:prod` names a pointer, not a version, and the deploy
# resolves it when the apply runs - which is not when CI ran it. Deploy runs are
# queued rather than cancelled (see deploy.yaml), so two pushes in a row leave
# the first deploy resolving the second push's build: the right manifests with
# the wrong code, and nothing anywhere reports it. ci therefore publishes every
# image it ships under `sha-<commit12>`, a name that cannot move, and that is
# the name resolved here.
#
# The fallback to the plain tag is for an image this pipeline never built. It
# reports itself, because a fallback nobody sees is the failure this removes.
pinned_digest() {
local ref="$1" commit pinned
commit="$(deploy_commit)"
if [ -n "$commit" ]; then
pinned="$(registry_digest "${ref%:*}:sha-$commit")"
if [ -n "$pinned" ]; then
printf '%s' "$pinned"
return 0
fi
fi
pinned="$(registry_digest "$ref")"
if [ -n "$pinned" ]; then
echo "WARNING: ${ref} carries no sha-${commit:-<unknown>} tag; resolved the moving tag instead" >&2
fi
printf '%s' "$pinned"
} }
# Rewrites our own images to immutable digests on the way into the cluster. # Rewrites our own images to immutable digests on the way into the cluster.
# Reads a manifest stream on stdin, writes the pinned stream to stdout. # Reads a manifest stream on stdin, writes the pinned stream to stdout.
# #
# A digest is not knowable when a manifest is written, so it is never committed: # A digest is not knowable when a manifest is written, so it is resolved here, at
# git keeps a readable `:prod` tag and the exact bytes are chosen here, at apply # apply time, and never committed: git keeps a readable `:prod` tag. That is what
# time, from the tag ci published for the commit being deployed. That is what
# makes rollback mean something. `kubectl rollout undo` restores the previous # makes rollback mean something. `kubectl rollout undo` restores the previous
# ReplicaSet's pod template verbatim, and a template naming a digest restores the # ReplicaSet's pod template verbatim, and a template naming a digest restores the
# exact bytes that were serving before. A template naming a moving tag does not — # exact bytes that were serving before. A template naming a moving tag does not —
@@ -334,7 +271,7 @@ render_pinned() {
while read -r ref; do while read -r ref; do
[ -n "$ref" ] || continue [ -n "$ref" ] || continue
digest="$(pinned_digest "$ref")" digest="$(registry_digest "$ref")"
if [ -z "$digest" ]; then if [ -z "$digest" ]; then
echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2 echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2
echo " The build job has to push that tag before the deploy resolves it." >&2 echo " The build job has to push that tag before the deploy resolves it." >&2
@@ -402,7 +339,7 @@ restart_stale_images() {
while read -r ns target image; do while read -r ns target image; do
[ -n "${target:-}" ] || continue [ -n "${target:-}" ] || continue
if [ -z "${digests[$image]:-}" ]; then if [ -z "${digests[$image]:-}" ]; then
digests[$image]="$(pinned_digest "$image")" digests[$image]="$(registry_digest "$image")"
fi fi
want="${digests[$image]}" want="${digests[$image]}"
if [ -z "$want" ]; then if [ -z "$want" ]; then
@@ -514,14 +451,6 @@ rollback_workloads() {
local -a recovered=() local -a recovered=()
while read -r kind ns name; do while read -r kind ns name; do
[ -n "${kind:-}" ] || continue [ -n "${kind:-}" ] || continue
# Helm-owned workloads are already rolled back by the release's --rollback-on-failure
# upgrade. `rollout undo` here would step back to the revision Helm just
# escaped (the failed one), so leave them for the operator instead.
if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then
echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}"
unrecovered+=("${kind}/${ns}/${name} (helm-managed)")
continue
fi
if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \ if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \
&& kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then && kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then
echo " rolled back: ${kind}/${ns}/${name}" echo " rolled back: ${kind}/${ns}/${name}"
@@ -560,66 +489,6 @@ helm_repo_for() {
esac esac
} }
# helm_release_status <release> <namespace>
# Prints the release status in lowercase (deployed, failed, pending-rollback,
# ...) or "not-found" when the release does not exist yet.
helm_release_status() {
local out
if ! out="$(helm status "$1" -n "$2" 2>&1)"; then
echo "not-found"
return 0
fi
awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]'
}
# recover_pending_release <release> <namespace>
# Rolls a release out of a pending-* state left by a failed upgrade with --rollback-on-failure
# whose own rollback never completed. Without this every future upgrade errors
# out until a human runs `helm rollback`. Passes through releases that are not
# pending (deployed, failed, not-found). Returns non-zero when the release is
# still not recoverable, so the pipeline fails loud instead of wedging.
recover_pending_release() {
local release="$1" namespace="$2" status
status="$(helm_release_status "$release" "$namespace")"
case "$status" in
pending-upgrade|pending-rollback|pending-install)
log "Release $release is $status, rolling back to the last deployed revision"
if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then
echo "WARN: helm rollback of $release did not complete"
return 1
fi
status="$(helm_release_status "$release" "$namespace")"
if [ "$status" != "deployed" ]; then
echo "WARN: $release is $status after rollback"
return 1
fi
;;
esac
return 0
}
# wait_for_calm <stage>
# The deploy itself is heavy enough to melt this single node (helm churn plus
# apply churn drove load past 40, killed netbird/ssh, left helm pending-*).
# Never pile a heavy step onto an already-hot node: wait up to 10 minutes for
# the 1-minute load average to drop below the ceiling, then proceed anyway
# with a warning so a permanently busy node cannot wedge the pipeline forever.
wait_for_calm() {
local load waited=0
while [ "$waited" -lt 600 ]; do
load="$(cut -d' ' -f1 /proc/loadavg | cut -d. -f1)"
if [ "$load" -lt 28 ]; then
return 0
fi
if [ "$((waited % 60))" -eq 0 ]; then
log "$1: load $load, waiting for calm (<28)..."
fi
sleep 15
waited=$((waited + 15))
done
echo "WARN: $1: node still loaded ($load) after 10m, proceeding anyway"
}
upgrade_helm_releases() { upgrade_helm_releases() {
local entry release chart namespace version values marker repo local entry release chart namespace version values marker repo
for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do
@@ -640,33 +509,13 @@ upgrade_helm_releases() {
helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true
helm repo update "${repo%% *}" >/dev/null 2>&1 || true helm repo update "${repo%% *}" >/dev/null 2>&1 || true
log "Upgrading $release ($chart $version)" log "Upgrading $release ($chart $version)"
wait_for_calm "helm $release" # --atomic rolls the release back when the upgrade times out or the workloads
# A previous run with --rollback-on-failure whose own rollback never finished leaves the # it touches never become ready, so a bad chart bump is not left half applied.
# release in pending-*, which blocks every future upgrade. Recover first helm upgrade --install "$release" "$chart" \
# so one wedged revision cannot wedge the pipeline forever.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: $release is stuck and automatic rollback did not recover it, run 'helm rollback $release -n $namespace' by hand."
return 1
fi
# --rollback-on-failure (+ --wait) rolls the release back when the upgrade
# times out or the workloads it touches never become ready, so a bad chart
# bump is not left half applied. (--atomic was this combo; deprecated.)
if ! helm upgrade --install "$release" "$chart" \
--namespace "$namespace" \ --namespace "$namespace" \
--version "$version" \ --version "$version" \
--values "$REPO/$values" \ --values "$REPO/$values" \
--wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then --atomic --cleanup-on-fail --timeout 10m
echo "WARN: upgrade of $release failed, checking release state"
# --rollback-on-failure already attempted its own rollback; finish the job when that
# rollback never completed, otherwise the release stays pending-* and
# blocks every future run.
if ! recover_pending_release "$release" "$namespace"; then
echo "ERROR: upgrade of $release failed and the release did not recover, run 'helm rollback $release -n $namespace' by hand."
else
echo "ERROR: upgrade of $release failed (release is back on its previous revision)."
fi
return 1
fi
done done
} }
@@ -696,53 +545,27 @@ stage_preflight() {
git -C "$REPO" reset --hard "$target" git -C "$REPO" reset --hard "$target"
} }
# Required pod Secrets, scoped to the resource namespace. TLS route Secrets are
# created by cert-manager and are not prerequisites for applying a Certificate.
check_referenced_secrets() {
local m k objects refs extracted ns name
local missing=()
refs=""
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
objects="$(kubectl create --dry-run=client --validate=false -f "$m" -o json)" || return 1
extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1
refs+="$extracted"$'\n'
done
for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do
objects="$(kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json)" || return 1
extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1
refs+="$extracted"$'\n'
done
while read -r ns name; do
[ -n "${name:-}" ] || continue
if kubectl get secret "$name" -n "$ns" -o name >/dev/null 2>&1; then
echo " ok: $ns/$name"
else
echo " MISSING OR UNREADABLE: $ns/$name"
missing+=("$ns/$name")
fi
done < <(printf '%s' "$refs" | sort -u)
if [ "${#missing[@]}" -gt 0 ]; then
echo "ERROR: required pod Secrets are missing or unreadable:"
printf ' - %s\n' "${missing[@]}"
echo "Create them in the listed namespaces from the service's secret example."
return 1
fi
}
stage_validate() { stage_validate() {
check_prune_mode || return 1
cd "$REPO" cd "$REPO"
select_manifests select_manifests
local m k cf local m k cf
# The deploy host has the local .env and secret files. Resolve them here so # Compose .env files and secret files are gitignored by design, so the
# missing configuration fails before either apply job changes workloads. # workstation never has real values for the inactive stacks. This stage only
# CI keeps the structure-only check for inactive stacks. # runs the full check on active stacks; the general structure check for every
# committed Compose file (active or not) lives in the ci workflow, which has no
# .env at all.
#
# Active stacks are still validated with interpolation and env-file resolution
# off, so required-variable guards (:?) and missing local files do not fail the
# deploy. Normalization and consistency checks stay enabled.
# shellcheck source=compose-lint.sh # shellcheck source=compose-lint.sh
source "$REPO/.gitea/workflows/compose-lint.sh" source "$REPO/.gitea/workflows/compose-lint.sh"
local compose_validate_flags=()
mapfile -t compose_validate_flags < <(compose_safe_flags)
log "Validate compose stacks" log "Validate compose stacks"
for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do
echo " config: $cf" echo " config: $cf"
validate_compose_file "$cf" validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"}
done done
log "Validate k8s manifests (kubectl dry-run=client)" log "Validate k8s manifests (kubectl dry-run=client)"
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
@@ -760,20 +583,48 @@ stage_validate() {
done done
log "Checking referenced Secrets exist" log "Checking referenced Secrets exist"
echo " (deploy never applies *secret*.yaml; create missing ones manually)" echo " (deploy never applies *secret*.yaml; create missing ones manually)"
check_referenced_secrets local ref_secrets=() missing_secrets=() all_secrets s
if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then
while IFS= read -r s; do
[ -n "$s" ] && ref_secrets+=("$s")
done < <(
{
grep -h -A1 -E 'secretRef:|secretKeyRef:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
grep -h -E 'secretName:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true
} | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true
)
fi
all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)"
for s in ${ref_secrets[@]+"${ref_secrets[@]}"}; do
if printf '%s\n' "$all_secrets" | grep -qx "$s"; then
echo " ok: $s"
else
echo " MISSING: $s"
missing_secrets+=("$s")
fi
done
if [ "${#missing_secrets[@]}" -gt 0 ]; then
echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:"
printf ' - %s\n' "${missing_secrets[@]}"
echo "Create them manually from the laptop, e.g.:"
echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example"
exit 1
fi
} }
stage_apply_k8s() { stage_apply_k8s() {
check_prune_mode || return 1
cd "$REPO" cd "$REPO"
select_manifests >/dev/null select_manifests >/dev/null
local ns_files=() other_files=() m k local ns_files=() other_files=() m k prune_opts=()
for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do
case "$m" in case "$m" in
*/namespace.y?ml) ns_files+=("$m") ;; */namespace.y?ml) ns_files+=("$m") ;;
*) other_files+=("$m") ;; *) other_files+=("$m") ;;
esac esac
done done
if [ "$APPLY_PRUNE" = "true" ]; then
prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy)
fi
# Record what is about to change, and publish it for the verify job, before # Record what is about to change, and publish it for the verify job, before
# the first apply. Both are fatal on failure: see snapshot_dir. # the first apply. Both are fatal on failure: see snapshot_dir.
@@ -794,11 +645,10 @@ stage_apply_k8s() {
fi fi
fi fi
upgrade_helm_releases upgrade_helm_releases
wait_for_calm "apply resources"
if [ "${#other_files[@]}" -gt 0 ]; then if [ "${#other_files[@]}" -gt 0 ]; then
log "Applying resources (${#other_files[@]} files, our images pinned to digests)" log "Applying resources (${#other_files[@]} files, our images pinned to digests)"
for m in "${other_files[@]}"; do for m in "${other_files[@]}"; do
if ! render_pinned <"$m" | kubectl apply -f -; then if ! render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -; then
echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 echo "ERROR: apply failed for ${m#"$REPO"/}" >&2
exit 1 exit 1
fi fi
@@ -811,6 +661,19 @@ stage_apply_k8s() {
exit 1 exit 1
fi fi
done done
if [ -f "$REPO/userbot/k8s/active" ]; then
log "userbot panel hook"
if kubectl get secret userbot-common-secrets -n userbot >/dev/null 2>&1; then
echo " userbot-common-secrets already present in userbot ns, not touching"
elif kubectl get secret userbot-common-secrets -n default >/dev/null 2>&1; then
echo " bootstrapping userbot-common-secrets into userbot ns"
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
else
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
fi
fi
restart_stale_images restart_stale_images
# No verification here on purpose. This stage may be killed at any point by # No verification here on purpose. This stage may be killed at any point by
@@ -962,32 +825,6 @@ smoke_hosts() {
| sort -u | sort -u
} }
# Traefik's own list of the routes it actually built. The Kubernetes CRs are the
# wrong source for this: when a middleware fails to load, Traefik drops the
# router that referenced it and leaves the CR behind looking perfectly healthy.
#
# api.insecure is already on for the internal `traefik` entrypoint, but the pod
# IP is not routable from the node, so read it through kubectl exec rather than
# standing up a port-forward. HTTP only: the TCP routers match on HostSNI(`*`)
# and the UDP ones carry no rule at all, both selected by entrypoint and port,
# so neither can answer whether a given host has a route.
traefik_http_routes() {
kubectl -n traefik exec deploy/traefik -- \
wget -qO- --timeout=10 http://127.0.0.1:8080/api/http/routers 2>/dev/null \
| jq -c '[.[] | {status, rule: (.rule // "")}]'
}
# The hosts Traefik currently routes to, one per line. Every backticked token of
# an enabled rule counts, which is a superset of the hosts -- PathPrefix values
# land here too, harmlessly -- but it keeps the host syntax in one place instead
# of a matcher per host. The scan keeps the delimiters, so strip them: what
# belongs in a comparison against a hostname is the bare name.
traefik_routed_hosts() {
jq -r '[.[] | select(.status == "enabled") | (.rule // "")
| scan("`[^`]+`") | ltrimstr("`") | rtrimstr("`")]
| unique | .[]' <<<"$1"
}
# stage_verify_k8s watches the rollout, which reports that the pods converged. # stage_verify_k8s watches the rollout, which reports that the pods converged.
# It cannot tell a converged pod from a serving one: a route pointing at the # It cannot tell a converged pod from a serving one: a route pointing at the
# wrong port, a Service selector that matches nothing the app listens on, a 500 # wrong port, a Service selector that matches nothing the app listens on, a 500
@@ -999,12 +836,6 @@ traefik_routed_hosts() {
# or a 404 from a path the service does not serve still means the chain is # or a 404 from a path the service does not serve still means the chain is
# intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means # intact. Only a transport failure (no DNS, refused, timeout) or a 5xx means
# the service is not serving, and only those fail the run. # the service is not serving, and only those fail the run.
#
# Except that a 404 is not evidence on its own. A router Traefik refused to
# build answers with the same 404 and nothing behind it, so a middleware that
# fails to load takes down every route that referenced it while
# this stage reports `ok` for all of them. No status code separates those two
# cases, so ask Traefik which routes it built and fail on the difference.
stage_smoke() { stage_smoke() {
cd "$REPO" cd "$REPO"
select_manifests >/dev/null select_manifests >/dev/null
@@ -1051,52 +882,11 @@ stage_smoke() {
esac esac
done done
# Second gate. The probe above only means something if a router matched the
# host in the first place, so compare the hosts we expect against the hosts
# Traefik reports and fail on the difference.
local routes routed
if ! routes="$(traefik_http_routes)"; then
echo "ERROR: could not read Traefik's router list, refusing to report success"
return 1
fi
routed="$(traefik_routed_hosts "$routes")"
local -a unrouted=()
local tries=3
while :; do
unrouted=()
for h in "${hosts[@]}"; do
grep -qxF "$h" <<<"$routed" || unrouted+=("$h")
done
if [ "${#unrouted[@]}" -eq 0 ]; then
break
fi
# A router mid-rollout is legitimately absent for a moment. A middleware
# that failed to load stays absent, so waiting cannot paper over it.
if [ "$tries" -le 1 ]; then
break
fi
tries=$((tries - 1))
warn "${#unrouted[@]} host(s) have no enabled route yet, re-checking in 10s"
sleep 10
if ! routes="$(traefik_http_routes)"; then
break
fi
routed="$(traefik_routed_hosts "$routes")"
done
if [ "${#unrouted[@]}" -ne 0 ]; then
for h in "${unrouted[@]}"; do
echo " NO ROUTE $h (Traefik has no enabled router for this host)"
done
bad=1
fi
if [ "$bad" -ne 0 ]; then if [ "$bad" -ne 0 ]; then
echo "ERROR: at least one active service is not serving over its public route" echo "ERROR: at least one active service is not serving over its public route"
return 1 return 1
fi fi
echo "all ${#hosts[@]} route(s) answered and have a router" echo "all ${#hosts[@]} route(s) answered"
} }
stage_apply_compose() { stage_apply_compose() {
+5 -52
View File
@@ -5,7 +5,6 @@ on:
# workflow_dispatch so a red lint/validate run can never reach the cluster. # workflow_dispatch so a red lint/validate run can never reach the cluster.
workflow_run: workflow_run:
workflows: [ci] workflows: [ci]
branches: [main]
types: [completed] types: [completed]
workflow_dispatch: workflow_dispatch:
@@ -40,15 +39,10 @@ env:
jobs: jobs:
preflight: preflight:
# Autodeploy defaults to OFF: pushes deploy only when the AUTODEPLOY repo
# variable is set to 'true' (Settings -> Actions -> Variables). A manual
# Run workflow always bypasses the switch: dispatching it is the explicit
# intent to deploy.
if: >- if: >-
(vars.AUTODEPLOY == 'true' || github.event_name == 'workflow_dispatch') && github.event_name != 'workflow_run' ||
(github.event_name != 'workflow_run' ||
(github.event.workflow_run.conclusion == 'success' && (github.event.workflow_run.conclusion == 'success' &&
github.event.workflow_run.head_branch == 'main')) github.event.workflow_run.head_branch == 'main')
runs-on: [self-hosted, linux, arch, homelab, prod] runs-on: [self-hosted, linux, arch, homelab, prod]
timeout-minutes: 10 timeout-minutes: 10
steps: steps:
@@ -79,34 +73,8 @@ jobs:
needs: [validate] needs: [validate]
runs-on: [self-hosted, linux, arch, homelab, prod] runs-on: [self-hosted, linux, arch, homelab, prod]
# Apply only, no verification, so this is just the work itself: snapshot, # Apply only, no verification, so this is just the work itself: snapshot,
# then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop. # then up to three sequential `helm upgrade --atomic --timeout 10m`, then the
# Verification has its own job and its own budget. # apply loop. Verification has its own job and its own budget.
#
# 45 is roughly four times the measured cost of the stage, which is
# deliberately not raised on a theory:
#
# helm, healthy 3 no-op upgrades ~3-5 min
# helm, one release bad rollback-on-failure spends its 10m, ~10-15 min
# then rolls that one back
# apply loop ~40 manifests, 4 of which ~1 min
# resolve an image digest
# restart_stale_images 7.6s to find 8 workloads, ~0.5 min
# 9.8s to resolve their digests
#
# The helm figure is one release, not three: `set -e` aborts
# upgrade_helm_releases on the first failure, so a broken release costs
# 10m and the other two are never attempted. Multiplying 10m by three
# overstates the worst case by 20 minutes.
#
# The 45 minutes this was last raised to 45 were still not enough, and the
# job logs for those runs no longer exist, so what actually consumed the
# budget is not known - the two measurable candidates above account for
# ~15 of it. The unbounded `docker manifest inspect` against the registry's
# known hang mode is now bounded inside registry_digest (25s timeout, 3
# attempts): a dead registry fails each owned image after ~85s instead of
# hanging the stage, and a blinking one is retried instead of failing the
# whole apply file. Still open: make the stage announce which manifest it
# is working on, so a killed run leaves a diagnosable last line.
timeout-minutes: 45 timeout-minutes: 45
steps: steps:
- name: Checkout repository - name: Checkout repository
@@ -144,22 +112,7 @@ jobs:
needs.apply-k8s.result != 'skipped' && needs.apply-k8s.result != 'skipped' &&
needs.apply-compose.result != 'skipped' needs.apply-compose.result != 'skipped'
runs-on: [self-hosted, linux, arch, homelab, prod] runs-on: [self-hosted, linux, arch, homelab, prod]
# Not raised, because the arithmetic does not close. # ceil(changed_workloads / 8) waves of ROLLOUT_TIMEOUT each, plus rollback.
#
# 32 workloads are under management and the wave width is 8, so the verify
# itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when
# every rollout times out rather than converging. That is already 20 of 30.
#
# The other 10 would have to absorb rollback, and rollback_workloads is a
# serial `while read` loop at 300s per failed workload. 10 minutes buys two.
# Any larger number is buying a bigger multiple of an unbounded term rather
# than covering a known cost: 60 minutes buys eight, and 60 minutes is
# therefore not a bound, it is a guess with two digits.
#
# The number becomes derivable the moment rollback uses the same wave width
# as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160,
# and 45 covers verify plus rollback at full width. That change is to the
# recovery path and is not folded into a timeout edit.
timeout-minutes: 30 timeout-minutes: 30
steps: steps:
- name: Checkout repository - name: Checkout repository
+4 -35
View File
@@ -16,10 +16,6 @@ here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}" TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}"
BIN_DIR="$TOOLS_DIR/bin" BIN_DIR="$TOOLS_DIR/bin"
mkdir -p "$BIN_DIR" mkdir -p "$BIN_DIR"
# The just-installed tools must resolve inside this script too: callers only
# prepend BIN_DIR to PATH after the script exits, so a bare `uv` below would
# miss the binary install_uv just placed (exit 127 on a clean runner).
export PATH="$BIN_DIR:$PATH"
arch="$(uname -m)" arch="$(uname -m)"
# Upstream projects disagree on arch spelling: kubeconform and actionlint use # Upstream projects disagree on arch spelling: kubeconform and actionlint use
@@ -57,31 +53,14 @@ fetch() {
fi fi
} }
# resolve <command>
# Absolute path to use for invoking a tool: the copy in BIN_DIR when present,
# otherwise the name for PATH lookup. Every version check and every in-script
# invocation goes through this, so a tool missing from both places reads as
# "not installed" instead of dying with 127 under `set -e`.
resolve() {
if [ -x "$BIN_DIR/$1" ]; then
printf '%s' "$BIN_DIR/$1"
else
printf '%s' "$1"
fi
}
# installed_version <command> # installed_version <command>
# Prints the version of an already-installed tool, or nothing. Each tool spells # Prints the version of an already-installed tool, or nothing. Each tool spells
# its version flag differently, hence the case. # its version flag differently, hence the case.
installed_version() { installed_version() {
local bin out local out
bin="$(resolve "$1")"
if ! command -v "$bin" >/dev/null 2>&1; then
return 0
fi
case "$1" in case "$1" in
kubeconform) out="$("$bin" -v 2>/dev/null | head -1 || true)" ;; kubeconform) out="$("$1" -v 2>/dev/null | head -1 || true)" ;;
*) out="$("$bin" --version 2>/dev/null | head -1 || true)" ;; *) out="$("$1" --version 2>/dev/null | head -1 || true)" ;;
esac esac
printf '%s' "$out" printf '%s' "$out"
} }
@@ -120,15 +99,6 @@ install_shellcheck() {
rm -rf "$tmp" rm -rf "$tmp"
} }
install_jq() {
if at_version jq "${JQ_VERSION}"; then
return 0
fi
fetch "https://github.com/jqlang/jq/releases/download/jq-${JQ_VERSION}/jq-linux-${goarch}" \
"$BIN_DIR/jq"
chmod 0755 "$BIN_DIR/jq"
}
install_uv() { install_uv() {
if at_version uv "${UV_VERSION}"; then if at_version uv "${UV_VERSION}"; then
return 0 return 0
@@ -160,7 +130,7 @@ install_uv_tool() {
return 0 return 0
fi fi
install_uv install_uv
UV_TOOL_BIN_DIR="$BIN_DIR" "$BIN_DIR/uv" tool install --force "$1==$2" >/dev/null UV_TOOL_BIN_DIR="$BIN_DIR" uv tool install --force "$1==$2" >/dev/null
} }
install_ruff() { install_ruff() {
@@ -245,7 +215,6 @@ for tool in "${wanted[@]}"; do
case "$tool" in case "$tool" in
kubeconform) install_kubeconform ;; kubeconform) install_kubeconform ;;
shellcheck) install_shellcheck ;; shellcheck) install_shellcheck ;;
jq) install_jq ;;
actionlint) install_actionlint ;; actionlint) install_actionlint ;;
prettier) install_prettier ;; prettier) install_prettier ;;
ruff) install_ruff ;; ruff) install_ruff ;;
-14
View File
@@ -2,23 +2,9 @@ name: renovate-ci
on: on:
pull_request: pull_request:
paths:
- "renovate/**"
- ".gitea/workflows/renovate-ci.yaml"
- ".gitea/workflows/sync-renovate-configmap.sh"
- ".gitea/workflows/compose-lint.sh"
- ".gitea/workflows/install-ci-tools.sh"
- ".gitea/workflows/tool-versions.env"
push: push:
branches: branches:
- main - main
paths:
- "renovate/**"
- ".gitea/workflows/renovate-ci.yaml"
- ".gitea/workflows/sync-renovate-configmap.sh"
- ".gitea/workflows/compose-lint.sh"
- ".gitea/workflows/install-ci-tools.sh"
- ".gitea/workflows/tool-versions.env"
workflow_dispatch: workflow_dispatch:
permissions: permissions:
+1 -1
View File
@@ -81,7 +81,7 @@ jobs:
docker run --rm \ docker run --rm \
-v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \ -v "$PWD/renovate/renovate.json:/opt/renovate/renovate.json:ro" \
-e RENOVATE_PLATFORM=gitea \ -e RENOVATE_PLATFORM=gitea \
-e RENOVATE_ENDPOINT=https://git.forust.xyz/api/v1 \ -e RENOVATE_ENDPOINT=https://gitea.forust.xyz/api/v1 \
-e RENOVATE_TOKEN="$RENOVATE_TOKEN" \ -e RENOVATE_TOKEN="$RENOVATE_TOKEN" \
-e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \ -e RENOVATE_GITHUB_COM_TOKEN="${RENOVATE_GITHUB_COM_TOKEN:-}" \
-e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \ -e RENOVATE_REPOSITORIES="${RENOVATE_REPOSITORIES:-forust/homelab}" \
-13
View File
@@ -1,13 +0,0 @@
# kubectl emits a List for files containing multiple resources.
(if .kind == "List" then .items[] else . end)
| (.metadata.namespace // "default") as $ns
| [
(.. | objects
| (.secretRef? // empty), (.secretKeyRef? // empty), (.secret? // empty)
| select(.optional != true)
| .name // .secretName // empty),
(.. | objects | .imagePullSecrets[]?.name)
]
| unique[]
| select(. != null and . != "")
| "\($ns) \(.)"
+4 -45
View File
@@ -21,51 +21,10 @@ ssh_key="$key_dir/deploy_key"
printf '%s\n' "$DEPLOY_KEY" > "$ssh_key" printf '%s\n' "$DEPLOY_KEY" > "$ssh_key"
chmod 600 "$ssh_key" chmod 600 "$ssh_key"
# A connection that died silently used to hang until the job timeout, and the ssh -i "$ssh_key" -p "$deploy_port" \
# stage was never re-run: one flaky TCP session cost a whole 45-minute apply. -o BatchMode=yes -o StrictHostKeyChecking=accept-new \
# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds "${DEPLOY_USER}@${DEPLOY_HOST}" \
# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that "REPO=$deploy_path APPLY_PRUNE=${APPLY_PRUNE:-false} DEPLOY_SHA=${DEPLOY_SHA:-} DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-} STAGE=$1 bash -se" <<'EOF'
# fails on its own merits exits with the remote's status, so a real failure
# still surfaces its own log instead of burning three attempts. The stages are
# declarative applies, so re-running one that had already committed is harmless.
ssh_opts=(
-i "$ssh_key" -p "$deploy_port"
-o BatchMode=yes -o StrictHostKeyChecking=accept-new
-o ConnectTimeout=15
-o ServerAliveInterval=15 -o ServerAliveCountMax=4
)
rc=0
# apply-k8s and apply-compose are separate workflow jobs so the graph stays
# intact for the verify job, but on a single node they must not run at once:
# host docker churn on top of cluster churn is what melts the node (load 40+,
# netbird/ssh die, helm is left pending-*). Serialize them on the workstation
# with a shared lock; whoever arrives second waits.
remote_cmd=(bash -se)
case "$1" in
apply-k8s | apply-compose)
remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se)
;;
esac
for attempt in 1 2 3; do
if [ "$attempt" -gt 1 ]; then
echo ":: warning::ssh transport failed, retrying (${attempt}/3)"
sleep $((attempt * 5))
fi
rc=0
# shellcheck disable=SC2029 # remote_cmd/ssh_opts expand on the client on purpose: they select the local ssh invocation, only the heredoc runs remotely.
ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \
env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \
"DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \
"STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$?
source "$REPO/.gitea/workflows/deploy-lib.sh" source "$REPO/.gitea/workflows/deploy-lib.sh"
run_stage "$STAGE" run_stage "$STAGE"
EOF EOF
[ "$rc" -eq 0 ] && break
[ "$rc" -ne 255 ] && break
done
if [ "$rc" -ne 0 ]; then
echo ":: error::stage $1 failed over ssh (exit $rc)"
fi
exit "$rc"
-3
View File
@@ -31,6 +31,3 @@ UV_VERSION="0.12.17"
# so the tree that gets tested is the tree that gets built. Renovate keeps this # so the tree that gets tested is the tree that gets built. Renovate keeps this
# in step with the Dockerfile's node: tag via the "node runtime" group. # in step with the Dockerfile's node: tag via the "node runtime" group.
NODE_VERSION="22.23.3" NODE_VERSION="22.23.3"
# Secret-reference regression tests parse rendered Kubernetes objects.
JQ_VERSION="1.8.1"
+3 -12
View File
@@ -51,8 +51,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: adguard-deployment name: adguard-deployment
namespace: adguard namespace: adguard
spec: spec:
@@ -60,12 +58,12 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: adguard app: adguard
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
app: adguard app: adguard
annotations:
reloader.stakater.com/auto: "true"
spec: spec:
containers: containers:
- name: adguard - name: adguard
@@ -75,7 +73,7 @@ spec:
memory: "1.5Gi" memory: "1.5Gi"
cpu: "300m" cpu: "300m"
requests: requests:
memory: "512Mi" memory: "500Mi"
cpu: "50m" cpu: "50m"
ports: ports:
- containerPort: 3000 - containerPort: 3000
@@ -84,13 +82,6 @@ spec:
name: dns name: dns
- containerPort: 853 - containerPort: 853
name: dot name: dot
readinessProbe:
tcpSocket:
port: dns
initialDelaySeconds: 5
periodSeconds: 5
successThreshold: 1
failureThreshold: 3
volumeMounts: volumeMounts:
- name: adguard-data - name: adguard-data
mountPath: /opt/adguardhome/work mountPath: /opt/adguardhome/work
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`dns.forust.xyz`) - match: Host(`dns.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: adguard-service - name: adguard-service
port: 3000 port: 3000
+5 -13
View File
@@ -27,8 +27,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: authentik-server-deployment name: authentik-server-deployment
namespace: authentik namespace: authentik
spec: spec:
@@ -36,8 +34,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: authentik-server app: authentik-server
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -56,8 +52,8 @@ spec:
- containerPort: 9000 - containerPort: 9000
resources: resources:
requests: requests:
memory: "768Mi" memory: "700Mi"
cpu: "100m" cpu: "300m"
limits: limits:
memory: "1.5Gi" memory: "1.5Gi"
cpu: "1000m" cpu: "1000m"
@@ -65,8 +61,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: authentik-worker-deployment name: authentik-worker-deployment
namespace: authentik namespace: authentik
spec: spec:
@@ -74,8 +68,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: authentik-worker app: authentik-worker
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -94,8 +86,8 @@ spec:
name: authentik-secrets name: authentik-secrets
resources: resources:
requests: requests:
memory: "320Mi" memory: "512Mi"
cpu: "100m" cpu: "300m"
limits: limits:
memory: "768Mi" memory: "1Gi"
cpu: "700m" cpu: "700m"
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`auth.forust.xyz`) - match: Host(`auth.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: authentik-server-service - name: authentik-server-service
port: 9000 port: 9000
+2 -6
View File
@@ -1,8 +1,6 @@
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: cfddns name: cfddns
labels: labels:
app: cfddns app: cfddns
@@ -11,8 +9,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: cfddns app: cfddns
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -26,10 +22,10 @@ spec:
imagePullPolicy: Always imagePullPolicy: Always
resources: resources:
requests: requests:
memory: "32Mi" memory: "20Mi"
cpu: "30m" cpu: "30m"
limits: limits:
memory: "128Mi" memory: "64Mi"
cpu: "50m" cpu: "50m"
envFrom: envFrom:
- secretRef: - secretRef:
-4
View File
@@ -17,8 +17,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: checkmk-deployment name: checkmk-deployment
namespace: checkmk namespace: checkmk
spec: spec:
@@ -26,8 +24,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: checkmk app: checkmk
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`cmk.forust.xyz`) - match: Host(`cmk.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: checkmk-service - name: checkmk-service
port: 5000 port: 5000
File renamed without changes.
+2 -6
View File
@@ -1,8 +1,6 @@
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: cloudflared name: cloudflared
labels: labels:
app: cloudflared app: cloudflared
@@ -11,8 +9,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: cloudflared app: cloudflared
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -34,8 +30,8 @@ spec:
key: TUNNEL_TOKEN key: TUNNEL_TOKEN
resources: resources:
requests: requests:
memory: "128Mi" memory: "32Mi"
cpu: "30m" cpu: "30m"
limits: limits:
memory: "256Mi" memory: "128Mi"
cpu: "200m" cpu: "200m"
+1 -1
View File
@@ -1,7 +1,7 @@
services: services:
convertx: convertx:
container_name: convertx container_name: convertx
image: ghcr.io/c4illin/convertx:v0.19.0 image: ghcr.io/c4illin/convertx:v0.18.0
restart: unless-stopped restart: unless-stopped
ports: ports:
- "9992:3000" - "9992:3000"
+2 -3
View File
@@ -31,13 +31,12 @@ spec:
name: bentopdf name: bentopdf
ports: ports:
- containerPort: 8080 - containerPort: 8080
# p95 4M, max 11M over 7 days. Was 50Mi/700Mi.
resources: resources:
requests: requests:
memory: "32Mi" memory: "50Mi"
cpu: "50m" cpu: "50m"
ephemeral-storage: "100Mi" ephemeral-storage: "100Mi"
limits: limits:
memory: "128Mi" memory: "700Mi"
cpu: "700m" cpu: "700m"
ephemeral-storage: "5Gi" ephemeral-storage: "5Gi"
+3 -8
View File
@@ -13,8 +13,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: convertx-deployment name: convertx-deployment
namespace: converters namespace: converters
spec: spec:
@@ -22,15 +20,13 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: convertx app: convertx
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
app: convertx app: convertx
spec: spec:
containers: containers:
- image: ghcr.io/c4illin/convertx:v0.19.0 - image: ghcr.io/c4illin/convertx:v0.18.0
name: convertx name: convertx
envFrom: envFrom:
- configMapRef: - configMapRef:
@@ -42,14 +38,13 @@ spec:
volumeMounts: volumeMounts:
- mountPath: /data - mountPath: /data
name: data name: data
# p95 85M, max 136M over 7 days, spikes while converting. Was 250Mi/1.5Gi.
resources: resources:
requests: requests:
memory: "128Mi" memory: "250Mi"
cpu: "100m" cpu: "100m"
limits: limits:
cpu: "1500m" cpu: "1500m"
memory: "512Mi" memory: "1.5Gi"
volumes: volumes:
- name: data - name: data
persistentVolumeClaim: persistentVolumeClaim:
+24
View File
@@ -0,0 +1,24 @@
apiVersion: traefik.io/v1alpha1
kind: Middleware
metadata:
name: crowdsec-bouncer
namespace: crowdsec
spec:
plugin:
crowdsec-bouncer:
enabled: true
LogLevel: INFO
CrowdsecMode: live
CrowdsecLapiScheme: http
CrowdsecLapiHost: crowdsec-service.crowdsec.svc.cluster.local:8080
CrowdsecLapiKeyFile: "/etc/traefik/secrets/traefik-api-key"
# LAPI lookup is SYNCHRONOUS and per-request: the plugin blocks on
# `GET /v1/decisions?ip=...&banned=true` before the request reaches
# the backend, and fails CLOSED (403) if the lookup exceeds the
# timeout. Unset, the fork defaults to 10s, which is an eternity for
# a request path: a single slow LAPI (idle 1.3-7.4s here) turned
# every request into a 10s hang and then a self-inflicted 403.
# 2s keeps the fail-closed path fast and bounded; with the LAPI
# resourced properly (see crowdsec-values.yaml) the lookup is
# sub-100ms and this budget is never hit.
CrowdsecLapiTimeout: "2s"
+15 -92
View File
@@ -16,12 +16,6 @@ agent:
value: crowdsecurity/traefik crowdsecurity/base-http-scenarios value: crowdsecurity/traefik crowdsecurity/base-http-scenarios
- name: DISABLE_COLLECTIONS - name: DISABLE_COLLECTIONS
value: crowdsecurity/sshd value: crowdsecurity/sshd
# Bans on 401/403 bursts hurt more than they protect: with L3 enforcement
# a false positive cuts the IP off everything (SSH included), and past
# incidents show legit automation (deploy runner, mesh peers, registry
# pulls) tripping this probe. Probing/XSS/SQLi/CVE scenarios stay.
- name: DISABLE_SCENARIOS
value: crowdsecurity/http-generic-bf
metrics: metrics:
enabled: true enabled: true
serviceMonitor: serviceMonitor:
@@ -64,44 +58,9 @@ config:
reason: "Mobile IP whitelist" reason: "Mobile IP whitelist"
cidr: cidr:
- "84.245.64.0/18" - "84.245.64.0/18"
# CrowdSec's own guidance: CIDR allowlisting belongs at the parser stage.
# A parser whitelist discards the event before it reaches a bucket, so
# these addresses never produce an overflow and never become a decision.
# A postoverflow whitelist is checked only *after* the ban exists, and
# the bouncer answers 403 for as long as it does - which is a window we
# do not want the deploy sitting in.
local-network.yaml: |
name: forust/local-network
description: "Whitelist loopback, private and VPN networks"
whitelist:
reason: "Local network"
cidr:
- "127.0.0.0/8"
- "10.0.0.0/8"
- "172.16.0.0/12"
- "192.168.0.0/16"
# CGNAT range (RFC 6598). The workstation and the k0s node live
# here on WireGuard, and 100.64.0.0/10 is not covered by the
# RFC 1918 blocks above.
- "100.64.0.0/10"
- "169.254.0.0/16"
- "fc00::/7"
- "fe80::/10"
vps-whitelist.yaml: |
name: forust/vps-whitelist
description: "Whitelist static VPS"
whitelist:
reason: "VPS"
ip:
- "193.181.211.79"
postoverflows: postoverflows:
s01-whitelist: s01-whitelist:
# The one whitelist that has to stay here: resolving a hostname is a
# network call, and the docs put expensive lookups in postoverflows on
# purpose - it runs only when a bucket actually overflows.
# ddns.forust.xyz is the public home address, not a private one, so
# forust/local-network does not cover it.
home-dynamic-ip.yaml: | home-dynamic-ip.yaml: |
name: forust/home-dynamic-ip name: forust/home-dynamic-ip
description: "Whitelist home dynamic IP" description: "Whitelist home dynamic IP"
@@ -109,59 +68,23 @@ config:
reason: "Home dynamic IP" reason: "Home dynamic IP"
expression: expression:
- evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz") - evt.Overflow.Alert.Source.IP in LookupHost("ddns.forust.xyz")
# The hairpin-NAT address of the router (192.168.88.1) is what the
# LAPI-only main config override, merged over config.yaml. NOTE: the # Gitea Actions runner presents to Traefik - it is NOT the home
# chart's own default for this key is REPLACED, not merged, so its # dynamic IP, so the whitelist above did not cover it. During a
# auto_registration block is repeated verbatim below - drop it and the # deploy the runner POSTs to the Actions API many times a second;
# agent can no longer register itself. # a single 403 storm was enough to earn it a 4h ban and break every
config.yaml.local: | # later job. Whitelisting the whole LAN also covers phones and
api: # tablets browsing over 192.168.88.0/24.
server: lan.yaml: |
auto_registration: # Activate if not using TLS for authentication name: forust/lan
enabled: true description: "Whitelist local network"
token: "${REGISTRATION_TOKEN}" # /!\ Do not modify this variable (auto-generated and handled by the chart) whitelist:
allowed_ranges: # /!\ Make sure to adapt to the pod IP ranges used by your cluster reason: "Local network"
- "127.0.0.1/32" cidr:
- "192.168.0.0/16" - "127.0.0.0/8"
- "10.0.0.0/8" - "10.0.0.0/8"
- "172.16.0.0/12" - "172.16.0.0/12"
# This homelab has no egress to console.crowdsec.cloud: DNS does - "192.168.0.0/16"
# not resolve. The LAPI kept trying anyway ("Signal push: N
# signals to push", "capi metrics: sending" every 10s) and each
# attempt sat on a resolver timeout WHILE HOLDING A WRITE
# TRANSACTION, which is what kept stalling per-request decision
# lookups even with WAL enabled. Nothing to share and nothing to
# pull - turn the Central API off instead of letting it block the
# only database writer we have.
online_client:
sharing: false
pull:
community: false
blocklists: false
disable_usage_metrics_export: true
db_config:
# SQLite without WAL serialises every reader behind the writer's
# rollback journal, and the LAPI writes constantly: the agent pushes
# Traefik alerts read from Loki, the metrics collector counts
# decisions, the bouncer touches "last pull" on every request.
# Symptom: decision lookups taking 10-30s (and a second connection
# that could not even open the database) while the LAPI sat at 28m
# CPU - the process was blocked in fsync, not computing. Every
# bouncer-protected request then blew through the plugin timeout and
# fail-closed with 403, on every site at once.
# The PVC is local-path-retain (hostPath), not a network share, so
# WAL is safe here; the crowdsec docs recommend it for exactly this
# ("allowing more concurrency in SQLite that will improve
# performances in most scenarios").
use_wal: true
# Keeps the alert table bounded. At the 5000/7d default the file
# reached 54MB in 15 days off the Traefik access log alone, and the
# metrics collector counts decisions on a timer; a smaller working
# set means fewer full scans. Crowdsec only prunes - SQLite never
# shrinks the file, so the size stays until a manual VACUUM.
flush:
max_items: 1000
max_age: 24h
lapi: lapi:
env: env:
+20 -8
View File
@@ -30,14 +30,10 @@
# 3. ensure the static machine exists, recreating it with the # 3. ensure the static machine exists, recreating it with the
# Secret password if missing (agent retry loops reconnect # Secret password if missing (agent retry loops reconnect
# on their own - same name + same password); # on their own - same name + same password);
# 4. prune bouncer entries idle for 30d. # 4. prune bouncer entries idle for 30d;
# # 5. delete decisions from LePresidente/http-generic-403-bf, a hub
# It used to also delete LePresidente/http-generic-403-bf decisions hourly. # scenario that bans an IP for 4h after 5 POST-403s in 10s and
# That was a workaround for the bouncer failing closed on a slow LAPI and # therefore bans us for our own bouncer's fail-closed 403s.
# 403-ing the deploy runner into a 4h ban. The bouncer now polls decisions
# into a cache and never blocks on an unreachable LAPI, so it cannot
# manufacture those 403s any more, and the scenario only fires against real
# scanners - deleting their decisions hourly was undoing a working ban.
# #
# Manual apply (crowdsec/k8s is NOT managed by deploy.yaml): # Manual apply (crowdsec/k8s is NOT managed by deploy.yaml):
# kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml # kubectl apply -f crowdsec/k8s/janitor-cronjob.yaml
@@ -200,3 +196,19 @@ spec:
fi fi
echo "== 4. prune stale bouncers (no pull for 30d) ==" echo "== 4. prune stale bouncers (no pull for 30d) =="
$LAPI_EXEC cscli bouncers prune -d 720h --force $LAPI_EXEC cscli bouncers prune -d 720h --force
echo "== 5. drop http-403-bf decisions (4h self-bans) =="
# `LePresidente/http-generic-403-bf` (hub item
# crowdsecurity/http-generic-bf v0.9) bans any source IP
# after 5 POSTs answered 403 within 10s, for 4h. That
# includes 403s this homelab generates ITSELF (any
# bouncer fail-closed, any app CSRF/rate-limit 403), and a
# 4h ban on the runner/home IP silently breaks deploys and
# browsing. The scenario cannot be removed per-scenario -
# it is baked into a hub item, and disabling the whole
# base-http-scenarios collection would drop ~40 useful
# detections. Instead we keep the detection and drop its
# decisions hourly; the LAN/home whitelists in
# crowdsec-values.yaml handle the legit sources, so this
# only ever hits real scanners (who are re-banned anyway).
$LAPI_EXEC cscli decisions delete \
--scenario LePresidente/http-generic-403-bf --all || true
+2
View File
@@ -18,6 +18,8 @@ spec:
- match: Host(`dockmon.forust.xyz`) - match: Host(`dockmon.forust.xyz`)
kind: Rule kind: Rule
middlewares: middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
- name: security-headers@file - name: security-headers@file
services: services:
- name: dockmon-service - name: dockmon-service
+1 -1
View File
@@ -1,7 +1,7 @@
services: services:
downtify: downtify:
container_name: downtify container_name: downtify
image: ghcr.io/henriquesebastiao/downtify:3.4.0 image: ghcr.io/henriquesebastiao/downtify:3.1.0
restart: unless-stopped restart: unless-stopped
# ports: # ports:
# - '7077:8000' # - '7077:8000'
+1 -3
View File
@@ -20,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: downtify app: downtify
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -29,7 +27,7 @@ spec:
spec: spec:
containers: containers:
- name: downtify - name: downtify
image: ghcr.io/henriquesebastiao/downtify:3.4.0 image: ghcr.io/henriquesebastiao/downtify:3.1.0
ports: ports:
- containerPort: 8000 - containerPort: 8000
volumeMounts: volumeMounts:
+2
View File
@@ -10,6 +10,8 @@ spec:
- match: Host(`downtify.forust.xyz`) - match: Host(`downtify.forust.xyz`)
kind: Rule kind: Rule
middlewares: middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
- name: security-chain@file - name: security-chain@file
services: services:
- name: downtify-service - name: downtify-service
+13
View File
@@ -0,0 +1,13 @@
FROM python:3.9-alpine
WORKDIR /app
# Установка зависимостей
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
# Копирование кода
COPY main.py .
COPY .env .
# Запуск бота
CMD ["python", "-u", "main.py"]
+373
View File
@@ -0,0 +1,373 @@
Mozilla Public License Version 2.0
==================================
1. Definitions
--------------
1.1. "Contributor"
means each individual or legal entity that creates, contributes to
the creation of, or owns Covered Software.
1.2. "Contributor Version"
means the combination of the Contributions of others (if any) used
by a Contributor and that particular Contributor's Contribution.
1.3. "Contribution"
means Covered Software of a particular Contributor.
1.4. "Covered Software"
means Source Code Form to which the initial Contributor has attached
the notice in Exhibit A, the Executable Form of such Source Code
Form, and Modifications of such Source Code Form, in each case
including portions thereof.
1.5. "Incompatible With Secondary Licenses"
means
(a) that the initial Contributor has attached the notice described
in Exhibit B to the Covered Software; or
(b) that the Covered Software was made available under the terms of
version 1.1 or earlier of the License, but not also under the
terms of a Secondary License.
1.6. "Executable Form"
means any form of the work other than Source Code Form.
1.7. "Larger Work"
means a work that combines Covered Software with other material, in
a separate file or files, that is not Covered Software.
1.8. "License"
means this document.
1.9. "Licensable"
means having the right to grant, to the maximum extent possible,
whether at the time of the initial grant or subsequently, any and
all of the rights conveyed by this License.
1.10. "Modifications"
means any of the following:
(a) any file in Source Code Form that results from an addition to,
deletion from, or modification of the contents of Covered
Software; or
(b) any new file in Source Code Form that contains any Covered
Software.
1.11. "Patent Claims" of a Contributor
means any patent claim(s), including without limitation, method,
process, and apparatus claims, in any patent Licensable by such
Contributor that would be infringed, but for the grant of the
License, by the making, using, selling, offering for sale, having
made, import, or transfer of either its Contributions or its
Contributor Version.
1.12. "Secondary License"
means either the GNU General Public License, Version 2.0, the GNU
Lesser General Public License, Version 2.1, the GNU Affero General
Public License, Version 3.0, or any later versions of those
licenses.
1.13. "Source Code Form"
means the form of the work preferred for making modifications.
1.14. "You" (or "Your")
means an individual or a legal entity exercising rights under this
License. For legal entities, "You" includes any entity that
controls, is controlled by, or is under common control with You. For
purposes of this definition, "control" means (a) the power, direct
or indirect, to cause the direction or management of such entity,
whether by contract or otherwise, or (b) ownership of more than
fifty percent (50%) of the outstanding shares or beneficial
ownership of such entity.
2. License Grants and Conditions
--------------------------------
2.1. Grants
Each Contributor hereby grants You a world-wide, royalty-free,
non-exclusive license:
(a) under intellectual property rights (other than patent or trademark)
Licensable by such Contributor to use, reproduce, make available,
modify, display, perform, distribute, and otherwise exploit its
Contributions, either on an unmodified basis, with Modifications, or
as part of a Larger Work; and
(b) under Patent Claims of such Contributor to make, use, sell, offer
for sale, have made, import, and otherwise transfer either its
Contributions or its Contributor Version.
2.2. Effective Date
The licenses granted in Section 2.1 with respect to any Contribution
become effective for each Contribution on the date the Contributor first
distributes such Contribution.
2.3. Limitations on Grant Scope
The licenses granted in this Section 2 are the only rights granted under
this License. No additional rights or licenses will be implied from the
distribution or licensing of Covered Software under this License.
Notwithstanding Section 2.1(b) above, no patent license is granted by a
Contributor:
(a) for any code that a Contributor has removed from Covered Software;
or
(b) for infringements caused by: (i) Your and any other third party's
modifications of Covered Software, or (ii) the combination of its
Contributions with other software (except as part of its Contributor
Version); or
(c) under Patent Claims infringed by Covered Software in the absence of
its Contributions.
This License does not grant any rights in the trademarks, service marks,
or logos of any Contributor (except as may be necessary to comply with
the notice requirements in Section 3.4).
2.4. Subsequent Licenses
No Contributor makes additional grants as a result of Your choice to
distribute the Covered Software under a subsequent version of this
License (see Section 10.2) or under the terms of a Secondary License (if
permitted under the terms of Section 3.3).
2.5. Representation
Each Contributor represents that the Contributor believes its
Contributions are its original creation(s) or it has sufficient rights
to grant the rights to its Contributions conveyed by this License.
2.6. Fair Use
This License is not intended to limit any rights You have under
applicable copyright doctrines of fair use, fair dealing, or other
equivalents.
2.7. Conditions
Sections 3.1, 3.2, 3.3, and 3.4 are conditions of the licenses granted
in Section 2.1.
3. Responsibilities
-------------------
3.1. Distribution of Source Form
All distribution of Covered Software in Source Code Form, including any
Modifications that You create or to which You contribute, must be under
the terms of this License. You must inform recipients that the Source
Code Form of the Covered Software is governed by the terms of this
License, and how they can obtain a copy of this License. You may not
attempt to alter or restrict the recipients' rights in the Source Code
Form.
3.2. Distribution of Executable Form
If You distribute Covered Software in Executable Form then:
(a) such Covered Software must also be made available in Source Code
Form, as described in Section 3.1, and You must inform recipients of
the Executable Form how they can obtain a copy of such Source Code
Form by reasonable means in a timely manner, at a charge no more
than the cost of distribution to the recipient; and
(b) You may distribute such Executable Form under the terms of this
License, or sublicense it under different terms, provided that the
license for the Executable Form does not attempt to limit or alter
the recipients' rights in the Source Code Form under this License.
3.3. Distribution of a Larger Work
You may create and distribute a Larger Work under terms of Your choice,
provided that You also comply with the requirements of this License for
the Covered Software. If the Larger Work is a combination of Covered
Software with a work governed by one or more Secondary Licenses, and the
Covered Software is not Incompatible With Secondary Licenses, this
License permits You to additionally distribute such Covered Software
under the terms of such Secondary License(s), so that the recipient of
the Larger Work may, at their option, further distribute the Covered
Software under the terms of either this License or such Secondary
License(s).
3.4. Notices
You may not remove or alter the substance of any license notices
(including copyright notices, patent notices, disclaimers of warranty,
or limitations of liability) contained within the Source Code Form of
the Covered Software, except that You may alter any license notices to
the extent required to remedy known factual inaccuracies.
3.5. Application of Additional Terms
You may choose to offer, and to charge a fee for, warranty, support,
indemnity or liability obligations to one or more recipients of Covered
Software. However, You may do so only on Your own behalf, and not on
behalf of any Contributor. You must make it absolutely clear that any
such warranty, support, indemnity, or liability obligation is offered by
You alone, and You hereby agree to indemnify every Contributor for any
liability incurred by such Contributor as a result of warranty, support,
indemnity or liability terms You offer. You may include additional
disclaimers of warranty and limitations of liability specific to any
jurisdiction.
4. Inability to Comply Due to Statute or Regulation
---------------------------------------------------
If it is impossible for You to comply with any of the terms of this
License with respect to some or all of the Covered Software due to
statute, judicial order, or regulation then You must: (a) comply with
the terms of this License to the maximum extent possible; and (b)
describe the limitations and the code they affect. Such description must
be placed in a text file included with all distributions of the Covered
Software under this License. Except to the extent prohibited by statute
or regulation, such description must be sufficiently detailed for a
recipient of ordinary skill to be able to understand it.
5. Termination
--------------
5.1. The rights granted under this License will terminate automatically
if You fail to comply with any of its terms. However, if You become
compliant, then the rights granted under this License from a particular
Contributor are reinstated (a) provisionally, unless and until such
Contributor explicitly and finally terminates Your grants, and (b) on an
ongoing basis, if such Contributor fails to notify You of the
non-compliance by some reasonable means prior to 60 days after You have
come back into compliance. Moreover, Your grants from a particular
Contributor are reinstated on an ongoing basis if such Contributor
notifies You of the non-compliance by some reasonable means, this is the
first time You have received notice of non-compliance with this License
from such Contributor, and You become compliant prior to 30 days after
Your receipt of the notice.
5.2. If You initiate litigation against any entity by asserting a patent
infringement claim (excluding declaratory judgment actions,
counter-claims, and cross-claims) alleging that a Contributor Version
directly or indirectly infringes any patent, then the rights granted to
You by any and all Contributors for the Covered Software under Section
2.1 of this License shall terminate.
5.3. In the event of termination under Sections 5.1 or 5.2 above, all
end user license agreements (excluding distributors and resellers) which
have been validly granted by You or Your distributors under this License
prior to termination shall survive termination.
************************************************************************
* *
* 6. Disclaimer of Warranty *
* ------------------------- *
* *
* Covered Software is provided under this License on an "as is" *
* basis, without warranty of any kind, either expressed, implied, or *
* statutory, including, without limitation, warranties that the *
* Covered Software is free of defects, merchantable, fit for a *
* particular purpose or non-infringing. The entire risk as to the *
* quality and performance of the Covered Software is with You. *
* Should any Covered Software prove defective in any respect, You *
* (not any Contributor) assume the cost of any necessary servicing, *
* repair, or correction. This disclaimer of warranty constitutes an *
* essential part of this License. No use of any Covered Software is *
* authorized under this License except under this disclaimer. *
* *
************************************************************************
************************************************************************
* *
* 7. Limitation of Liability *
* -------------------------- *
* *
* Under no circumstances and under no legal theory, whether tort *
* (including negligence), contract, or otherwise, shall any *
* Contributor, or anyone who distributes Covered Software as *
* permitted above, be liable to You for any direct, indirect, *
* special, incidental, or consequential damages of any character *
* including, without limitation, damages for lost profits, loss of *
* goodwill, work stoppage, computer failure or malfunction, or any *
* and all other commercial damages or losses, even if such party *
* shall have been informed of the possibility of such damages. This *
* limitation of liability shall not apply to liability for death or *
* personal injury resulting from such party's negligence to the *
* extent applicable law prohibits such limitation. Some *
* jurisdictions do not allow the exclusion or limitation of *
* incidental or consequential damages, so this exclusion and *
* limitation may not apply to You. *
* *
************************************************************************
8. Litigation
-------------
Any litigation relating to this License may be brought only in the
courts of a jurisdiction where the defendant maintains its principal
place of business and such litigation shall be governed by laws of that
jurisdiction, without reference to its conflict-of-law provisions.
Nothing in this Section shall prevent a party's ability to bring
cross-claims or counter-claims.
9. Miscellaneous
----------------
This License represents the complete agreement concerning the subject
matter hereof. If any provision of this License is held to be
unenforceable, such provision shall be reformed only to the extent
necessary to make it enforceable. Any law or regulation which provides
that the language of a contract shall be construed against the drafter
shall not be used to construe this License against a Contributor.
10. Versions of the License
---------------------------
10.1. New Versions
Mozilla Foundation is the license steward. Except as provided in Section
10.3, no one other than the license steward has the right to modify or
publish new versions of this License. Each version will be given a
distinguishing version number.
10.2. Effect of New Versions
You may distribute the Covered Software under the terms of the version
of the License under which You originally received the Covered Software,
or under the terms of any subsequent version published by the license
steward.
10.3. Modified Versions
If you create software not governed by this License, and you want to
create a new license for such software, you may create and use a
modified version of this License if you rename the license and remove
any references to the name of the license steward (except to note that
such modified license differs from this License).
10.4. Distributing Source Code Form that is Incompatible With Secondary
Licenses
If You choose to distribute Source Code Form that is Incompatible With
Secondary Licenses under the terms of this version of the License, the
notice described in Exhibit B of this License must be attached.
Exhibit A - Source Code Form License Notice
-------------------------------------------
This Source Code Form is subject to the terms of the Mozilla Public
License, v. 2.0. If a copy of the MPL was not distributed with this
file, You can obtain one at https://mozilla.org/MPL/2.0/.
If it is not possible or desirable to put the notice in a particular
file, then You may include the notice in a location (such as a LICENSE
file in a relevant directory) where a recipient would be likely to look
for such a notice.
You may add additional accurate notices of copyright ownership.
Exhibit B - "Incompatible With Secondary Licenses" Notice
---------------------------------------------------------
This Source Code Form is "Incompatible With Secondary Licenses", as
defined by the Mozilla Public License, v. 2.0.
+15
View File
@@ -0,0 +1,15 @@
services:
dtek_notif:
build:
context: .
dockerfile: Dockerfile
image: gcr.forust.xyz/forust/dtek-notif:latest
pull_policy: build
restart: unless-stopped
environment:
- TZ=Europe/Kyiv
dns:
- 1.1.1.1
- 8.8.8.8
networks:
- default
+748
View File
@@ -0,0 +1,748 @@
import asyncio
import contextlib
import logging
import os
from datetime import datetime, timedelta
import requests
from aiogram import Bot, Dispatcher
from aiogram.filters import Command
from aiogram.types import KeyboardButton, Message
from aiogram.utils.keyboard import ReplyKeyboardBuilder
from bs4 import BeautifulSoup
from dotenv import load_dotenv
# Загрузка переменных окружения
load_dotenv()
# Настройки
TELEGRAM_TOKEN = os.getenv('TELEGRAM_TOKEN', 'YOUR_TOKEN_HERE')
ALLOWED_CHAT_IDS = list(map(int, os.getenv('ALLOWED_CHAT_IDS', '').split(','))) if os.getenv('ALLOWED_CHAT_IDS') else []
CHECK_INTERVAL = int(os.getenv('CHECK_INTERVAL', '120'))
# Параметры для запроса
VOE_CITY_ID = int(os.getenv('VOE_CITY_ID', 'VOE_CITY_ID'))
VOE_STREET_ID = int(os.getenv('VOE_STREET_ID', 'VOE_STREET_ID'))
VOE_HOUSE_ID = int(os.getenv('VOE_HOUSE_ID', 'VOE_HOUSE_ID'))
# Настройка логирования
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)
# Глобальные переменные
bot = Bot(token=TELEGRAM_TOKEN)
dp = Dispatcher()
last_schedule: list[dict] | None = None
last_notification_time: dict[str, datetime] = {}
# ============================================================================
# УТИЛИТЫ
# ============================================================================
def format_time_duration(minutes: int) -> str:
"""Форматирует время из минут в часы и минуты"""
hours = minutes // 60
mins = minutes % 60
if hours == 0:
return f'{mins}м'
elif mins == 0:
return f'{hours}ч'
return f'{hours}ч {mins}м'
def get_day_statistics(day_blocks: list[dict]) -> dict[str, int]:
"""Получает статистику по дню"""
total_minutes = 0
confirmed_minutes = 0
possible_minutes = 0
for block in day_blocks:
for half in [block['first_half'], block['second_half']]:
if half['status'] == 'off':
total_minutes += 30
if half['confirmed']:
confirmed_minutes += 30
else:
possible_minutes += 30
return {'total': total_minutes, 'confirmed': confirmed_minutes, 'possible': possible_minutes}
# ============================================================================
# ПАРСИНГ ДАННЫХ
# ============================================================================
def parse_html(html: str) -> list[dict]:
"""Парсит HTML с графиком отключений (логика от 15.11.2024)"""
soup = BeautifulSoup(html, 'html.parser')
cells = soup.select('.disconnection-detailed-table-cell.cell')
schedule = []
current_hour = 0
current_day = 0
for cell in cells:
if 'legend' in cell.get('class', []) or 'head' in cell.get('class', []):
continue
cell_classes = cell.get('class', [])
# ПРоверка статуса отключения на весь час
full_hour_off = 'has_disconnection' in cell_classes and 'full_hour' in cell_classes
hour_block = cell.select_one('.hour_block')
if not hour_block:
continue
# Проверка подтверждённости отключения для всего часа
cell_confirmed = None
if 'confirm_1' in cell_classes:
cell_confirmed = True
elif 'confirm_0' in cell_classes:
cell_confirmed = False
# Проверка половин часа
left = hour_block.select_one('.half.left')
right = hour_block.select_one('.half.right')
def parse_half(half, is_full_hour_off: bool, cell_confirmed: bool | None = None) -> dict:
"""Парсит половину часа"""
if not half:
return {'status': 'on', 'queue': None, 'confirmed': None}
half_classes = half.get('class', [])
# Если вся ячейка full_hour - используем статус ячейки
if is_full_hour_off:
return {'status': 'off', 'queue': None, 'confirmed': cell_confirmed}
# Определяем статус половины
if 'has_disconnection' in half_classes:
status = 'off'
elif 'no_disconnection' in half_classes:
status = 'on'
else:
status = 'on' # По умолчанию считаем включенным
# Если выключено - ищем подробности
queue = None
confirmed = None
if status == 'off':
disconnection_div = half.select_one('.disconnection')
if disconnection_div:
# Ищем номер черги в title
if disconnection_div.has_attr('title'):
title = disconnection_div['title']
if 'Номер черги' in title or 'Номер черги:' in title:
with contextlib.suppress(BaseException):
queue = title.split(':')[-1].strip()
# Определяем подтверждение
disc_classes = disconnection_div.get('class', [])
if 'disconnection_confirm_1' in disc_classes:
confirmed = True
elif 'disconnection_confirm_0' in disc_classes:
confirmed = False
return {'status': status, 'queue': queue, 'confirmed': confirmed}
first_half_data = parse_half(left, full_hour_off, cell_confirmed)
second_half_data = parse_half(right, full_hour_off, cell_confirmed)
schedule.append(
{
'hour': current_hour,
'day': current_day,
'first_half': first_half_data,
'second_half': second_half_data,
}
)
current_hour += 1
if current_hour >= 24:
current_hour = 0
current_day += 1
return schedule
def get_voe_html(city_id: int, street_id: int, house_id: int) -> str:
"""Получает HTML с сайта VOE"""
url = 'https://www.voe.com.ua/disconnection/detailed?ajax_form=1&_wrapper_format=drupal_ajax'
headers = {
'Content-Type': 'application/x-www-form-urlencoded; charset=UTF-8',
'X-Requested-With': 'XMLHttpRequest',
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
}
data = {
'search_type': 0,
'city_id': city_id,
'street_id': street_id,
'house_id': house_id,
'form_build_id': 'form-Irv5aHw1R2FT_Ik2apyHOZ47hTH5xPNH_LQnBrmpSTc',
'form_id': 'disconnection_detailed_search_form',
'_triggering_element_name': 'search',
'_triggering_element_value': 'Показати',
'_drupal_ajax': 1,
}
try:
response = requests.post(url, headers=headers, data=data, timeout=10)
response.raise_for_status()
resp_json = response.json()
insert_html = next((item['data'] for item in resp_json if item.get('command') == 'insert'), None)
if not insert_html:
raise ValueError('HTML не найден в ответе')
return insert_html
except requests.exceptions.RequestException as e:
logger.error(f'Ошибка запроса VOE: {e}')
raise
# ============================================================================
# ФОРМАТИРОВАНИЕ СООБЩЕНИЙ
# ============================================================================
def get_main_keyboard():
"""Создает главную клавиатуру"""
builder = ReplyKeyboardBuilder()
builder.row(KeyboardButton(text='📊 Графік'), KeyboardButton(text='🔄 Оновити'))
builder.row(KeyboardButton(text='📅 Сьогодні'), KeyboardButton(text='📅 Завтра'))
builder.row(KeyboardButton(text='ℹ️ Про бота'))
return builder.as_markup(resize_keyboard=True)
def format_schedule_message(schedule: list[dict], days_to_show: int = 2) -> str:
"""Форматирует полный график на несколько дней"""
lines = [
'⚡️ <b>Графік відключень світла</b>',
f'🕐 Оновлено: {datetime.now().strftime("%d.%m.%Y %H:%M:%S")}',
'─' * 30,
'',
]
start_date = datetime.now()
for day in range(min(days_to_show, 2)):
day_blocks = [b for b in schedule if b['day'] == day]
if not day_blocks:
continue
date_str = (start_date + timedelta(days=day)).strftime('%d.%m.%Y')
day_name = '🌅 <b>Сьогодні</b>' if day == 0 else '🌄 <b>Завтра</b>'
lines.append(f'{day_name} ({date_str})')
# Статистика
stats = get_day_statistics(day_blocks)
if stats['total'] > 0:
lines.append(f'⏱ Всього: <code>{format_time_duration(stats["total"])}</code>')
if stats['confirmed'] > 0:
lines.append(f'🔴 Підтверджено: <code>{format_time_duration(stats["confirmed"])}</code>')
if stats['possible'] > 0:
lines.append(f'🟠 Можливо: <code>{format_time_duration(stats["possible"])}</code>')
else:
lines.append('🟢 <b>Відключень немає!</b>')
lines.append('')
# Детальный список отключений
disconnections = []
current_status = None
start_time = None
current_confirmed = None
current_queue = None
for block in day_blocks:
hour = block['hour']
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
time_str = f'{hour:02d}:00' if half_idx == 0 else f'{hour:02d}:30'
if half['status'] == 'off':
if current_status != 'off':
start_time = time_str
current_confirmed = half['confirmed']
current_queue = half['queue']
current_status = 'off'
else:
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч{current_queue})' if current_queue else ''
disconnections.append(f'{icon} <code>{start_time} - {time_str}</code>{queue_text}')
current_status = half['status']
# Если день закончился на отключении
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч{current_queue})' if current_queue else ''
next_hour = (day_blocks[-1]['hour'] + 1) % 24
end_time = f'{next_hour:02d}:00'
disconnections.append(f'{icon} <code>{start_time} - {end_time}</code>{queue_text}')
if disconnections:
for idx, disc in enumerate(disconnections, 1):
lines.append(f'{idx}. {disc}')
lines.append('')
lines.append('<i>🔴 = підтверджено • 🟠 = можливо • 🟢 = світло</i>')
return '\n'.join(lines)
def format_single_day_schedule(schedule: list[dict], day: int) -> str:
"""Форматирует график на один день"""
day_blocks = [b for b in schedule if b['day'] == day]
if not day_blocks:
return '❌ Немає даних для цього дня'
start_date = datetime.now()
date_str = (start_date + timedelta(days=day)).strftime('%d.%m.%Y')
day_name = '🟠 <b>Сьогодні</b>' if day == 0 else '🔶 <b>Завтра</b>'
lines = [f'{day_name} • {date_str}', '']
# Статистика
lines.append('<b>📊 Статистика</b>')
stats = get_day_statistics(day_blocks)
if stats['total'] == 0:
lines.append('└ 🟢 <b>Відключень немає!</b>')
else:
total_time = format_time_duration(stats['total'])
lines.append(f'├ ⏱ Всього: <code>{total_time}</code>')
if stats['confirmed'] > 0:
confirmed_time = format_time_duration(stats['confirmed'])
lines.append(f'├ 🔴 Підтверджено: <code>{confirmed_time}</code>')
if stats['possible'] > 0:
possible_time = format_time_duration(stats['possible'])
lines.append(f'└ 🟠 Можливо: <code>{possible_time}</code>')
else:
lines.append('└ 🟢 Решта часу світло')
lines.append('')
# Детальный список отключений
disconnections = []
current_status = None
start_time = None
current_confirmed = None
current_queue = None
for block in day_blocks:
hour = block['hour']
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
time_str = f'{hour:02d}:00' if half_idx == 0 else f'{hour:02d}:30'
if half['status'] == 'off':
if current_status != 'off':
start_time = time_str
current_confirmed = half['confirmed']
current_queue = half['queue']
current_status = 'off'
else:
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч.{current_queue})' if current_queue else ''
disconnections.append(f'{icon} <code>{start_time} - {time_str}</code>{queue_text}')
current_status = half['status']
# Если день закончился на отключении
if current_status == 'off':
icon = '🔴' if current_confirmed else '🟠'
queue_text = f' (Ч.{current_queue})' if current_queue else ''
next_hour = (day_blocks[-1]['hour'] + 1) % 24
end_time = f'{next_hour:02d}:00'
disconnections.append(f'{icon} <code>{start_time} - {end_time}</code>{queue_text}')
if disconnections:
lines.append('<b>⚡️ Розклад відключень</b>')
for idx, disc in enumerate(disconnections, 1):
lines.append(f'{idx}. {disc}')
lines.append('')
lines.append('<i>🔴 підтверджено • 🟠 можливо • 🟢 світло</i>')
return '\n'.join(lines)
def schedules_differ(old_schedule: list[dict] | None, new_schedule: list[dict] | None) -> bool:
"""Проверяет отличия между графиками"""
if old_schedule is None or new_schedule is None:
return True
if len(old_schedule) != len(new_schedule):
return True
for old, new in zip(old_schedule, new_schedule, strict=False):
if old['day'] >= 2:
break
if old['first_half'] != new['first_half'] or old['second_half'] != new['second_half']:
return True
return False
# ============================================================================
# УВЕДОМЛЕНИЯ
# ============================================================================
async def send_to_all_users(message_text: str, parse_mode: str = 'HTML'):
"""Отправляет сообщение всем пользователям"""
if not ALLOWED_CHAT_IDS:
logger.warning('Нет допущенных ID чатов для отправки уведомлений')
return
for chat_id in ALLOWED_CHAT_IDS:
try:
await bot.send_message(chat_id, message_text, parse_mode=parse_mode)
logger.info(f'✅ Сообщение отправлено пользователю {chat_id}')
except Exception as e:
logger.error(f'❌ Ошибка отправки пользователю {chat_id}: {e}')
await asyncio.sleep(0.5)
async def check_schedule():
"""Проверяет график и отправляет уведомления"""
global last_schedule
try:
logger.info('🔍 Проверка графика...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
new_schedule = parse_html(html)
if schedules_differ(last_schedule, new_schedule):
logger.info('✨ Обнаружены изменения!')
message = format_schedule_message(new_schedule, days_to_show=2)
if last_schedule is not None:
await send_to_all_users(f'🔄 <b>Графік оновлено!</b>\n\n{message}')
last_schedule = new_schedule
else:
logger.info('✓ Графік без змін')
except Exception as e:
logger.error(f'❌ Ошибка при проверке графика: {e}')
async def check_upcoming_disconnections():
"""Проверяет предстоящие события и отправляет предупреждения за 5 минут"""
global last_notification_time
if last_schedule is None:
return
now = datetime.now()
today_blocks = [b for b in last_schedule if b['day'] == 0]
# Создаем список всех переходов (off -> on или on -> off)
transitions = []
prev_status = None
for block in today_blocks:
hour = block['hour']
for half_idx, half in enumerate([block['first_half'], block['second_half']]):
minute = 0 if half_idx == 0 else 30
time_str = f'{hour:02d}:{minute:02d}'
current_status = half['status']
# Если статус изменился - это переход
if prev_status is not None and prev_status != current_status:
transitions.append(
{
'hour': hour,
'minute': minute,
'time_str': time_str,
'from_status': prev_status,
'to_status': current_status,
'confirmed': half.get('confirmed'),
'queue': half.get('queue'),
}
)
prev_status = current_status
# Проверяем переходы
for transition in transitions:
event_time = now.replace(hour=transition['hour'], minute=transition['minute'], second=0, microsecond=0)
time_until = (event_time - now).total_seconds() / 60
notification_key = f'{transition["hour"]}:{transition["minute"]}_{transition["to_status"]}'
# Если за 5 минут до события (±1 минута) и еще не отправляли
if 4 <= time_until <= 6:
# Проверяем, не отправляли ли уже уведомление сегодня
if notification_key in last_notification_time:
last_notif_time = last_notification_time[notification_key]
if last_notif_time.date() == now.date():
continue # Уже отправляли сегодня
# Переход на ОТКЛЮЧЕНИЕ (on -> off)
if transition['from_status'] == 'on' and transition['to_status'] == 'off':
icon = '🔴' if transition['confirmed'] else '🟠'
status = 'підтверджено' if transition['confirmed'] else 'можливе'
queue_info = f' (Черга {transition["queue"]})' if transition['queue'] else ''
warning = (
f'⚠️ <b>УВАГА! ВІДКЛЮЧЕННЯ</b>\n\n'
f'Через ~5 хвилин\n'
f'Час: <code>{transition["time_str"]}</code>\n'
f'Статус: {icon} {status}{queue_info}'
)
await send_to_all_users(warning)
last_notification_time[notification_key] = now
logger.info(f'📢 Відправлено попередження про ВІДКЛЮЧЕННЯ в {transition["time_str"]}')
# Переход на ВКЛЮЧЕНИЕ (off -> on)
elif transition['from_status'] == 'off' and transition['to_status'] == 'on':
warning = (
f'✅ <b>УВАГА! ВКЛЮЧЕННЯ</b>\n\n'
f'Через ~5 хвилин буде світло\n'
f'Час: <code>{transition["time_str"]}</code>'
)
await send_to_all_users(warning)
last_notification_time[notification_key] = now
logger.info(f'📢 Відправлено попередження про ВКЛЮЧЕННЯ в {transition["time_str"]}')
async def monitoring_loop():
"""Основной цикл мониторинга"""
await check_schedule()
while True:
try:
await asyncio.sleep(CHECK_INTERVAL)
await check_schedule()
await check_upcoming_disconnections()
except Exception as e:
logger.error(f'Ошибка в цикле мониторинга: {e}')
await asyncio.sleep(5)
# ============================================================================
# ОБРАБОТЧИКИ КОМАНД
# ============================================================================
@dp.message(Command('start'))
async def cmd_start(message: Message):
"""Обработчик /start"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу до цього бота.')
return
await message.answer(
'👋 <b>Ласкаво просимо!</b>\n\n'
'🤖 <b>Бот для моніторингу графіку відключень світла</b>\n\n'
'✨ <b>Можливості:</b>\n'
'• 📊 Перегляд графіку на сьогодні і завтра\n'
'• 🔔 Автоматичні сповіщення за 5 хвилин до подій\n'
'• 🔄 Моніторинг змін графіку\n\n'
'Використовуйте кнопки нижче 👇',
parse_mode='HTML',
reply_markup=get_main_keyboard(),
)
@dp.message(lambda msg: msg.text == 'ℹ️ Про бота')
async def cmd_info(message: Message):
"""Показывает информацию о боте"""
if message.chat.id not in ALLOWED_CHAT_IDS:
return
await message.answer(
'<b>ℹ️ Про бота</b>\n\n'
'🚀 <b>Версія:</b> 2.2 (Стабільна)\n\n'
'📝 <b>Реліз-ноути:</b>\n'
'├ 15.11.2024: Адаптація під оновлену логіку сайту VOE\n'
'├ Виправлено парсинг half.left та half.right\n'
'├ Покращено визначення підтвердження відключень\n'
'└ Оптимізовано обробку статусу для всієї години\n\n'
'⚡ <b>Функціональність:</b>\n'
'├ Моніторинг графіку 24/7\n'
'├ Сповіщення за 5 хвилин\n'
'├ Детальна статистика дня\n'
'└ Красива візуалізація\n\n'
'🔐 <b>Безпека:</b> Використовуються .env файли\n'
'💾 <b>Джерело:</b> voe.com.ua',
parse_mode='HTML',
reply_markup=get_main_keyboard(),
)
@dp.message(Command('schedule'))
async def cmd_schedule(message: Message):
"""Показывает полный график"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('⏳ Завантаження графіку...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
schedule = parse_html(html)
text = format_schedule_message(schedule, days_to_show=2)
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message(Command('today'))
async def cmd_today(message: Message):
"""Показывает график на сегодня"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('⏳ Завантаження графіку сьогодні...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
schedule = parse_html(html)
text = format_single_day_schedule(schedule, 0)
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message(Command('tomorrow'))
async def cmd_tomorrow(message: Message):
"""Показывает график на завтра"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('⏳ Завантаження графіку завтра...')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
schedule = parse_html(html)
text = format_single_day_schedule(schedule, 1)
await message.answer(text, parse_mode='HTML', reply_markup=get_main_keyboard())
# await message.answer("❌ Функція тимчасово недоступна. Чекаємо на оновлення сайту", parse_mode="HTML", reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message(Command('check'))
async def cmd_check(message: Message):
"""Принудительная проверка графика"""
if message.chat.id not in ALLOWED_CHAT_IDS:
await message.answer('❌ У вас немає доступу.')
return
try:
await message.answer('🔄 <b>Перевіряю графік...</b>', parse_mode='HTML')
html = get_voe_html(VOE_CITY_ID, VOE_STREET_ID, VOE_HOUSE_ID)
new_schedule = parse_html(html)
prefix = (
'✅ <b>Знайдено зміни!</b>\n\n'
if schedules_differ(last_schedule, new_schedule)
else '✓ <b>Графік без змін</b>\n\n'
)
result = prefix + format_schedule_message(new_schedule, days_to_show=2)
await message.answer(result, parse_mode='HTML', reply_markup=get_main_keyboard())
except Exception as e:
await message.answer(f'❌ <b>Помилка:</b> {str(e)}', parse_mode='HTML', reply_markup=get_main_keyboard())
@dp.message()
async def handle_text(message: Message):
"""Обработчик текстовых сообщений и кнопок"""
if message.chat.id not in ALLOWED_CHAT_IDS:
return
text = message.text
# Кнопка "Графік"
if text == '📊 Графік':
await cmd_schedule(message)
# Кнопка "Сьогодні"
elif text == '📅 Сьогодні':
await cmd_today(message)
# Кнопка "Завтра"
elif text == '📅 Завтра':
await cmd_tomorrow(message)
# Кнопка "Оновити"
elif text == '🔄 Оновити':
await cmd_check(message)
# Кнопка "Про бота"
elif text == 'ℹ️ Про бота':
await cmd_info(message)
# Неизвестная команда
else:
await message.answer(
'❓ <b>Команда не розпізнана</b>\n\n'
'Використовуйте кнопки на клавіатурі або команди:\n'
'/start • /today • /tomorrow • /schedule • /check',
parse_mode='HTML',
reply_markup=get_main_keyboard(),
)
# ============================================================================
# ГЛАВНАЯ ФУНКЦИЯ
# ============================================================================
async def main():
"""Главная функция"""
logger.info('=' * 50)
logger.info('ЗАПУСК БОТА V2.2 (stable 2.2, 15.11.2025)')
logger.info('=' * 50)
if not TELEGRAM_TOKEN or os.getenv('TELEGRAM_TOKEN', 'YOUR_TOKEN_HERE') == TELEGRAM_TOKEN:
logger.error('❌ TELEGRAM_TOKEN не конфігурований! Напишіть токен в .env файл')
return
if not ALLOWED_CHAT_IDS:
logger.error('❌ ALLOWED_CHAT_IDS не конфігуровані! Напишіть ID в .env файл')
return
logger.info(f'📌 Allowed chat ids: {ALLOWED_CHAT_IDS}')
logger.info(f'⏱ Інтервал перевірки: {CHECK_INTERVAL} сек')
logger.info('=' * 50)
# Запускаем мониторинг
monitoring_task = asyncio.create_task(monitoring_loop())
try:
await dp.start_polling(bot)
except KeyboardInterrupt:
logger.info('⏹ Бот зупинений користувачем')
finally:
monitoring_task.cancel()
await bot.session.close()
logger.info('✓ Підключення закрито')
if __name__ == '__main__':
try:
asyncio.run(main())
except KeyboardInterrupt:
logger.info('⏹ Завершено')
+7
View File
@@ -0,0 +1,7 @@
[project]
name = "dtek-notif"
version = "0.1.0"
description = "Add your description here"
readme = "README.md"
requires-python = ">=3.13"
dependencies = []
+5
View File
@@ -0,0 +1,5 @@
requests>=2.31.0
beautifulsoup4>=4.12.0
aiogram>=3.3.0
python-dotenv>=1.0.0
aiohttp>=3.9.0
+2 -2
View File
@@ -17,7 +17,7 @@ services:
session-keeper: session-keeper:
build: ./phpsessid-bot build: ./phpsessid-bot
image: gcr.forust.xyz/forust/session-keeper:prod image: gcr.forust.xyz/forust/session-keeper:latest
pull_policy: build pull_policy: build
env_file: .env env_file: .env
restart: unless-stopped restart: unless-stopped
@@ -33,7 +33,7 @@ services:
webinar-checker: webinar-checker:
build: ./webinar-checker build: ./webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:prod image: gcr.forust.xyz/forust/webinar-checker:latest
pull_policy: build pull_policy: build
env_file: .env env_file: .env
restart: unless-stopped restart: unless-stopped
-11
View File
@@ -10,8 +10,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: edu-master-playwright app: edu-master-playwright
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -22,15 +20,6 @@ spec:
# renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker # renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
image: mcr.microsoft.com/playwright:v1.56.0-jammy image: mcr.microsoft.com/playwright:v1.56.0-jammy
imagePullPolicy: IfNotPresent imagePullPolicy: IfNotPresent
# p95 412M, max 478M over 7 days, no limit before. Request is set at p95
# so the pod is not an eviction candidate; the limit stays above 2x the
# request because browser page lifetimes are unpredictable.
resources:
requests:
cpu: "200m"
memory: "416Mi"
limits:
memory: "1Gi"
command: command:
- npx - npx
- -y - -y
+2 -2
View File
@@ -28,10 +28,10 @@ spec:
resources: resources:
requests: requests:
cpu: 25m cpu: 25m
memory: 32Mi memory: 64Mi
limits: limits:
cpu: 250m cpu: 250m
memory: 128Mi memory: 256Mi
readinessProbe: readinessProbe:
exec: exec:
command: ["redis-cli", "ping"] command: ["redis-cli", "ping"]
+2 -6
View File
@@ -1,8 +1,6 @@
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: session-keeper name: session-keeper
namespace: edu-master namespace: edu-master
labels: labels:
@@ -12,8 +10,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: edu-master-session-keeper app: edu-master-session-keeper
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -42,10 +38,10 @@ spec:
resources: resources:
requests: requests:
cpu: 25m cpu: 25m
memory: 32Mi memory: 96Mi
limits: limits:
cpu: 250m cpu: 250m
memory: 128Mi memory: 256Mi
readinessProbe: readinessProbe:
exec: exec:
command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"] command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"]
+2 -6
View File
@@ -1,8 +1,6 @@
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: webinar-checker name: webinar-checker
namespace: edu-master namespace: edu-master
labels: labels:
@@ -12,8 +10,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: edu-master-webinar-checker app: edu-master-webinar-checker
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -71,7 +67,7 @@ spec:
resources: resources:
requests: requests:
cpu: "50m" cpu: "50m"
memory: "192Mi" memory: "128Mi"
limits: limits:
cpu: "600m" cpu: "600m"
memory: "384Mi" memory: "512Mi"
+1 -1
View File
@@ -1,7 +1,7 @@
services: services:
errorpage: errorpage:
build: . build: .
image: gcr.forust.xyz/forust/error-pages:prod image: gcr.forust.xyz/forust/error-pages:latest
pull_policy: build pull_policy: build
container_name: error-pages container_name: error-pages
restart: unless-stopped restart: unless-stopped
-9
View File
@@ -20,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: error-pages app: error-pages
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -30,13 +28,6 @@ spec:
containers: containers:
- name: error-pages - name: error-pages
image: gcr.forust.xyz/forust/error-pages:prod image: gcr.forust.xyz/forust/error-pages:prod
# p95 6M, max 10M, no limit before.
resources:
requests:
cpu: "10m"
memory: "32Mi"
limits:
memory: "128Mi"
ports: ports:
- containerPort: 80 - containerPort: 80
readinessProbe: readinessProbe:
+3 -6
View File
@@ -1,6 +1,6 @@
services: services:
server: server:
image: docker.gitea.com/gitea:28.0.0 image: docker.gitea.com/gitea:1.27.3
container_name: gitea container_name: gitea
restart: always restart: always
environment: environment:
@@ -13,12 +13,9 @@ services:
- GITEA__database__PASSWD=gitea - GITEA__database__PASSWD=gitea
- GITEA__database__NAME=gitea - GITEA__database__NAME=gitea
# Server # Server
- GITEA__server__ROOT_URL=https://git.forust.xyz - GITEA__server__ROOT_URL=https://gitea.forust.xyz
- GITEA__server__SSH_DOMAIN=gitssh.forust.xyz - GITEA__server__SSH_DOMAIN=gitssh.forust.xyz
- GITEA__server__SSH_PORT=2221 - GITEA__server__SSH_PORT=2221
# Pin 28.0 defaults explicitly (see k8s/config.yaml for rationale)
- GITEA__service__DISABLE_REGISTRATION=true
- GITEA__actions__RUN_RETENTION_DAYS=90
# Mailer # Mailer
- GITEA__mailer__ENABLED=true - GITEA__mailer__ENABLED=true
- GITEA__mailer__FROM=${SERVICE_EMAIL} - GITEA__mailer__FROM=${SERVICE_EMAIL}
@@ -37,7 +34,7 @@ services:
- "traefik.http.services.gitea.loadbalancer.server.port=3000" - "traefik.http.services.gitea.loadbalancer.server.port=3000"
# Prod Router # Prod Router
- "traefik.http.routers.gitea.rule=Host(`git.forust.xyz`) || Host(`gitea.forust.xyz`)" - "traefik.http.routers.gitea.rule=Host(`gitea.forust.xyz`)"
- "traefik.http.routers.gitea.entrypoints=websecure" - "traefik.http.routers.gitea.entrypoints=websecure"
- "traefik.http.routers.gitea.tls.certresolver" - "traefik.http.routers.gitea.tls.certresolver"
# Local Router # Local Router
-1
View File
@@ -8,7 +8,6 @@ spec:
dnsNames: dnsNames:
- gcr.forust.xyz - gcr.forust.xyz
- gitea.forust.xyz - gitea.forust.xyz
- git.forust.xyz
issuerRef: issuerRef:
name: letsencrypt-prod name: letsencrypt-prod
kind: ClusterIssuer kind: ClusterIssuer
+2 -11
View File
@@ -4,14 +4,11 @@ metadata:
name: gitea-config name: gitea-config
namespace: gitea namespace: gitea
data: data:
GITEA__server__ROOT_URL: "https://git.forust.xyz" GITEA__server__DOMAIN: "gitea.forust.xyz"
GITEA__server__ROOT_URL: "https://gitea.forust.xyz"
GITEA__server__SSH_DOMAIN: "gitssh.forust.xyz" GITEA__server__SSH_DOMAIN: "gitssh.forust.xyz"
GITEA__server__SSH_PORT: "2221" GITEA__server__SSH_PORT: "2221"
GITEA__service__DISABLE_REGISTRATION: "true"
GITEA__actions__RUN_RETENTION_DAYS: "90"
GITEA__database__DB_TYPE: "postgres" GITEA__database__DB_TYPE: "postgres"
GITEA__database__HOST: "postgres.database.svc.cluster.local:5432" GITEA__database__HOST: "postgres.database.svc.cluster.local:5432"
GITEA__database__NAME: "gitea" GITEA__database__NAME: "gitea"
@@ -20,12 +17,6 @@ data:
GITEA__mailer__ENABLED: "false" GITEA__mailer__ENABLED: "false"
# No code/issue search needed: bleve reindexes the whole issue index on
# every pod restart (cron.rebuild_issue_indexer RUN_AT_START) and hammers
# the rotational disk for an hour. "db" serves issue search from postgres.
GITEA__indexer__ISSUE_INDEXER_TYPE: "db"
GITEA__indexer__REPO_INDEXER_ENABLED: "false"
GITEA__log__logger.access.MODE: "console, file" GITEA__log__logger.access.MODE: "console, file"
USER_UID: "1000" USER_UID: "1000"
USER_GID: "1000" USER_GID: "1000"
+4 -8
View File
@@ -17,8 +17,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: gitea-deployment name: gitea-deployment
namespace: gitea namespace: gitea
spec: spec:
@@ -26,8 +24,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: gitea app: gitea
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -35,7 +31,7 @@ spec:
spec: spec:
containers: containers:
- name: gitea - name: gitea
image: gitea/gitea:28.0.0 image: gitea/gitea:1.27.3
envFrom: envFrom:
- configMapRef: - configMapRef:
name: gitea-config name: gitea-config
@@ -51,10 +47,10 @@ spec:
mountPath: /data mountPath: /data
resources: resources:
requests: requests:
memory: "320Mi" memory: "512Mi"
cpu: "100m" cpu: "300m"
limits: limits:
memory: "1Gi" memory: "1.5Gi"
cpu: "1300m" cpu: "1300m"
volumes: volumes:
- name: gitea-data - name: gitea-data
+15 -2
View File
@@ -7,11 +7,24 @@ spec:
entryPoints: entryPoints:
- websecure - websecure
routes: routes:
- match: Host(`gitea.forust.xyz`) || Host(`git.forust.xyz`) - match: Host(`gitea.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: gitea-service - name: gitea-service
port: 3000 port: 3000
# Registry route: NO crowdsec-bouncer.
# The bouncer plugin does a blocking `GET /v1/decisions` to the LAPI on
# *every* request. A deploy burst (runner Action API polls, `docker
# manifest inspect` per own image, containerd pulls, smoke probes) fires
# hundreds of parallel registry calls; LAPI saturation pushed the lookup
# past the plugin timeout, and the bouncer fail-closed with 403 - which
# containerd surfaces as ErrImagePull/ImagePullBackOff on the next pod.
# This route only serves authenticated OCI traffic (registry tokens,
# basic-auth already handled by gitea) and scanners get nothing useful
# from /v2, so there is no bruteforce surface to protect here.
- match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`) - match: Host(`gcr.forust.xyz`) && PathPrefix(`/v2`)
kind: Rule kind: Rule
services: services:
@@ -29,7 +42,7 @@ spec:
entryPoints: entryPoints:
- websecure - websecure
routes: routes:
- match: (Host(`gitea.workstation.internal`) || Host(`gitea.gigaforust.internal`)) || (Host(`git.workstation.internal`) || Host(`git.gigaforust.internal`)) - match: Host(`gitea.workstation.internal`) || Host(`gitea.gigaforust.internal`)
kind: Rule kind: Rule
services: services:
- name: gitea-service - name: gitea-service
File renamed without changes.
+2 -2
View File
@@ -177,8 +177,8 @@ data:
# url: https://gitssh.forust.xyz # url: https://gitssh.forust.xyz
# - title: gcr.forust.xyz # - title: gcr.forust.xyz
# url: https://gcr.forust.xyz/v2/ # url: https://gcr.forust.xyz/v2/
- title: git.forust.xyz - title: gitea.forust.xyz
url: https://git.forust.xyz url: https://gitea.forust.xyz
- title: nextcloud.forust.xyz - title: nextcloud.forust.xyz
url: https://nextcloud.forust.xyz url: https://nextcloud.forust.xyz
- title: mc.forust.xyz - title: mc.forust.xyz
+2 -6
View File
@@ -13,8 +13,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: glance-deployment name: glance-deployment
namespace: glance namespace: glance
spec: spec:
@@ -22,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: glance app: glance
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -61,10 +57,10 @@ spec:
resources: resources:
requests: requests:
cpu: "50m" cpu: "50m"
memory: "32Mi" memory: "64Mi"
limits: limits:
cpu: "200m" cpu: "200m"
memory: "128Mi" memory: "256Mi"
volumes: volumes:
- name: glance-config - name: glance-config
configMap: configMap:
+8
View File
@@ -23,11 +23,17 @@ spec:
port: 8080 port: 8080
- match: Host(`hs.forust.xyz`) && PathPrefix(`/admin`) - match: Host(`hs.forust.xyz`) && PathPrefix(`/admin`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: headscale-ui-external - name: headscale-ui-external
port: 80 port: 80
- match: Host(`hs.forust.xyz`) && PathPrefix(`/metrics`) - match: Host(`hs.forust.xyz`) && PathPrefix(`/metrics`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: headscale-server-external - name: headscale-server-external
port: 9090 port: 9090
@@ -47,6 +53,8 @@ spec:
kind: Rule kind: Rule
middlewares: middlewares:
- name: headplane-prefix - name: headplane-prefix
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: headplane-external - name: headplane-external
port: 3000 port: 3000
-4
View File
@@ -1,4 +0,0 @@
SECRET_ENCRYPTION_KEY="REPLACE_ME"
TZ="Europe/Bratislava"
PUID="1000"
PGID="1000"
-38
View File
@@ -1,38 +0,0 @@
services:
homarr:
container_name: homarr
image: ghcr.io/homarr-labs/homarr:v2.1.2
restart: unless-stopped
volumes:
- ./appdata:/appdata
- /var/run/docker.sock:/var/run/docker.sock:ro
- ./kubeconfig:/app/config/kubeconfig:ro
env_file: .env
ports:
- 80:7575
- 81:3000
environment:
- TZ=${TZ:-Europe/Bratislava}
- TURBO_TELEMETRY_DISABLED=1
- KUBECONFIG=/app/config/kubeconfig
labels:
- "traefik.enable=true"
- "traefik.http.services.homarr.loadbalancer.server.port=7575"
# Prod Router
- "traefik.http.routers.homarr.rule=Host(`homarr.forust.xyz`)"
- "traefik.http.routers.homarr.entrypoints=websecure"
- "traefik.http.routers.homarr.tls.certresolver=letsencrypt"
# Local Router
- "traefik.http.routers.homarr-local.rule=Host(`homarr.workstation.internal`)"
- "traefik.http.routers.homarr-local.entrypoints=websecure"
- "traefik.http.routers.homarr-local.tls=true"
# Dev Router
- "traefik.http.routers.homarr-dev.rule=Host(`homarr.gigaforust.internal`)"
- "traefik.http.routers.homarr-dev.entrypoints=websecure"
- "traefik.http.routers.homarr-dev.tls=true"
networks:
- proxy
networks:
proxy:
external: true
-28
View File
@@ -1,28 +0,0 @@
# apiVersion: cert-manager.io/v1
# kind: Certificate
# metadata:
# name: home-prod-tls
# namespace: homarr
# spec:
# secretName: home-prod-tls
# dnsNames:
# - home.forust.xyz
# issuerRef:
# name: letsencrypt-prod
# kind: ClusterIssuer
# ---
apiVersion: cert-manager.io/v1
kind: Certificate
metadata:
name: internal-wildcard-tls
namespace: homarr
spec:
secretName: internal-wildcard-tls
dnsNames:
- "*.workstation.internal"
- "*.gigaforust.internal"
- workstation.internal
- gigaforust.internal
issuerRef:
name: internal-ca
kind: ClusterIssuer
-9
View File
@@ -1,9 +0,0 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: homarr-config
namespace: homarr
data:
TZ: "Europe/Bratislava"
TURBO_TELEMETRY_DISABLED: "1"
ENABLE_KUBERNETES: "true"
-83
View File
@@ -1,83 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: homarr-service
namespace: homarr
spec:
selector:
app: homarr
ports:
- port: 7575
targetPort: 7575
---
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: homarr-deployment
namespace: homarr
spec:
replicas: 1
selector:
matchLabels:
app: homarr
strategy:
type: Recreate
template:
metadata:
labels:
app: homarr
spec:
serviceAccountName: homarr
containers:
- name: homarr
image: ghcr.io/homarr-labs/homarr:v2.1.2
envFrom:
- configMapRef:
name: homarr-config
- secretRef:
name: homarr-secrets
ports:
- containerPort: 7575
readinessProbe:
httpGet:
path: /
port: 7575
initialDelaySeconds: 30
periodSeconds: 10
failureThreshold: 6
livenessProbe:
httpGet:
path: /
port: 7575
initialDelaySeconds: 60
periodSeconds: 30
failureThreshold: 3
volumeMounts:
- name: homarr-data
mountPath: /appdata
resources:
requests:
cpu: "250m"
memory: "350Mi"
limits:
cpu: "500m"
memory: "700Mi"
volumes:
- name: homarr-data
persistentVolumeClaim:
claimName: homarr-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: homarr-pvc
namespace: homarr
spec:
resources:
requests:
storage: 2Gi
volumeMode: Filesystem
accessModes:
- ReadWriteOnce
-33
View File
@@ -1,33 +0,0 @@
# apiVersion: traefik.io/v1alpha1
# kind: IngressRoute
# metadata:
# name: homarr-prod
# namespace: homarr
# spec:
# entryPoints:
# - websecure
# routes:
# - match: Host(`home.forust.xyz`)
# kind: Rule
# services:
# - name: homarr-service
# port: 7575
# tls:
# secretName: home-prod-tls
# ---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: homarr-local
namespace: homarr
spec:
entryPoints:
- websecure
routes:
- match: Host(`home.workstation.internal`) || Host(`home.gigaforust.internal`)
kind: Rule
services:
- name: homarr-service
port: 7575
tls:
secretName: internal-wildcard-tls
-4
View File
@@ -1,4 +0,0 @@
apiVersion: v1
kind: Namespace
metadata:
name: homarr
-58
View File
@@ -1,58 +0,0 @@
apiVersion: v1
kind: ServiceAccount
metadata:
name: homarr
namespace: homarr
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
name: homarr-readonly
rules:
- apiGroups: [""]
resources:
- pods
- services
- endpoints
- namespaces
- nodes
- configmaps
- persistentvolumeclaims
- events
verbs: ["get", "list", "watch"]
- apiGroups: ["apps"]
resources:
- deployments
- statefulsets
- daemonsets
- replicasets
verbs: ["get", "list", "watch"]
- apiGroups: ["networking.k8s.io"]
resources:
- ingresses
verbs: ["get", "list", "watch"]
- apiGroups: ["traefik.io"]
resources:
- ingressroutes
- ingressroutetcps
- ingressrouteudps
- middlewares
verbs: ["get", "list", "watch"]
- apiGroups: ["metrics.k8s.io"]
resources:
- pods
- nodes
verbs: ["get", "list"]
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: homarr-readonly
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: homarr-readonly
subjects:
- kind: ServiceAccount
name: homarr
namespace: homarr
-9
View File
@@ -1,9 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: homarr-secrets
namespace: homarr
type: Opaque
stringData:
# openssl rand -hex 32
SECRET_ENCRYPTION_KEY: "REPLACE_ME"
+2 -2
View File
@@ -3,7 +3,7 @@ services:
build: build:
context: . context: .
dockerfile: Dockerfile.forust dockerfile: Dockerfile.forust
image: gcr.forust.xyz/forust/forust-homepage:prod image: gcr.forust.xyz/forust/forust-homepage:latest
pull_policy: build pull_policy: build
# ports: # ports:
# - "8085:80" # - "8085:80"
@@ -35,7 +35,7 @@ services:
build: build:
context: . context: .
dockerfile: Dockerfile.xdfnx dockerfile: Dockerfile.xdfnx
image: gcr.forust.xyz/forust/xdfnx-homepage:prod image: gcr.forust.xyz/forust/xdfnx-homepage:latest
pull_policy: build pull_policy: build
restart: unless-stopped restart: unless-stopped
# ports: # ports:
+1 -1
View File
@@ -174,7 +174,7 @@
<h2>./projects</h2> <h2>./projects</h2>
<ul class="repo-list"> <ul class="repo-list">
<li> <li>
<a href="https://git.forust.xyz/forust/gosleep" target="_blank">forust/gosleep</a> <a href="https://gitea.forust.xyz/forust/gosleep" target="_blank">forust/gosleep</a>
<span class="comment">// linux sleep timer written in rust (originally in go)</span> <span class="comment">// linux sleep timer written in rust (originally in go)</span>
</li> </li>
</ul> </ul>
-31
View File
@@ -1,31 +0,0 @@
# Gateway API PoC for homepages. Lives next to the TLS secrets so
# certificateRefs stay same-namespace and no ReferenceGrant is needed.
# Listener ports must match the Traefik entryPoints (80/443),
# otherwise Traefik marks the listener Invalid.
# Local .internal hosts are deliberately left on IngressRoute,
# only prod is migrated here.
apiVersion: gateway.networking.k8s.io/v1
kind: Gateway
metadata:
name: homepages
namespace: homepages
spec:
gatewayClassName: traefik
listeners:
- name: http
protocol: HTTP
port: 80
allowedRoutes:
namespaces:
from: Same
- name: https
protocol: HTTPS
port: 443
tls:
mode: Terminate
certificateRefs:
- name: forust-homepage-prod-tls
- name: xdfnx-homepage-prod-tls
allowedRoutes:
namespaces:
from: Same
+4 -8
View File
@@ -20,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: forust-homepage app: forust-homepage
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -41,10 +39,10 @@ spec:
failureThreshold: 3 failureThreshold: 3
resources: resources:
requests: requests:
memory: "32Mi" memory: "10Mi"
cpu: "20m" cpu: "20m"
limits: limits:
memory: "128Mi" memory: "100Mi"
cpu: "50m" cpu: "50m"
--- ---
apiVersion: v1 apiVersion: v1
@@ -69,8 +67,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: xdfnx-homepage app: xdfnx-homepage
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -90,8 +86,8 @@ spec:
failureThreshold: 3 failureThreshold: 3
resources: resources:
requests: requests:
memory: "32Mi" memory: "10Mi"
cpu: "20m" cpu: "20m"
limits: limits:
memory: "128Mi" memory: "100Mi"
cpu: "50m" cpu: "50m"
-69
View File
@@ -1,69 +0,0 @@
# PoC: homepages prod hosts via Gateway API.
# Runs alongside k8s/ingress.yaml - delete the prod IngressRoutes only after verification.
# There are no local .internal hosts here, they stay on the local IngressRoute.
# No per-route security middlewares: L3 enforcement moved to the host
# firewall bouncer, so HTTPRoutes stay clean.
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: homepages-http-redirect
namespace: homepages
spec:
parentRefs:
- name: homepages
kind: Gateway
sectionName: http
hostnames:
- forust.xyz
- www.forust.xyz
- xdfnx.cfd
rules:
- filters:
- type: RequestRedirect
requestRedirect:
scheme: https
statusCode: 301
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: forust-homepage-https
namespace: homepages
spec:
parentRefs:
- name: homepages
kind: Gateway
sectionName: https
hostnames:
- forust.xyz
- www.forust.xyz
rules:
- matches:
- path:
type: PathPrefix
value: /
backendRefs:
- name: forust-homepage-service
port: 80
---
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: xdfnx-homepage-https
namespace: homepages
spec:
parentRefs:
- name: homepages
kind: Gateway
sectionName: https
hostnames:
- xdfnx.cfd
rules:
- matches:
- path:
type: PathPrefix
value: /
backendRefs:
- name: xdfnx-homepage-service
port: 80
+6
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`forust.xyz`) || Host(`www.forust.xyz`) - match: Host(`forust.xyz`) || Host(`www.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
priority: 10 priority: 10
services: services:
- name: forust-homepage-service - name: forust-homepage-service
@@ -46,6 +49,9 @@ spec:
routes: routes:
- match: Host(`xdfnx.cfd`) - match: Host(`xdfnx.cfd`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: xdfnx-homepage-service - name: xdfnx-homepage-service
port: 80 port: 80
-24
View File
@@ -1,24 +0,0 @@
# You can find documentation for all the supported env variables at https://docs.immich.app/install/environment-variables
# The location where your uploaded files are stored. The k8s manifests bind
# mount /mnt/immich/library, which is the sdc9 partition - the same place, so
# the two deployment paths look at one library.
UPLOAD_LOCATION=/mnt/immich/library
# The location where your database files are stored. Network shares are not supported for the database
DB_DATA_LOCATION=./postgres
# To set a timezone, uncomment the next line and change Etc/UTC to a TZ identifier from this list: https://en.wikipedia.org/wiki/List_of_tz_database_time_zones#List
# TZ=Etc/UTC
# The Immich version to use. You can pin this to a specific version like "v2.1.0"
IMMICH_VERSION=v3
# Connection secret for postgres. You should change it to a random password
# Please use only the characters `A-Za-z0-9`, without special characters or spaces
DB_PASSWORD=postgres
# The values below this line do not need to be changed
###################################################################################
DB_USERNAME=postgres
DB_DATABASE_NAME=immich
-63
View File
@@ -1,63 +0,0 @@
name: immich
services:
immich-server:
container_name: immich_server
image: ghcr.io/immich-app/immich-server:v3
volumes:
- ${UPLOAD_LOCATION}:/data
- /etc/localtime:/etc/localtime:ro
env_file:
- .env
ports:
- "2283:2283"
depends_on:
- redis
- database
restart: always
healthcheck:
disable: false
immich-machine-learning:
container_name: immich_machine_learning
# For hardware acceleration, add one of -[armnn, cuda, rocm, openvino, rknn] to the image tag.
# Example tag: ${IMMICH_VERSION:-release}-cuda
image: ghcr.io/immich-app/immich-machine-learning:${IMMICH_VERSION:-release}
# extends: # uncomment this section for hardware acceleration - see https://docs.immich.app/features/ml-hardware-acceleration
# file: hwaccel.ml.yml
# service: cpu # set to one of [armnn, cuda, rocm, openvino, openvino-wsl, rknn] for accelerated inference - use the `-wsl` version for WSL2 where applicable
volumes:
- model-cache:/cache
env_file:
- .env
restart: always
healthcheck:
disable: false
redis:
container_name: immich_redis
image: docker.io/valkey/valkey:9@sha256:418652cfb58ef879d4978c33553735d7147016032d5aefaa14c828e611eb9dfd
healthcheck:
test: redis-cli ping | grep -q PONG || exit 1
restart: always
database:
container_name: immich_postgres
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0@sha256:1a078b237c1d9b420b0ee59147386b4aa60d3a07a8e6a402fc84a57e41b043a4
environment:
POSTGRES_PASSWORD: ${DB_PASSWORD}
POSTGRES_USER: ${DB_USERNAME}
POSTGRES_DB: ${DB_DATABASE_NAME}
POSTGRES_INITDB_ARGS: "--data-checksums"
# Uncomment the DB_STORAGE_TYPE: 'HDD' var if your database isn't stored on SSDs
# DB_STORAGE_TYPE: 'HDD'
volumes:
# Do not edit the next line. If you want to change the database storage location on your system, edit the value of DB_DATA_LOCATION in the .env file
- ${DB_DATA_LOCATION}:/var/lib/postgresql/data
shm_size: 128mb
restart: always
healthcheck:
disable: false
volumes:
model-cache:
-24
View File
@@ -1,24 +0,0 @@
apiVersion: v1
kind: ConfigMap
metadata:
name: immich-config
namespace: immich
data:
TZ: "Europe/Bratislava"
# The database in this namespace, not the shared one in the database
# namespace: v3 needs VectorChord, and only the dedicated image carries it.
DB_HOSTNAME: "immich-postgres"
DB_PORT: "5432"
DB_USERNAME: "immich"
DB_DATABASE_NAME: "immich"
DB_SSL_MODE: "disable"
DB_VECTOR_EXTENSION: "vectorchord"
REDIS_HOSTNAME: "immich-valkey"
REDIS_PORT: "6379"
# Traefik is the only client of the server, and it is a pod: the address immich
# sees is inside the node's pod CIDR. Without this the server does not trust
# X-Forwarded-For and every request looks like it came from Traefik itself.
IMMICH_TRUSTED_PROXIES: "10.244.0.0/24"
-97
View File
@@ -1,97 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: immich-service
namespace: immich
spec:
selector:
app: immich
ports:
- name: http
port: 2283
targetPort: 2283
---
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: immich-deployment
namespace: immich
labels:
app: immich
spec:
replicas: 2
selector:
matchLabels:
app: immich
template:
metadata:
labels:
app: immich
spec:
containers:
- name: immich
image: ghcr.io/immich-app/immich-server:v3
envFrom:
- configMapRef:
name: immich-config
- secretRef:
name: immich-secrets
ports:
- name: http
containerPort: 2283
volumeMounts:
- name: immich-data
mountPath: /data
# The first boot runs migrations and warms the transcoder, which can
# take minutes, so liveness has to wait on the startup probe.
startupProbe:
httpGet:
path: /api/server/ping
port: http
failureThreshold: 60
periodSeconds: 10
timeoutSeconds: 5
readinessProbe:
httpGet:
path: /api/server/ping
port: http
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
httpGet:
path: /api/server/ping
port: http
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
# Only the request is scheduled against, and the node is already
# oversubscribed (5.58 of 6 cores requested) while actually running
# at about 1.5. So the request states what this sits at while idle -
# tens of millicores - and the limit leaves room for the burst that
# matters: thumbnails, transcodes and metadata extraction.
#
# The limit used to be 2Gi, but the server OOMKilled on boot while
# chewing through a backlog of unprocessed assets (API +Workers in
# one container spike well past idle).
resources:
requests:
cpu: "100m"
memory: "512Mi"
limits:
cpu: "1500m"
memory: "4Gi"
volumes:
- name: immich-data
# The library lives on the node's own disk, not in a PVC. A PVC here
# meant declaring a size up front for data that does not exist yet,
# on a provisioner that cannot grow it, and the only copy of the
# photos was one `kubectl delete namespace` away.
#
# Directory, not DirectoryOrCreate, on purpose: if sdc9 is not
# mounted, this must fail loudly instead of quietly writing the
# library onto the root filesystem.
hostPath:
path: /mnt/immich/library
type: Directory
-33
View File
@@ -1,33 +0,0 @@
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: immich-prod
namespace: immich
spec:
entryPoints:
- websecure
routes:
- match: Host(`immich.forust.xyz`)
kind: Rule
services:
- name: immich-service
port: 2283
tls:
secretName: immich-prod-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: immich-local
namespace: immich
spec:
entryPoints:
- websecure
routes:
- match: Host(`immich.workstation.internal`) || Host(`immich.gigaforust.internal`)
kind: Rule
services:
- name: immich-service
port: 2283
tls:
secretName: internal-wildcard-tls
-96
View File
@@ -1,96 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: immich-machine-learning
namespace: immich
spec:
selector:
app: immich-machine-learning
ports:
- name: http
port: 3003
targetPort: 3003
---
apiVersion: apps/v1
kind: Deployment
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: immich-machine-learning-deployment
namespace: immich
labels:
app: immich-machine-learning
spec:
replicas: 1
selector:
matchLabels:
app: immich-machine-learning
strategy:
type: Recreate
template:
metadata:
labels:
app: immich-machine-learning
spec:
containers:
- name: immich-machine-learning
image: ghcr.io/immich-app/immich-machine-learning:v3
envFrom:
- configMapRef:
name: immich-config
- secretRef:
name: immich-secrets
ports:
- name: http
containerPort: 3003
volumeMounts:
- name: model-cache
mountPath: /cache
# The first request pulls a model over the internet, so a cold start
# is slower than a container start.
startupProbe:
httpGet:
path: /ping
port: http
failureThreshold: 60
periodSeconds: 5
timeoutSeconds: 5
readinessProbe:
httpGet:
path: /ping
port: http
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
httpGet:
path: /ping
port: http
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
# Same reasoning as the server: the request covers the idle cost
# only, because the node has no spare cores to schedule against.
# Recognition is the burst - a busy import wants both cores.
resources:
requests:
cpu: "100m"
memory: "1Gi"
limits:
cpu: "2000m"
memory: "3Gi"
volumes:
- name: model-cache
persistentVolumeClaim:
claimName: immich-model-cache-pvc
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: immich-model-cache-pvc
namespace: immich
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 2Gi
-5
View File
@@ -1,5 +0,0 @@
# yaml-language-server: $schema=kubernetes
apiVersion: v1
kind: Namespace
metadata:
name: immich
-145
View File
@@ -1,145 +0,0 @@
# Immich's own database, separate from the shared postgres in the database
# namespace. It has to be separate: v3 checks the VectorChord version at startup
# and refuses to boot without it, VectorChord needs its .so in
# shared_preload_libraries, and that can only be read when postmaster starts.
# So the shared instance would have to be rebuilt on a custom image carrying
# vchord and restarted - for every consumer of it (authentik, gitea, netbox,
# netronome, penpot, statuspage). Not worth it for one photo library.
apiVersion: v1
kind: Service
metadata:
name: immich-postgres
namespace: immich
labels:
app: immich-postgres
spec:
selector:
app: immich-postgres
ports:
- name: postgres
port: 5432
targetPort: postgres
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: immich-postgres
namespace: immich
labels:
app: immich-postgres
spec:
serviceName: immich-postgres
replicas: 1
selector:
matchLabels:
app: immich-postgres
template:
metadata:
labels:
app: immich-postgres
spec:
containers:
- name: postgres
# v3.x expects vchord for its vector work and vectors (pgvecto.rs)
# for some index types. This image ships both and preloads them, plus
# its own shared_buffers and wal settings, through
# /etc/postgresql/postgresql.conf - which its entrypoint reaches via
# `postgres -c config_file=...` in the image CMD.
#
# So there is deliberately no `command:` here. Overriding it replaces
# that config_file, and it also loses the step where the entrypoint
# drops from root to the postgres user: postmaster then starts as
# root and refuses to run.
image: ghcr.io/immich-app/postgres:16-vectorchord0.4.3-pgvectors0.2.0@sha256:1a078b237c1d9b420b0ee59147386b4aa60d3a07a8e6a402fc84a57e41b043a4
env:
- name: POSTGRES_USER
value: immich
- name: POSTGRES_DB
value: immich
- name: POSTGRES_PASSWORD
valueFrom:
secretKeyRef:
name: immich-secrets
key: DB_PASSWORD
# Only read when the data directory is empty, so the checksums are
# decided here and never again.
- name: POSTGRES_INITDB_ARGS
value: --data-checksums
# The postgres-data volume lives on sdc, which is rotational. The
# SSD template is the default; HDD only changes the planner costs
# (effective_io_concurrency, random_page_cost), nothing structural.
- name: DB_STORAGE_TYPE
value: HDD
- name: TZ
valueFrom:
configMapKeyRef:
name: immich-config
key: TZ
ports:
- name: postgres
containerPort: 5432
volumeMounts:
- name: postgres-data
mountPath: /var/lib/postgresql/data
# The upstream compose file asks docker for 128mb of shm. Kubernetes
# gives every container 64mb, which is not what postmaster expects
# for parallel query workers and the WAL writer.
- name: shm
mountPath: /dev/shm
# Probes use a generous timeout on purpose: the data lives on a
# rotational disk on a loaded single node, and pg_isready can take
# seconds during WAL recovery. A 1s timeout kills the container
# mid-recovery and restarts the spiral.
#
# Budgets are sized for HDD stalls, not for a healthy disk: fsync of
# a single file was observed taking 70s under node IO pressure, so
# the startup budget is 15 minutes and liveness tolerates 5 minutes
# of unresponsiveness. Killing a stalled-but-healthy postmaster only
# buys another full WAL replay, which is more IO, not less.
startupProbe:
exec:
command: ["sh", "-c", "pg_isready -U immich -d immich"]
failureThreshold: 180
periodSeconds: 5
timeoutSeconds: 5
readinessProbe:
exec:
command: ["sh", "-c", "pg_isready -U immich -d immich"]
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
exec:
command: ["sh", "-c", "pg_isready -U immich -d immich"]
initialDelaySeconds: 30
periodSeconds: 60
timeoutSeconds: 10
failureThreshold: 5
# The image template sets shared_buffers to 512MB, and the vchord and
# vectors workers are Rust binaries with a real RSS footprint on top
# of postmaster, checkpointer and friends. 1Gi was enough to start
# the server but the vectors worker kept dying in it, so the limit
# sits at 2Gi. The request stays at the idle cost.
resources:
requests:
cpu: "50m"
memory: "256Mi"
limits:
cpu: "1000m"
memory: "2Gi"
volumes:
- name: shm
emptyDir:
medium: Memory
sizeLimit: 128Mi
volumeClaimTemplates:
- metadata:
name: postgres-data
spec:
accessModes: ["ReadWriteOnce"]
# Retain: this is the metadata for a library that only exists in one
# place, and local-path cannot expand a bound volume, so this size has
# to hold until the library is rebuilt or dumped elsewhere.
storageClassName: local-path-retain
resources:
requests:
storage: 32Gi
-13
View File
@@ -1,13 +0,0 @@
apiVersion: v1
kind: Secret
metadata:
name: immich-secrets
namespace: immich
type: Opaque
stringData:
# Creates the immich superuser in this namespace's own postgres on first
# boot, and is the same value the server connects with. Nothing outside the
# immich namespace needs it. Letters and digits only: immich reads this into
# a connection string.
DB_PASSWORD: "changeme"
REDIS_PASSWORD: "changeme"
-86
View File
@@ -1,86 +0,0 @@
apiVersion: v1
kind: Service
metadata:
name: immich-valkey
namespace: immich
labels:
app: immich-valkey
spec:
clusterIP: None
selector:
app: immich-valkey
ports:
- name: valkey
port: 6379
targetPort: valkey
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
annotations:
reloader.stakater.com/auto: "true"
name: immich-valkey
namespace: immich
labels:
app: immich-valkey
spec:
serviceName: immich-valkey
replicas: 1
selector:
matchLabels:
app: immich-valkey
template:
metadata:
labels:
app: immich-valkey
spec:
containers:
- name: valkey
image: docker.io/valkey/valkey:9.1.2-alpine
command:
- sh
- -c
- valkey-server --appendonly yes --save 30 1 --loglevel warning --requirepass "$REDIS_PASSWORD"
envFrom:
- secretRef:
name: immich-secrets
ports:
- name: valkey
containerPort: 6379
volumeMounts:
- name: valkey-data
mountPath: /data
# Same reasoning as postgres: 1s probe timeouts flap on a loaded
# single node with rotational storage.
startupProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
failureThreshold: 20
periodSeconds: 5
timeoutSeconds: 5
readinessProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
exec:
command: ["sh", "-c", 'valkey-cli --pass "$REDIS_PASSWORD" ping | grep -q PONG']
initialDelaySeconds: 20
periodSeconds: 20
timeoutSeconds: 5
resources:
requests:
cpu: "25m"
memory: "64Mi"
limits:
cpu: "250m"
memory: "256Mi"
volumeClaimTemplates:
- metadata:
name: valkey-data
spec:
accessModes: ["ReadWriteOnce"]
resources:
requests:
storage: 1Gi
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`status.forust.xyz`) - match: Host(`status.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: kener-service - name: kener-service
port: 3000 port: 3000
-4
View File
@@ -13,8 +13,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: kener-deployment name: kener-deployment
namespace: kener namespace: kener
spec: spec:
@@ -22,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: kener app: kener
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
+4 -18
View File
@@ -7,32 +7,18 @@
controller: controller:
type: daemonset type: daemonset
# config-reloader sidecar: p95 33M, max 43M. The chart keeps it at the top level,
# not under `alloy:`.
configReloader:
resources: resources:
requests: requests:
memory: "32Mi"
cpu: "10m"
limits:
memory: "128Mi" memory: "128Mi"
cpu: "50m"
limits:
memory: "512Mi"
cpu: "500m"
image: image:
tag: "v1.19.2" tag: "v1.19.2"
alloy: alloy:
# p95 275M, max 287M. Alloy tails every pod log and ships it to Loki, so it sits
# on the same IronWolf read path the node is I/O bound on. Request is set at p95.
# The chart key is `alloy.resources`. `controller.resources` is ignored silently,
# which is why this pod shipped with no limits at all.
resources:
requests:
memory: "288Mi"
cpu: "50m"
limits:
memory: "512Mi"
configMap: configMap:
create: true create: true
content: | content: |
+4 -27
View File
@@ -39,16 +39,6 @@ loki:
local: local:
directory: /var/loki/rules directory: /var/loki/rules
# p95 84M, max 85M for the rules sidecar that shares the singleBinary pod.
# The chart exposes it as `sidecar.resources`, shared with any other sidecar.
sidecar:
resources:
requests:
memory: "96Mi"
cpu: "10m"
limits:
memory: "192Mi"
singleBinary: singleBinary:
replicas: 1 replicas: 1
persistence: persistence:
@@ -57,10 +47,10 @@ singleBinary:
storageClass: local-path-retain storageClass: local-path-retain
resources: resources:
requests: requests:
memory: "256Mi" memory: "512Mi"
cpu: "200m" cpu: "200m"
limits: limits:
memory: "1Gi" memory: "2Gi"
cpu: "1000m" cpu: "1000m"
# Zeroed: unused in SingleBinary mode (chart validation requires it). # Zeroed: unused in SingleBinary mode (chart validation requires it).
@@ -73,25 +63,12 @@ backend:
gateway: gateway:
replicas: 1 replicas: 1
# Single node: chart default is required podAntiAffinity on hostname +
# RollingUpdate 25%/25% (effective maxUnavailable=0 at replicas=1).
# That deadlocks the rollout: the new pod stays Unschedulable while the
# old one lives, and the old one never leaves while the new one is not
# Ready. Null clears the default (an empty map would deep-merge with it
# and keep the required rule); maxUnavailable=1 allows a brief gateway
# outage during rollouts instead of a stuck deploy.
affinity: null
deploymentStrategy:
type: RollingUpdate
rollingUpdate:
maxSurge: 1
maxUnavailable: 1
resources: resources:
requests: requests:
memory: "32Mi" memory: "64Mi"
cpu: "50m" cpu: "50m"
limits: limits:
memory: "128Mi" memory: "256Mi"
cpu: "300m" cpu: "300m"
monitoring: monitoring:
+1 -1
View File
@@ -1,6 +1,6 @@
services: services:
metube: metube:
image: ghcr.io/alexta69/metube:2026.09.29 image: ghcr.io/alexta69/metube:2026.09.27
container_name: metube container_name: metube
restart: unless-stopped restart: unless-stopped
# ports: # ports:
+3 -8
View File
@@ -13,8 +13,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: metube-deployment name: metube-deployment
namespace: metube namespace: metube
spec: spec:
@@ -22,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: metube app: metube
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -31,7 +27,7 @@ spec:
spec: spec:
containers: containers:
- name: metube - name: metube
image: ghcr.io/alexta69/metube:2026.09.29 image: ghcr.io/alexta69/metube:2026.09.27
envFrom: envFrom:
- configMapRef: - configMapRef:
name: metube-config name: metube-config
@@ -40,13 +36,12 @@ spec:
volumeMounts: volumeMounts:
- name: downloads - name: downloads
mountPath: /downloads mountPath: /downloads
# p95 72M, max 80M over 7 days. Was 600Mi/2Gi.
resources: resources:
requests: requests:
memory: "96Mi" memory: "600Mi"
cpu: "400m" cpu: "400m"
limits: limits:
memory: "384Mi" memory: "2Gi"
cpu: "1700m" cpu: "1700m"
volumes: volumes:
- name: downloads - name: downloads
+1 -1
View File
@@ -1,6 +1,6 @@
services: services:
n8n: n8n:
image: docker.n8n.io/n8nio/n8n:2.42.3 image: docker.n8n.io/n8nio/n8n:2.41.3
container_name: n8n container_name: n8n
restart: unless-stopped restart: unless-stopped
environment: environment:
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`n8n.forust.xyz`) - match: Host(`n8n.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: n8n-service - name: n8n-service
port: 5678 port: 5678
+1 -5
View File
@@ -13,8 +13,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: n8n-deployment name: n8n-deployment
namespace: n8n namespace: n8n
spec: spec:
@@ -22,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: n8n app: n8n
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -31,7 +27,7 @@ spec:
spec: spec:
containers: containers:
- name: n8n - name: n8n
image: docker.n8n.io/n8nio/n8n:2.42.3 image: docker.n8n.io/n8nio/n8n:2.41.3
envFrom: envFrom:
- configMapRef: - configMapRef:
name: n8n-config name: n8n-config
+2 -2
View File
@@ -2,7 +2,7 @@ name: netbird
services: services:
netbird-server: netbird-server:
image: netbirdio/netbird-server:0.80.0 image: netbirdio/netbird-server:0.79.0
container_name: netbird-server container_name: netbird-server
restart: unless-stopped restart: unless-stopped
environment: environment:
@@ -87,7 +87,7 @@ services:
- proxy - proxy
dashboard: dashboard:
image: netbirdio/dashboard:v2.94.0 image: netbirdio/dashboard:v2.93.0
container_name: netbird-dashboard container_name: netbird-dashboard
restart: unless-stopped restart: unless-stopped
environment: environment:
+9 -9
View File
@@ -7,18 +7,12 @@ spec:
entryPoints: entryPoints:
- websecure - websecure
routes: routes:
# NO crowdsec-bouncer on the API routes. These are the mesh client's own
# endpoints: gRPC-gateway management calls plus signal/relay long-polling,
# authenticated by NetBird's token rather than by a login form. A ban here
# is self-defeating - the client needs the mesh to reach anything else, so
# CrowdSec banning it locks the peer out of the network it needs to
# function. It also backfires: a banned peer keeps retrying, every retry
# is another 403, and LePresidente/http-generic-403-bf turns five 403s in
# ten seconds into a 4h ban, so one 403 loop kept re-arming the ban.
# netbird-local below has always been exempt; this makes prod match.
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`)) - match: Host(`nb.forust.xyz`) && (PathPrefix(`/signalexchange.SignalExchange/`) || PathPrefix(`/management.ManagementService/`) || PathPrefix(`/management.ProxyService/`))
kind: Rule kind: Rule
priority: 100 priority: 100
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: netbird-server-service - name: netbird-server-service
port: 80 port: 80
@@ -26,12 +20,18 @@ spec:
- match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`)) - match: Host(`nb.forust.xyz`) && (PathPrefix(`/relay`) || PathPrefix(`/ws-proxy/`) || PathPrefix(`/api`) || PathPrefix(`/oauth2`))
kind: Rule kind: Rule
priority: 100 priority: 100
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: netbird-server-service - name: netbird-server-service
port: 80 port: 80
- match: Host(`nb.forust.xyz`) - match: Host(`nb.forust.xyz`)
kind: Rule kind: Rule
priority: 1 priority: 1
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: netbird-dashboard-service - name: netbird-dashboard-service
port: 80 port: 80
+7 -16
View File
@@ -32,8 +32,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: netbird-server-deployment name: netbird-server-deployment
namespace: netbird namespace: netbird
spec: spec:
@@ -41,8 +39,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: netbird-server app: netbird-server
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -50,7 +46,7 @@ spec:
spec: spec:
containers: containers:
- name: netbird-server - name: netbird-server
image: netbirdio/netbird-server:0.80.0 image: netbirdio/netbird-server:0.79.0
command: ["/bin/sh", "/opt/netbird/entrypoint.sh", "--config", "/run/netbird/config.yaml"] command: ["/bin/sh", "/opt/netbird/entrypoint.sh", "--config", "/run/netbird/config.yaml"]
envFrom: envFrom:
- configMapRef: - configMapRef:
@@ -92,13 +88,12 @@ spec:
periodSeconds: 30 periodSeconds: 30
timeoutSeconds: 5 timeoutSeconds: 5
failureThreshold: 5 failureThreshold: 5
# p95 97M, max 102M over 7 days. Was 256Mi/1Gi.
resources: resources:
requests: requests:
memory: "128Mi" memory: "256Mi"
cpu: "100m" cpu: "250m"
limits: limits:
memory: "384Mi" memory: "1Gi"
cpu: "1000m" cpu: "1000m"
volumes: volumes:
- name: netbird-data - name: netbird-data
@@ -128,8 +123,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: netbird-dashboard-deployment name: netbird-dashboard-deployment
namespace: netbird namespace: netbird
spec: spec:
@@ -137,8 +130,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: netbird-dashboard app: netbird-dashboard
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -146,7 +137,7 @@ spec:
spec: spec:
containers: containers:
- name: dashboard - name: dashboard
image: netbirdio/dashboard:v2.94.0 image: netbirdio/dashboard:v2.93.0
envFrom: envFrom:
- configMapRef: - configMapRef:
name: netbird-config name: netbird-config
@@ -171,10 +162,10 @@ spec:
failureThreshold: 5 failureThreshold: 5
resources: resources:
requests: requests:
memory: "32Mi" memory: "64Mi"
cpu: "50m" cpu: "50m"
limits: limits:
memory: "128Mi" memory: "256Mi"
cpu: "300m" cpu: "300m"
--- ---
apiVersion: v1 apiVersion: v1
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`netbox.forust.xyz`) - match: Host(`netbox.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: netbox-service - name: netbox-service
port: 8080 port: 8080
+1 -5
View File
@@ -14,8 +14,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: netbox-deployment name: netbox-deployment
namespace: netbox namespace: netbox
labels: labels:
@@ -120,8 +118,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: netbox-worker-deployment name: netbox-worker-deployment
namespace: netbox namespace: netbox
labels: labels:
@@ -168,7 +164,7 @@ spec:
memory: "256Mi" memory: "256Mi"
limits: limits:
cpu: "1" cpu: "1"
memory: "512Mi" memory: "1Gi"
volumes: volumes:
- name: netbox-config - name: netbox-config
configMap: configMap:
-2
View File
@@ -17,8 +17,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: StatefulSet kind: StatefulSet
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: netbox-valkey name: netbox-valkey
namespace: netbox namespace: netbox
labels: labels:
+1 -1
View File
@@ -1,6 +1,6 @@
services: services:
netronome: netronome:
image: ghcr.io/autobrr/netronome:v0.15.0 image: ghcr.io/autobrr/netronome:v0.14.1
restart: unless-stopped restart: unless-stopped
container_name: netronome container_name: netronome
ports: ports:
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`nm.forust.xyz`) - match: Host(`nm.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: netronome-service - name: netronome-service
port: 7575 port: 7575
+3 -7
View File
@@ -14,8 +14,6 @@ spec:
apiVersion: apps/v1 apiVersion: apps/v1
kind: Deployment kind: Deployment
metadata: metadata:
annotations:
reloader.stakater.com/auto: "true"
name: netronome-deployment name: netronome-deployment
namespace: netronome namespace: netronome
labels: labels:
@@ -25,8 +23,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: netronome app: netronome
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
@@ -34,7 +30,7 @@ spec:
spec: spec:
containers: containers:
- name: netronome - name: netronome
image: ghcr.io/autobrr/netronome:v0.15.0 image: ghcr.io/autobrr/netronome:v0.14.1
ports: ports:
- name: netronome-port - name: netronome-port
protocol: TCP protocol: TCP
@@ -55,8 +51,8 @@ spec:
key: NETRONOME__DB_PASSWORD key: NETRONOME__DB_PASSWORD
resources: resources:
requests: requests:
memory: "64Mi" memory: "100Mi"
cpu: "100m" cpu: "100m"
limits: limits:
memory: "256Mi" memory: "512Mi"
cpu: "500m" cpu: "500m"
+4
View File
@@ -12,6 +12,8 @@ spec:
kind: Rule kind: Rule
middlewares: middlewares:
- name: nextcloud-chain@file - name: nextcloud-chain@file
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: nextcloud-apache - name: nextcloud-apache
port: 11000 port: 11000
@@ -30,6 +32,8 @@ spec:
- match: Host(`nextcloud.workstation.internal`) || Host(`nextcloud.gigaforust.internal`) - match: Host(`nextcloud.workstation.internal`) || Host(`nextcloud.gigaforust.internal`)
kind: Rule kind: Rule
middlewares: middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
- name: nextcloud-chain@file - name: nextcloud-chain@file
services: services:
- name: nextcloud-apache - name: nextcloud-apache
File renamed without changes.
+3
View File
@@ -9,6 +9,9 @@ spec:
routes: routes:
- match: Host(`portainer.forust.xyz`) - match: Host(`portainer.forust.xyz`)
kind: Rule kind: Rule
middlewares:
- name: crowdsec-bouncer
namespace: crowdsec
services: services:
- name: portainer-service - name: portainer-service
port: 9000 port: 9000
-2
View File
@@ -20,8 +20,6 @@ spec:
selector: selector:
matchLabels: matchLabels:
app: portainer app: portainer
strategy:
type: Recreate
template: template:
metadata: metadata:
labels: labels:
+1 -11
View File
@@ -60,26 +60,16 @@ spec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"] command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
initialDelaySeconds: 10 initialDelaySeconds: 10
periodSeconds: 10 periodSeconds: 10
# Generous timeout: on an I/O-bound single node even exec can take
# seconds, and a 1s default kills a healthy postgres mid-recovery.
timeoutSeconds: 5
startupProbe: startupProbe:
exec: exec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"] command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
# Crash recovery on an I/O-starved single node can fsync for 10+ failureThreshold: 30
# minutes; killing postgres mid-recovery restarts the fsync from
# zero and loops forever. 90x10s = 15 minutes of grace.
failureThreshold: 90
periodSeconds: 10 periodSeconds: 10
livenessProbe: livenessProbe:
exec: exec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"] command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
initialDelaySeconds: 30 initialDelaySeconds: 30
periodSeconds: 20 periodSeconds: 20
# Same I/O reasoning as readiness, plus more misses before a kill:
# restarting postgres on a loaded node only makes recovery longer.
timeoutSeconds: 5
failureThreshold: 5
resources: resources:
requests: requests:
memory: "512Mi" memory: "512Mi"
Loaded 100 of 311 files, more files were not shown because too many files have changed in this diff. Show more