From ff40a71145bb32c3f987cbec42a4ee25affd74bc Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 15:58:44 +0200 Subject: [PATCH 01/53] fix(deploy): resolve Compose config and check namespaced pod Secrets --- .gitea/tests/deploy-validation.sh | 80 +++++++++++++++++++++++++++ .gitea/workflows/ci.yaml | 1 + .gitea/workflows/compose-lint.sh | 3 +- .gitea/workflows/deploy-lib.sh | 77 +++++++++++++------------- .gitea/workflows/secret-references.jq | 13 +++++ 5 files changed, 133 insertions(+), 41 deletions(-) create mode 100755 .gitea/tests/deploy-validation.sh create mode 100644 .gitea/workflows/secret-references.jq diff --git a/.gitea/tests/deploy-validation.sh b/.gitea/tests/deploy-validation.sh new file mode 100755 index 0000000..b03a429 --- /dev/null +++ b/.gitea/tests/deploy-validation.sh @@ -0,0 +1,80 @@ +#!/usr/bin/env bash +# Local regressions only: kubectl is mocked and Docker is used for config parsing. +set -euo pipefail +repo="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +scratch="$(mktemp -d)" +trap 'rm -rf "$scratch"' EXIT + +mkdir -p "$scratch/repo/app" "$scratch/repo/postgres" "$scratch/repo/netbird" "$scratch/repo/renovate" +git -C "$scratch/repo" init -q +for file in app/compose.yaml postgres/shared-compose.yaml netbird/client.compose.yaml renovate/renovate-compose.yaml; do + touch "$scratch/repo/$file" +done +git -C "$scratch/repo" add . +# shellcheck source=../workflows/compose-lint.sh +source "$repo/.gitea/workflows/compose-lint.sh" +actual="$(cd "$scratch/repo" && compose_files)" +expected=$'app/compose.yaml\nnetbird/client.compose.yaml\npostgres/shared-compose.yaml\nrenovate/renovate-compose.yaml' +[ "$actual" = "$expected" ] || { echo 'Compose discovery missed a file' >&2; exit 1; } + +cat >"$scratch/compose.yaml" <<'YAML' +services: + example: + image: busybox:1.37.0 + environment: + REQUIRED: ${HOMELAB_TEST_REQUIRED:?required for this regression} +YAML +unset HOMELAB_TEST_REQUIRED +if validate_compose_file "$scratch/compose.yaml" >"$scratch/config.log" 2>&1; then + echo 'Full Compose validation accepted a missing variable' >&2 + exit 1 +fi +rg -q 'required for this regression' "$scratch/config.log" +HOMELAB_TEST_REQUIRED=present validate_compose_file "$scratch/compose.yaml" + +cat >"$scratch/resources.json" <<'JSON' +{"kind":"List","items":[ + {"kind":"Deployment","metadata":{"namespace":"app"},"spec":{"template":{"spec":{ + "containers":[{"envFrom":[{"secretRef":{"name":"credentials"}},{"secretRef":{"name":"optional","optional":true}}],"env":[{"valueFrom":{"secretKeyRef":{"name":"credentials","key":"password"}}}]}], + "initContainers":[{"envFrom":[{"secretRef":{"name":"init"}}]}], + "imagePullSecrets":[{"name":"registry"}], + "volumes":[{"secret":{"secretName":"mounted"}},{"projected":{"sources":[{"secret":{"name":"projected"}},{"secret":{"name":"optional-projected","optional":true}}]}}] + }}}}, + {"kind":"CronJob","metadata":{},"spec":{"jobTemplate":{"spec":{"template":{"spec":{"containers":[{"envFrom":[{"secretRef":{"name":"cron"}}]}]}}}}}}, + {"kind":"IngressRoute","metadata":{"namespace":"app"},"spec":{"tls":{"secretName":"controller-issued-tls"}}} +]} +JSON +actual="$(jq -r -f "$repo/.gitea/workflows/secret-references.jq" "$scratch/resources.json" | sort)" +expected=$'app credentials\napp init\napp mounted\napp projected\napp registry\ndefault cron' +[ "$actual" = "$expected" ] || { echo "Unexpected Secret references: $actual" >&2; exit 1; } + +REPO="$repo" +# shellcheck source=../workflows/deploy-lib.sh +source "$repo/.gitea/workflows/deploy-lib.sh" +K8S_MANIFESTS=("$scratch/resources.json") +KUSTOMIZE_APPS=() +# No live cluster access. Reject credentials in app even if they exist elsewhere. +kubectl() { + case "$1" in + create) cat "$scratch/resources.json" ;; + get) + if [ "$3" = credentials ] && [ "$5" = app ]; then + return 1 + fi + return 0 + ;; + *) echo "Unexpected kubectl invocation: $*" >&2; return 1 ;; + esac +} +if check_referenced_secrets >"$scratch/secrets.log"; then + echo 'Namespace-scoped Secret check accepted a missing Secret' >&2 + exit 1 +fi +rg -q 'MISSING OR UNREADABLE: app/credentials' "$scratch/secrets.log" +# API/rendering errors must not produce an empty reference list and pass. +kubectl() { return 1; } +if check_referenced_secrets >"$scratch/secrets.log"; then + echo 'Secret check accepted a failed manifest render' >&2 + exit 1 +fi +printf '%s\n' 'Deploy validation regressions passed.' diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 2480c13..232534d 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -98,6 +98,7 @@ jobs: exit 0 fi shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" + bash .gitea/tests/deploy-validation.sh lint-prettier: runs-on: [self-hosted, linux, arch, homelab] diff --git a/.gitea/workflows/compose-lint.sh b/.gitea/workflows/compose-lint.sh index c27e29f..30e79bb 100644 --- a/.gitea/workflows/compose-lint.sh +++ b/.gitea/workflows/compose-lint.sh @@ -21,8 +21,7 @@ # All committed Compose files, including the ones deploy never starts. compose_files() { git ls-files \ - '*/compose.yaml' '*/compose.yml' 'compose.yaml' 'compose.yml' \ - '*/docker-compose.yaml' '*/docker-compose.yml' + '*compose.yaml' '*compose.yml' } # Prints the flags that turn `docker compose config` into the general check. diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index fa7452e..d96880e 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -686,27 +686,52 @@ stage_preflight() { git -C "$REPO" reset --hard "$target" } +# Required pod Secrets, scoped to the resource namespace. TLS route Secrets are +# created by cert-manager and are not prerequisites for applying a Certificate. +check_referenced_secrets() { + local m k objects refs extracted ns name + local missing=() + refs="" + for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do + objects="$(kubectl create --dry-run=client --validate=false -f "$m" -o json)" || return 1 + extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1 + refs+="$extracted"$'\n' + done + for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do + objects="$(kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json)" || return 1 + extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1 + refs+="$extracted"$'\n' + done + while read -r ns name; do + [ -n "${name:-}" ] || continue + if kubectl get secret "$name" -n "$ns" -o name >/dev/null 2>&1; then + echo " ok: $ns/$name" + else + echo " MISSING OR UNREADABLE: $ns/$name" + missing+=("$ns/$name") + fi + done < <(printf '%s' "$refs" | sort -u) + if [ "${#missing[@]}" -gt 0 ]; then + echo "ERROR: required pod Secrets are missing or unreadable:" + printf ' - %s\n' "${missing[@]}" + echo "Create them in the listed namespaces from the service's secret example." + return 1 + fi +} + stage_validate() { cd "$REPO" select_manifests local m k cf - # Compose .env files and secret files are gitignored by design, so the - # workstation never has real values for the inactive stacks. This stage only - # runs the full check on active stacks; the general structure check for every - # committed Compose file (active or not) lives in the ci workflow, which has no - # .env at all. - # - # Active stacks are still validated with interpolation and env-file resolution - # off, so required-variable guards (:?) and missing local files do not fail the - # deploy. Normalization and consistency checks stay enabled. + # The deploy host has the local .env and secret files. Resolve them here so + # missing configuration fails before either apply job changes workloads. + # CI keeps the structure-only check for inactive stacks. # shellcheck source=compose-lint.sh source "$REPO/.gitea/workflows/compose-lint.sh" - local compose_validate_flags=() - mapfile -t compose_validate_flags < <(compose_safe_flags) log "Validate compose stacks" for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do echo " config: $cf" - validate_compose_file "$cf" ${compose_validate_flags[@]+"${compose_validate_flags[@]}"} + validate_compose_file "$cf" done log "Validate k8s manifests (kubectl dry-run=client)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do @@ -724,33 +749,7 @@ stage_validate() { done log "Checking referenced Secrets exist" echo " (deploy never applies *secret*.yaml; create missing ones manually)" - local ref_secrets=() missing_secrets=() all_secrets s - if [ "${#K8S_MANIFESTS[@]}" -gt 0 ]; then - while IFS= read -r s; do - [ -n "$s" ] && ref_secrets+=("$s") - done < <( - { - grep -h -A1 -E 'secretRef:|secretKeyRef:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true - grep -h -E 'secretName:' "${K8S_MANIFESTS[@]}" 2>/dev/null || true - } | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true - ) - fi - all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)" - for s in ${ref_secrets[@]+"${ref_secrets[@]}"}; do - if printf '%s\n' "$all_secrets" | grep -qx "$s"; then - echo " ok: $s" - else - echo " MISSING: $s" - missing_secrets+=("$s") - fi - done - if [ "${#missing_secrets[@]}" -gt 0 ]; then - echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:" - printf ' - %s\n' "${missing_secrets[@]}" - echo "Create them manually from the laptop, e.g.:" - echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example" - exit 1 - fi + check_referenced_secrets } stage_apply_k8s() { diff --git a/.gitea/workflows/secret-references.jq b/.gitea/workflows/secret-references.jq new file mode 100644 index 0000000..a205f61 --- /dev/null +++ b/.gitea/workflows/secret-references.jq @@ -0,0 +1,13 @@ +# kubectl emits a List for files containing multiple resources. +(if .kind == "List" then .items[] else . end) +| (.metadata.namespace // "default") as $ns +| [ + (.. | objects + | (.secretRef? // empty), (.secretKeyRef? // empty), (.secret? // empty) + | select(.optional != true) + | .name // .secretName // empty), + (.. | objects | .imagePullSecrets[]?.name) + ] +| unique[] +| select(. != null and . != "") +| "\($ns) \(.)" From 8d3185f8abceb86fed0ccfa262e5aa972c24fe33 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 16:31:48 +0200 Subject: [PATCH 02/53] fix(ci): install jq for deploy validation regressions --- .gitea/tests/deploy-validation.sh | 4 ++-- .gitea/workflows/ci.yaml | 2 +- .gitea/workflows/install-ci-tools.sh | 10 ++++++++++ .gitea/workflows/tool-versions.env | 3 +++ 4 files changed, 16 insertions(+), 3 deletions(-) diff --git a/.gitea/tests/deploy-validation.sh b/.gitea/tests/deploy-validation.sh index b03a429..94e6480 100755 --- a/.gitea/tests/deploy-validation.sh +++ b/.gitea/tests/deploy-validation.sh @@ -29,7 +29,7 @@ if validate_compose_file "$scratch/compose.yaml" >"$scratch/config.log" 2>&1; th echo 'Full Compose validation accepted a missing variable' >&2 exit 1 fi -rg -q 'required for this regression' "$scratch/config.log" +grep -q 'required for this regression' "$scratch/config.log" HOMELAB_TEST_REQUIRED=present validate_compose_file "$scratch/compose.yaml" cat >"$scratch/resources.json" <<'JSON' @@ -70,7 +70,7 @@ if check_referenced_secrets >"$scratch/secrets.log"; then echo 'Namespace-scoped Secret check accepted a missing Secret' >&2 exit 1 fi -rg -q 'MISSING OR UNREADABLE: app/credentials' "$scratch/secrets.log" +grep -q 'MISSING OR UNREADABLE: app/credentials' "$scratch/secrets.log" # API/rendering errors must not produce an empty reference list and pass. kubectl() { return 1; } if check_referenced_secrets >"$scratch/secrets.log"; then diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 232534d..b9f7ebc 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -88,7 +88,7 @@ jobs: shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck)" + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" export PATH="$tools_dir:$PATH" mapfile -t scripts < <( git ls-files '*.sh' ':(glob)**/*.bash' diff --git a/.gitea/workflows/install-ci-tools.sh b/.gitea/workflows/install-ci-tools.sh index 988fe64..2fcb3b3 100755 --- a/.gitea/workflows/install-ci-tools.sh +++ b/.gitea/workflows/install-ci-tools.sh @@ -120,6 +120,15 @@ install_shellcheck() { rm -rf "$tmp" } +install_jq() { + if at_version jq "${JQ_VERSION}"; then + return 0 + fi + fetch "https://github.com/jqlang/jq/releases/download/jq-${JQ_VERSION}/jq-linux-${goarch}" \ + "$BIN_DIR/jq" + chmod 0755 "$BIN_DIR/jq" +} + install_uv() { if at_version uv "${UV_VERSION}"; then return 0 @@ -236,6 +245,7 @@ for tool in "${wanted[@]}"; do case "$tool" in kubeconform) install_kubeconform ;; shellcheck) install_shellcheck ;; + jq) install_jq ;; actionlint) install_actionlint ;; prettier) install_prettier ;; ruff) install_ruff ;; diff --git a/.gitea/workflows/tool-versions.env b/.gitea/workflows/tool-versions.env index d010495..cf8edb9 100644 --- a/.gitea/workflows/tool-versions.env +++ b/.gitea/workflows/tool-versions.env @@ -31,3 +31,6 @@ UV_VERSION="0.12.17" # so the tree that gets tested is the tree that gets built. Renovate keeps this # in step with the Dockerfile's node: tag via the "node runtime" group. NODE_VERSION="22.23.3" + +# Secret-reference regression tests parse rendered Kubernetes objects. +JQ_VERSION="1.8.1" From ff83daed1eb35366d7c6cc060c0fc31acb746437 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 16:14:05 +0200 Subject: [PATCH 03/53] feat(reloader): enable deployment and reload runtime-config consumers --- adguardhome/k8s/adguard.yaml | 4 ++-- authentik/k8s/authentik.yaml | 4 ++++ cfddns/k8s/deployment.yaml | 2 ++ checkmk/k8s/checkmk.yaml | 2 ++ cloudflared/k8s/deployment.yaml | 2 ++ converters/k8s/convertx.yaml | 2 ++ edu_master/k8s/session-keeper.yaml | 2 ++ edu_master/k8s/webinar-checker.yaml | 2 ++ gitea/k8s/gitea.yaml | 2 ++ glance/k8s/glance.yaml | 2 ++ homarr/k8s/homarr.yaml | 2 ++ immich/k8s/immich.yaml | 2 ++ immich/k8s/machine-learning.yaml | 2 ++ immich/k8s/valkey.yaml | 2 ++ kener/k8s/kener.yaml | 2 ++ metube/k8s/metube.yaml | 2 ++ n8n/k8s/n8n.yaml | 2 ++ netbird/k8s/netbird.yaml | 4 ++++ netbox/k8s/netbox.yaml | 4 ++++ netbox/k8s/valkey.yaml | 2 ++ netronome/k8s/netronome.yaml | 2 ++ reloader/k8s/active | 0 reloader/k8s/reloader-values.yaml | 8 +++++++- searxng/k8s/searxng.yaml | 2 ++ termix/k8s/termix.yaml | 2 ++ vaultwarden/k8s/vaultwarden.yaml | 2 ++ vpn/xui/k8s/xui.yaml | 2 ++ 27 files changed, 63 insertions(+), 3 deletions(-) create mode 100644 reloader/k8s/active diff --git a/adguardhome/k8s/adguard.yaml b/adguardhome/k8s/adguard.yaml index 290dfc4..40db3ea 100644 --- a/adguardhome/k8s/adguard.yaml +++ b/adguardhome/k8s/adguard.yaml @@ -51,6 +51,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: adguard-deployment namespace: adguard spec: @@ -64,8 +66,6 @@ spec: metadata: labels: app: adguard - annotations: - reloader.stakater.com/auto: "true" spec: containers: - name: adguard diff --git a/authentik/k8s/authentik.yaml b/authentik/k8s/authentik.yaml index 5cfcbdf..2afecb6 100644 --- a/authentik/k8s/authentik.yaml +++ b/authentik/k8s/authentik.yaml @@ -27,6 +27,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: authentik-server-deployment namespace: authentik spec: @@ -63,6 +65,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: authentik-worker-deployment namespace: authentik spec: diff --git a/cfddns/k8s/deployment.yaml b/cfddns/k8s/deployment.yaml index a5598ec..b10f2b0 100644 --- a/cfddns/k8s/deployment.yaml +++ b/cfddns/k8s/deployment.yaml @@ -1,6 +1,8 @@ apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: cfddns labels: app: cfddns diff --git a/checkmk/k8s/checkmk.yaml b/checkmk/k8s/checkmk.yaml index 3fb5fc2..ddb7095 100644 --- a/checkmk/k8s/checkmk.yaml +++ b/checkmk/k8s/checkmk.yaml @@ -17,6 +17,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: checkmk-deployment namespace: checkmk spec: diff --git a/cloudflared/k8s/deployment.yaml b/cloudflared/k8s/deployment.yaml index c164c46..7be3a6c 100644 --- a/cloudflared/k8s/deployment.yaml +++ b/cloudflared/k8s/deployment.yaml @@ -1,6 +1,8 @@ apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: cloudflared labels: app: cloudflared diff --git a/converters/k8s/convertx.yaml b/converters/k8s/convertx.yaml index bc81cc1..d5924b8 100644 --- a/converters/k8s/convertx.yaml +++ b/converters/k8s/convertx.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: convertx-deployment namespace: converters spec: diff --git a/edu_master/k8s/session-keeper.yaml b/edu_master/k8s/session-keeper.yaml index d9f5cda..f3c330d 100644 --- a/edu_master/k8s/session-keeper.yaml +++ b/edu_master/k8s/session-keeper.yaml @@ -1,6 +1,8 @@ apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: session-keeper namespace: edu-master labels: diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml index a8f275e..3c53074 100644 --- a/edu_master/k8s/webinar-checker.yaml +++ b/edu_master/k8s/webinar-checker.yaml @@ -1,6 +1,8 @@ apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: webinar-checker namespace: edu-master labels: diff --git a/gitea/k8s/gitea.yaml b/gitea/k8s/gitea.yaml index 94b20a7..1b75c91 100644 --- a/gitea/k8s/gitea.yaml +++ b/gitea/k8s/gitea.yaml @@ -17,6 +17,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: gitea-deployment namespace: gitea spec: diff --git a/glance/k8s/glance.yaml b/glance/k8s/glance.yaml index b9b2707..14b3dac 100644 --- a/glance/k8s/glance.yaml +++ b/glance/k8s/glance.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: glance-deployment namespace: glance spec: diff --git a/homarr/k8s/homarr.yaml b/homarr/k8s/homarr.yaml index b09e754..e1812b3 100644 --- a/homarr/k8s/homarr.yaml +++ b/homarr/k8s/homarr.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: homarr-deployment namespace: homarr spec: diff --git a/immich/k8s/immich.yaml b/immich/k8s/immich.yaml index d22397d..448e5c1 100644 --- a/immich/k8s/immich.yaml +++ b/immich/k8s/immich.yaml @@ -14,6 +14,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: immich-deployment namespace: immich labels: diff --git a/immich/k8s/machine-learning.yaml b/immich/k8s/machine-learning.yaml index 3ff65a2..1fd7096 100644 --- a/immich/k8s/machine-learning.yaml +++ b/immich/k8s/machine-learning.yaml @@ -14,6 +14,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: immich-machine-learning-deployment namespace: immich labels: diff --git a/immich/k8s/valkey.yaml b/immich/k8s/valkey.yaml index 0a408eb..7e587d6 100644 --- a/immich/k8s/valkey.yaml +++ b/immich/k8s/valkey.yaml @@ -17,6 +17,8 @@ spec: apiVersion: apps/v1 kind: StatefulSet metadata: + annotations: + reloader.stakater.com/auto: "true" name: immich-valkey namespace: immich labels: diff --git a/kener/k8s/kener.yaml b/kener/k8s/kener.yaml index 61b04d0..db3962a 100644 --- a/kener/k8s/kener.yaml +++ b/kener/k8s/kener.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: kener-deployment namespace: kener spec: diff --git a/metube/k8s/metube.yaml b/metube/k8s/metube.yaml index 1f773e7..6caac66 100644 --- a/metube/k8s/metube.yaml +++ b/metube/k8s/metube.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: metube-deployment namespace: metube spec: diff --git a/n8n/k8s/n8n.yaml b/n8n/k8s/n8n.yaml index 585eb41..54804ed 100644 --- a/n8n/k8s/n8n.yaml +++ b/n8n/k8s/n8n.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: n8n-deployment namespace: n8n spec: diff --git a/netbird/k8s/netbird.yaml b/netbird/k8s/netbird.yaml index 550b498..bdd8f2f 100644 --- a/netbird/k8s/netbird.yaml +++ b/netbird/k8s/netbird.yaml @@ -32,6 +32,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: netbird-server-deployment namespace: netbird spec: @@ -126,6 +128,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: netbird-dashboard-deployment namespace: netbird spec: diff --git a/netbox/k8s/netbox.yaml b/netbox/k8s/netbox.yaml index e5a3f08..d5badc7 100644 --- a/netbox/k8s/netbox.yaml +++ b/netbox/k8s/netbox.yaml @@ -14,6 +14,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: netbox-deployment namespace: netbox labels: @@ -118,6 +120,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: netbox-worker-deployment namespace: netbox labels: diff --git a/netbox/k8s/valkey.yaml b/netbox/k8s/valkey.yaml index fc8cef9..503da8a 100644 --- a/netbox/k8s/valkey.yaml +++ b/netbox/k8s/valkey.yaml @@ -17,6 +17,8 @@ spec: apiVersion: apps/v1 kind: StatefulSet metadata: + annotations: + reloader.stakater.com/auto: "true" name: netbox-valkey namespace: netbox labels: diff --git a/netronome/k8s/netronome.yaml b/netronome/k8s/netronome.yaml index b5ba753..5fc558f 100644 --- a/netronome/k8s/netronome.yaml +++ b/netronome/k8s/netronome.yaml @@ -14,6 +14,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: netronome-deployment namespace: netronome labels: diff --git a/reloader/k8s/active b/reloader/k8s/active new file mode 100644 index 0000000..e69de29 diff --git a/reloader/k8s/reloader-values.yaml b/reloader/k8s/reloader-values.yaml index f26caea..ef8ba27 100644 --- a/reloader/k8s/reloader-values.yaml +++ b/reloader/k8s/reloader-values.yaml @@ -1,11 +1,17 @@ # Pinned chart: stakater/reloader 2.2.17 (app v1.4.22). # Deployed by the deploy workflow, namespace reloader. # Restarts pods when a ConfigMap or Secret they consume changes. Opt-in per workload -# via the reloader.stakater.com/auto: "true" pod annotation; watchGlobally because +# via the reloader.stakater.com/auto: "true" workload annotation; watchGlobally because # the workloads that need it are spread across a few dozen namespaces. reloader: watchGlobally: true + # Only opted-in workloads are restarted. Keep scheduled jobs on their schedule. + autoReloadAll: false + ignoreJobs: true + ignoreCronJobs: true + # Change pod-template annotations rather than injecting STAKATER_* env vars. + reloadStrategy: annotations deployment: replicas: 1 diff --git a/searxng/k8s/searxng.yaml b/searxng/k8s/searxng.yaml index cbea1e5..f84664d 100644 --- a/searxng/k8s/searxng.yaml +++ b/searxng/k8s/searxng.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: searxng-deployment namespace: searxng spec: diff --git a/termix/k8s/termix.yaml b/termix/k8s/termix.yaml index b8d9e18..f0d13a0 100644 --- a/termix/k8s/termix.yaml +++ b/termix/k8s/termix.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: termix-deployment namespace: termix spec: diff --git a/vaultwarden/k8s/vaultwarden.yaml b/vaultwarden/k8s/vaultwarden.yaml index f0968f3..b210a67 100644 --- a/vaultwarden/k8s/vaultwarden.yaml +++ b/vaultwarden/k8s/vaultwarden.yaml @@ -13,6 +13,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: vaultwarden-deployment namespace: vaultwarden spec: diff --git a/vpn/xui/k8s/xui.yaml b/vpn/xui/k8s/xui.yaml index 69f02de..103e0c2 100644 --- a/vpn/xui/k8s/xui.yaml +++ b/vpn/xui/k8s/xui.yaml @@ -20,6 +20,8 @@ spec: apiVersion: apps/v1 kind: Deployment metadata: + annotations: + reloader.stakater.com/auto: "true" name: xui-deployment namespace: xui spec: From c098807aa4d7e1495327ef2fdb6f1e475b740a10 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 16:03:53 +0200 Subject: [PATCH 04/53] fix(deploy): reject destructive per-file pruning before apply --- .gitea/workflows/deploy-lib.sh | 19 ++++++++++++++----- 1 file changed, 14 insertions(+), 5 deletions(-) diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index d96880e..d3cb30a 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -31,6 +31,16 @@ warn() { echo "WARNING: $*" >&2 } +# Prune needs the complete desired set in one invocation. Per-file pruning +# treats resources from the other files as absent and can delete them. +check_prune_mode() { + if [ "$APPLY_PRUNE" = "true" ]; then + echo "ERROR: APPLY_PRUNE=true is unsupported by the per-file deploy loop." >&2 + echo "Disable it; remove obsolete resources explicitly after review." >&2 + return 1 + fi +} + collect_k8s() { git -C "$REPO" ls-files -- "$1" \ | grep -E '\.ya?ml$' \ @@ -720,6 +730,7 @@ check_referenced_secrets() { } stage_validate() { + check_prune_mode || return 1 cd "$REPO" select_manifests local m k cf @@ -753,18 +764,16 @@ stage_validate() { } stage_apply_k8s() { + check_prune_mode || return 1 cd "$REPO" select_manifests >/dev/null - local ns_files=() other_files=() m k prune_opts=() + local ns_files=() other_files=() m k for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do case "$m" in */namespace.y?ml) ns_files+=("$m") ;; *) other_files+=("$m") ;; esac done - if [ "$APPLY_PRUNE" = "true" ]; then - prune_opts=(--prune -l app.kubernetes.io/managed-by=homelab-deploy) - fi # Record what is about to change, and publish it for the verify job, before # the first apply. Both are fatal on failure: see snapshot_dir. @@ -789,7 +798,7 @@ stage_apply_k8s() { if [ "${#other_files[@]}" -gt 0 ]; then log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do - if ! render_pinned <"$m" | kubectl apply "${prune_opts[@]}" -f -; then + if ! render_pinned <"$m" | kubectl apply -f -; then echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 exit 1 fi From 67422663b7833dc7bb1c3b73573fe1444b3ccf72 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 16:59:26 +0200 Subject: [PATCH 05/53] fix(ci): avoid duplicate branch and PR runs --- .gitea/workflows/ci.yaml | 2 +- .gitea/workflows/renovate-ci.yaml | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index b9f7ebc..77587bf 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -3,7 +3,7 @@ name: ci on: push: branches: - - "**" + - main pull_request: workflow_dispatch: diff --git a/.gitea/workflows/renovate-ci.yaml b/.gitea/workflows/renovate-ci.yaml index 00dfee9..0886d50 100644 --- a/.gitea/workflows/renovate-ci.yaml +++ b/.gitea/workflows/renovate-ci.yaml @@ -2,9 +2,23 @@ name: renovate-ci on: pull_request: + paths: + - "renovate/**" + - ".gitea/workflows/renovate-ci.yaml" + - ".gitea/workflows/sync-renovate-configmap.sh" + - ".gitea/workflows/compose-lint.sh" + - ".gitea/workflows/install-ci-tools.sh" + - ".gitea/workflows/tool-versions.env" push: branches: - main + paths: + - "renovate/**" + - ".gitea/workflows/renovate-ci.yaml" + - ".gitea/workflows/sync-renovate-configmap.sh" + - ".gitea/workflows/compose-lint.sh" + - ".gitea/workflows/install-ci-tools.sh" + - ".gitea/workflows/tool-versions.env" workflow_dispatch: permissions: From b357ef95d8f9cc925b47d25caa81fde413417b29 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 17:00:05 +0200 Subject: [PATCH 06/53] fix(deploy): only create automatic runs for main CI --- .gitea/workflows/deploy.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 159ba25..40a158b 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -5,6 +5,7 @@ on: # workflow_dispatch so a red lint/validate run can never reach the cluster. workflow_run: workflows: [ci] + branches: [main] types: [completed] workflow_dispatch: From 8729cb50620f487d9e3b2c82fa73fde5a4b38036 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 15:58:44 +0200 Subject: [PATCH 07/53] fix(netbird): restore Compose setup and server entrypoint --- .gitea/workflows/ci.yaml | 1 + netbird/entrypoint.sh | 109 ++++++++++++++++++++++++++++++++++ netbird/setup.sh | 38 ++++++++++++ tests/test_netbird_runtime.py | 87 +++++++++++++++++++++++++++ 4 files changed, 235 insertions(+) create mode 100755 netbird/entrypoint.sh create mode 100755 netbird/setup.sh create mode 100644 tests/test_netbird_runtime.py diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 77587bf..5552327 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -142,6 +142,7 @@ jobs: export PATH="$tools_dir:$PATH" ruff check . ruff format --check . + python3 -m unittest discover -s tests -v lint-yaml: runs-on: [self-hosted, linux, arch, homelab] diff --git a/netbird/entrypoint.sh b/netbird/entrypoint.sh new file mode 100755 index 0000000..fe5420b --- /dev/null +++ b/netbird/entrypoint.sh @@ -0,0 +1,109 @@ +#!/bin/sh +set -eu + +umask 077 + +TEMPLATE_PATH=/opt/netbird/config.template.yaml +RENDERED_PATH=/run/netbird/config.yaml +RELAY_SECRET_PATH=/run/secrets/relay_auth_secret +ENCRYPTION_KEY_PATH=/run/secrets/datastore_encryption_key + +is_valid_proxy_subnet() { + candidate="$1" + case "$candidate" in + 0.0.0.0/0) + return 1 + ;; + */*) + address="${candidate%%/*}" + prefix="${candidate#*/}" + ;; + *) + return 1 + ;; + esac + + case "$prefix" in + 0|[1-9]|[1-2][0-9]|3[0-2]) ;; + *) + return 1 + ;; + esac + + old_ifs="$IFS" + IFS=. + # shellcheck disable=SC2086 + set -- $address + IFS="$old_ifs" + [ "$#" -eq 4 ] || return 1 + + for octet do + case "$octet" in + 0|[1-9]|[1-9][0-9]|1[0-9][0-9]|2[0-4][0-9]|25[0-5]) ;; + *) + return 1 + ;; + esac + done +} + +read_secret() { + secret_path="$1" + + if [ ! -r "$secret_path" ]; then + echo "Required secret is not readable: $secret_path" >&2 + exit 1 + fi + + secret_value="$(cat "$secret_path")" + if [ -z "$secret_value" ]; then + echo "Required secret is empty: $secret_path" >&2 + exit 1 + fi + + printf '%s' "$secret_value" +} + +if [ -z "${NETBIRD_DOMAIN:-}" ]; then + echo "NETBIRD_DOMAIN must be set" >&2 + exit 1 +fi + +case "$NETBIRD_DOMAIN" in + *[!A-Za-z0-9.-]*) + echo "NETBIRD_DOMAIN contains unsupported characters" >&2 + exit 1 + ;; +esac + +if [ -z "${NETBIRD_PROXY_SUBNET:-}" ] || [ "$NETBIRD_PROXY_SUBNET" = "auto" ]; then + echo "NETBIRD_PROXY_SUBNET must be an explicit IPv4 CIDR; run netbird/setup.sh first" >&2 + exit 1 +fi +if ! is_valid_proxy_subnet "$NETBIRD_PROXY_SUBNET"; then + echo "NETBIRD_PROXY_SUBNET must be a non-default IPv4 CIDR, for example 172.20.0.0/16" >&2 + exit 1 +fi + +if [ "$#" -ne 2 ] || [ "$1" != "--config" ] || [ "$2" != "$RENDERED_PATH" ]; then + echo "Expected: --config $RENDERED_PATH" >&2 + exit 1 +fi + +relay_secret="$(read_secret "$RELAY_SECRET_PATH")" +encryption_key="$(read_secret "$ENCRYPTION_KEY_PATH")" + +mkdir -p "$(dirname "$RENDERED_PATH")" +sed \ + -e "s|__NETBIRD_DOMAIN__|${NETBIRD_DOMAIN}|g" \ + -e "s|__NETBIRD_AUTH_SECRET__|${relay_secret}|g" \ + -e "s|__NETBIRD_ENCRYPTION_KEY__|${encryption_key}|g" \ + -e "s|__NETBIRD_PROXY_SUBNET__|${NETBIRD_PROXY_SUBNET}|g" \ + "$TEMPLATE_PATH" >"$RENDERED_PATH" + +if grep -q '__NETBIRD_' "$RENDERED_PATH"; then + echo "Rendered NetBird configuration still contains unresolved placeholders" >&2 + exit 1 +fi + +exec /go/bin/netbird-server "$@" diff --git a/netbird/setup.sh b/netbird/setup.sh new file mode 100755 index 0000000..3b4829b --- /dev/null +++ b/netbird/setup.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# Prepare local Compose configuration without replacing existing credentials. +set -euo pipefail +cd "$(dirname "${BASH_SOURCE[0]}")" +umask 077 +if [ ! -f .env ]; then + cp .env.example .env +fi + +if grep -q '^NETBIRD_PROXY_SUBNET=auto$' .env; then + subnet="$(docker network inspect proxy --format '{{range .IPAM.Config}}{{println .Subnet}}{{end}}' | awk '/^[0-9]+\./ { print; exit }')" + if [ -z "$subnet" ]; then + echo "No IPv4 subnet found on the Docker proxy network. Set NETBIRD_PROXY_SUBNET in .env." >&2 + exit 1 + fi + # The detected value must be safe to substitute into the env file. + if [[ ! "$subnet" =~ ^[0-9.]+/[0-9]+$ ]]; then + echo "Unexpected Docker network subnet: $subnet" >&2 + exit 1 + fi + sed -i "s|^NETBIRD_PROXY_SUBNET=auto$|NETBIRD_PROXY_SUBNET=$subnet|" .env +fi + +mkdir -p secrets +chmod 700 secrets +for name in relay-auth-secret datastore-encryption-key; do + path="secrets/$name" + if [ -e "$path" ]; then + if [ ! -s "$path" ]; then + echo "Existing secret is empty: $path. Restore it before continuing." >&2 + exit 1 + fi + else + openssl rand -base64 32 >"$path" + fi + chmod 600 "$path" +done +printf '%s\n' 'Local files are ready. Review .env, then run docker compose config --quiet.' diff --git a/tests/test_netbird_runtime.py b/tests/test_netbird_runtime.py new file mode 100644 index 0000000..373980e --- /dev/null +++ b/tests/test_netbird_runtime.py @@ -0,0 +1,87 @@ +"""Exercise local setup and config rendering without a Docker daemon.""" + +import os +import shutil +import subprocess +import tempfile +import unittest +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + + +class NetbirdRuntimeTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.stack = self.root / 'netbird' + self.stack.mkdir() + for name in ('setup.sh', '.env.example', 'config.template.yaml'): + shutil.copy(ROOT / 'netbird' / name, self.stack / name) + binary = self.root / 'bin' + binary.mkdir() + docker = binary / 'docker' + docker.write_text('#!/bin/sh\nprintf "%s\\n" 172.20.0.0/16\n') + docker.chmod(0o755) + self.env = dict(os.environ, PATH=f'{binary}:{os.environ["PATH"]}') + + def setup(self): + return subprocess.run( # noqa: S603 - executes the repository script copied into this test's temp dir + ['/bin/bash', str(self.stack / 'setup.sh')], env=self.env, capture_output=True, check=False + ) + + def test_setup_preserves_existing_secrets_and_env(self): + self.assertEqual(self.setup().returncode, 0) + paths = [self.stack / '.env', *sorted((self.stack / 'secrets').iterdir())] + before = [p.read_bytes() for p in paths] + self.assertIn(b'NETBIRD_PROXY_SUBNET=172.20.0.0/16', before[0]) + self.assertEqual(self.setup().returncode, 0) + self.assertEqual(before, [p.read_bytes() for p in paths]) + for p in paths[1:]: + self.assertEqual(p.stat().st_mode & 0o777, 0o600) + + def test_setup_rejects_empty_existing_secret(self): + (self.stack / 'secrets').mkdir() + secret = self.stack / 'secrets/datastore-encryption-key' + secret.touch() + self.assertNotEqual(self.setup().returncode, 0) + self.assertEqual(secret.read_bytes(), b'') + + def render(self, subnet): + self.assertEqual(self.setup().returncode, 0) + rendered = self.root / 'run/config.yaml' + script = (ROOT / 'netbird/entrypoint.sh').read_text() + replacements = { + '/opt/netbird/config.template.yaml': str(self.stack / 'config.template.yaml'), + '/run/netbird/config.yaml': str(rendered), + '/run/secrets/relay_auth_secret': str(self.stack / 'secrets/relay-auth-secret'), + '/run/secrets/datastore_encryption_key': str(self.stack / 'secrets/datastore-encryption-key'), + '/go/bin/netbird-server': '/bin/true', + } + for original, local in replacements.items(): + script = script.replace(original, local) + result = subprocess.run( # noqa: S603 - repository renderer, with test-local paths + ['/bin/sh', '-c', script, 'entrypoint', '--config', str(rendered)], + env=dict(self.env, NETBIRD_DOMAIN='nb.example.com', NETBIRD_PROXY_SUBNET=subnet), + capture_output=True, + check=False, + ) + return result, rendered + + def test_renderer_replaces_placeholders_and_restricts_file_permissions(self): + result, rendered = self.render('172.20.0.0/16') + self.assertEqual(result.returncode, 0, result.stderr) + self.assertNotIn('__NETBIRD_', rendered.read_text()) + self.assertIn('nb.example.com', rendered.read_text()) + self.assertEqual(rendered.stat().st_mode & 0o777, 0o600) + + def test_renderer_rejects_auto_and_default_route(self): + for subnet in ('auto', '0.0.0.0/0', '999.1.1.1/24'): + with self.subTest(subnet=subnet): + result, _ = self.render(subnet) + self.assertNotEqual(result.returncode, 0) + + +if __name__ == '__main__': + unittest.main() From 2fe9b7632f8b96cd7575fa4967b437afb0c3c81c Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 16:05:17 +0200 Subject: [PATCH 08/53] fix(traefik): correct AdGuard and SearXNG Compose router expressions --- adguardhome/compose.yaml | 2 +- searxng/compose.yaml | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/adguardhome/compose.yaml b/adguardhome/compose.yaml index 4ce64d1..69b4e5d 100644 --- a/adguardhome/compose.yaml +++ b/adguardhome/compose.yaml @@ -31,7 +31,7 @@ services: - "traefik.http.routers.adguard-dev.entrypoints=websecure" - "traefik.http.routers.adguard-dev.tls=true" # DoH Router - - "traefik.http.routers.dns-over-https.rule=(Host(`dns.forust.xyz` || Host(`adguard.forust.xyz`)) && PathPrefix(`/dns-query`))" + - "traefik.http.routers.dns-over-https.rule=(Host(`dns.forust.xyz`) || Host(`adguard.forust.xyz`)) && PathPrefix(`/dns-query`)" - "traefik.http.routers.dns-over-https.entrypoints=websecure" - "traefik.http.routers.dns-over-https.tls.certresolver=letsencrypt" diff --git a/searxng/compose.yaml b/searxng/compose.yaml index 89b4c8a..ead695c 100644 --- a/searxng/compose.yaml +++ b/searxng/compose.yaml @@ -12,11 +12,11 @@ services: - "traefik.enable=true" - "traefik.http.services.searxng.loadbalancer.server.port=8080" # Prod Router - - "traefik.http.routers.searxng.rule=Host(`s.forust.xyz` || `search.forust.xyz`)" + - "traefik.http.routers.searxng.rule=Host(`s.forust.xyz`) || Host(`search.forust.xyz`)" - "traefik.http.routers.searxng.entrypoints=websecure" - "traefik.http.routers.searxng.tls.certresolver=letsencrypt" # Local Router - - "traefik.http.routers.searxng-local.rule=Host(`s.workstation.internal` || `searxng.workstation.internal`)" + - "traefik.http.routers.searxng-local.rule=Host(`s.workstation.internal`) || Host(`searxng.workstation.internal`)" - "traefik.http.routers.searxng-local.entrypoints=websecure" - "traefik.http.routers.searxng-local.tls=true" # Dev Router From 8aea0f0d13cd87793d002615c4e0ecae9931829c Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 15:55:50 +0200 Subject: [PATCH 09/53] fix(postgres): include required NetBox password in Compose env example --- postgres/.env.example | 1 + 1 file changed, 1 insertion(+) diff --git a/postgres/.env.example b/postgres/.env.example index c99ea0b..2255fdf 100644 --- a/postgres/.env.example +++ b/postgres/.env.example @@ -1,6 +1,7 @@ POSTGRES_ADMIN_PASSWORD= AUTHENTIK_DB_PASSWORD= GITEA_DB_PASSWORD= +NETBOX_DB_PASSWORD= NETRONOME_DB_PASSWORD= PENPOT_DB_PASSWORD= STATUSPAGE_DB_PASSWORD= From e9a68aae77f3b8b8158f1fdd678eccbe58c684f0 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 15:55:35 +0200 Subject: [PATCH 10/53] fix(glance): mount CSS from the assets ConfigMap --- glance/k8s/glance.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/glance/k8s/glance.yaml b/glance/k8s/glance.yaml index 14b3dac..2b4c456 100644 --- a/glance/k8s/glance.yaml +++ b/glance/k8s/glance.yaml @@ -71,7 +71,7 @@ spec: name: glance-config - name: glance-assets configMap: - name: glance-config + name: glance-assets - name: docker-socket hostPath: path: /var/run/docker.sock From 24f84bab2fcb89d04ffd0cbe5dde41916d02a9ef Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 17:22:06 +0200 Subject: [PATCH 11/53] fix(deploy): skip verification when deployment is disabled --- .gitea/workflows/deploy.yaml | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 40a158b..6c455b8 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -138,9 +138,10 @@ jobs: # lets it start after a failed dependency; the needs on apply-compose are a # barrier, so verification begins only once both applies are done. verify-k8s: - needs: [apply-k8s, apply-compose] + needs: [preflight, apply-k8s, apply-compose] if: >- always() && + needs.preflight.result == 'success' && needs.apply-k8s.result != 'skipped' && needs.apply-compose.result != 'skipped' runs-on: [self-hosted, linux, arch, homelab, prod] @@ -183,8 +184,11 @@ jobs: # suppressing them on a rollback would hide the one run where the answer # matters most. smoke: - needs: [verify-k8s] - if: always() && needs.verify-k8s.result != 'skipped' + needs: [preflight, verify-k8s] + if: >- + always() && + needs.preflight.result == 'success' && + needs.verify-k8s.result != 'skipped' runs-on: [self-hosted, linux, arch, homelab, prod] timeout-minutes: 10 steps: From 3756e60c95c0e1690dce196d6494918503ddcaf6 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 17:18:10 +0200 Subject: [PATCH 12/53] fix(renovate): restore custom extraction and update groups --- renovate/k8s/configmap.yaml | 61 ++++++++++++++++++++-------------- renovate/renovate-compose.yaml | 2 +- renovate/renovate.json | 61 ++++++++++++++++++++-------------- 3 files changed, 73 insertions(+), 51 deletions(-) diff --git a/renovate/k8s/configmap.yaml b/renovate/k8s/configmap.yaml index f839ba5..ce589cd 100644 --- a/renovate/k8s/configmap.yaml +++ b/renovate/k8s/configmap.yaml @@ -29,7 +29,7 @@ data: { "customType": "regex", "description": "singlesource: playwright npm version pinned in npx command (k8s + compose)", - "managerFilePatterns": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"], + "managerFilePatterns": ["edu_master/k8s/playwright.yaml", "edu_master/compose.yaml"], "matchStrings": ["playwright@(?\\d+\\.\\d+\\.\\d+)"], "datasourceTemplate": "npm", "depNameTemplate": "playwright" @@ -37,15 +37,24 @@ data: { "customType": "regex", "description": "singlesource: PLAYWRIGHT_VERSION file", - "managerFilePatterns": ["^edu_master/PLAYWRIGHT_VERSION$"], - "matchStrings": ["^(?\\d+\\.\\d+\\.\\d+)$"], + "managerFilePatterns": ["edu_master/PLAYWRIGHT_VERSION"], + "matchStrings": ["^(?\\d+\\.\\d+\\.\\d+)(?:\\r?\\n)?$"], "datasourceTemplate": "pypi", "depNameTemplate": "playwright" }, + { + "customType": "regex", + "description": "singlesource: playwright Python client version pinned in Dockerfile ARG", + "managerFilePatterns": ["edu_master/webinar-checker/Dockerfile"], + "matchStrings": ["(?:^|\\n)ARG PLAYWRIGHT_VERSION=(?\\d+\\.\\d+\\.\\d+)(?:\\r?\\n|$)"], + "datasourceTemplate": "pypi", + "depNameTemplate": "playwright", + "versioningTemplate": "pep440" + }, { "customType": "regex", "description": "kube-prometheus-stack chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|prometheus-community/kube-prometheus-stack\\|prometheus\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "kube-prometheus-stack", @@ -54,7 +63,7 @@ data: { "customType": "regex", "description": "grafana/loki chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|grafana/loki\\|prometheus\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "loki", @@ -63,7 +72,7 @@ data: { "customType": "regex", "description": "grafana/alloy chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|grafana/alloy\\|prometheus\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "alloy", @@ -72,7 +81,7 @@ data: { "customType": "regex", "description": "actionlint version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)ACTIONLINT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "rhysd/actionlint" @@ -80,7 +89,7 @@ data: { "customType": "regex", "description": "shellcheck version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)SHELLCHECK_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "koalaman/shellcheck" @@ -88,7 +97,7 @@ data: { "customType": "regex", "description": "kubeconform version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)KUBECONFORM_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "yannh/kubeconform" @@ -96,7 +105,7 @@ data: { "customType": "regex", "description": "uv version used to build the pytest venv", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)UV_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "astral-sh/uv" @@ -104,7 +113,7 @@ data: { "customType": "regex", "description": "prettier version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)PRETTIER_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "npm", "depNameTemplate": "prettier" @@ -112,7 +121,7 @@ data: { "customType": "regex", "description": "ruff version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)RUFF_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "pypi", "depNameTemplate": "ruff" @@ -120,7 +129,7 @@ data: { "customType": "regex", "description": "pip-audit version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)PIP_AUDIT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "pypi", "depNameTemplate": "pip-audit" @@ -128,7 +137,7 @@ data: { "customType": "regex", "description": "yamllint version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)YAMLLINT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "pypi", "depNameTemplate": "yamllint" @@ -136,7 +145,7 @@ data: { "customType": "regex", "description": "hadolint version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)HADOLINT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "hadolint/hadolint" @@ -144,7 +153,7 @@ data: { "customType": "regex", "description": "node version the ci workflow runs npm with", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)NODE_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "node", "depNameTemplate": "node" @@ -152,7 +161,7 @@ data: { "customType": "regex", "description": "stakater/reloader chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|stakater/reloader\\|reloader\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "reloader", @@ -161,7 +170,7 @@ data: ], "packageRules": [ { - "description": "Automerge digest and patch updates - safe by definition, review adds nothing, keeps the renovate queue and the deploy line short. Specific no-automerge rules below still override this for playwright, helm and majors.", + "description": "Automerge ordinary digest and patch updates after successful checks; specific manual-review rules below override this.", "matchUpdateTypes": ["digest", "patch"], "automerge": true }, @@ -172,6 +181,12 @@ data: "groupSlug": "all-minor", "automerge": false }, + { + "description": "Group ordinary patch updates; the specific groups and manual-review rules below take precedence", + "matchUpdateTypes": ["patch"], + "groupName": "all patch updates", + "groupSlug": "all-patch" + }, { "description": "Keep private homelab images unchanged", "matchDatasources": ["docker"], @@ -196,7 +211,7 @@ data: "automerge": false }, { - "description": "CI runs npm on the node the panel image is built from - the NODE_VERSION pin in tool-versions.env and node:22-alpine in the Dockerfile are the same dependency and move as one", + "description": "Keep CI Node runtime updates in a separate, manually reviewed group", "matchPackageNames": ["node"], "groupName": "node runtime", "groupSlug": "node", @@ -205,6 +220,8 @@ data: { "description": "Helm chart bumps change PVC fields and admission behaviour, keep them reviewable", "matchDatasources": ["helm"], + "groupName": "Helm chart {{depName}}", + "groupSlug": "helm-{{depName}}", "automerge": false }, { @@ -213,12 +230,6 @@ data: "dependencyDashboardApproval": true, "automerge": false }, - { - "description": "Group patch updates from all sources - automerge still applies via the digest/patch rule above (helm/playwright stay manual via their own rules)", - "matchUpdateTypes": ["patch"], - "groupName": "all patch updates", - "groupSlug": "all-patch" - }, { "description": "Python Y-bumps break compat (3.11->3.12->3.13->3.14) - keep the base image out of the shared minor/patch groups, review every bump separately. Placed last so its groupName wins.", "matchDatasources": ["docker"], diff --git a/renovate/renovate-compose.yaml b/renovate/renovate-compose.yaml index 0d63628..f0f5442 100644 --- a/renovate/renovate-compose.yaml +++ b/renovate/renovate-compose.yaml @@ -2,7 +2,7 @@ services: renovate: # Kept in step with renovate/k8s/cronjob.yaml by the "renovate self-update" # package rule in renovate/renovate.json. - image: renovate/renovate:44.115.9 + image: renovate/renovate:44.136.0 container_name: renovate restart: "no" env_file: diff --git a/renovate/renovate.json b/renovate/renovate.json index c74bbd4..79df300 100644 --- a/renovate/renovate.json +++ b/renovate/renovate.json @@ -18,7 +18,7 @@ { "customType": "regex", "description": "singlesource: playwright npm version pinned in npx command (k8s + compose)", - "managerFilePatterns": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"], + "managerFilePatterns": ["edu_master/k8s/playwright.yaml", "edu_master/compose.yaml"], "matchStrings": ["playwright@(?\\d+\\.\\d+\\.\\d+)"], "datasourceTemplate": "npm", "depNameTemplate": "playwright" @@ -26,15 +26,24 @@ { "customType": "regex", "description": "singlesource: PLAYWRIGHT_VERSION file", - "managerFilePatterns": ["^edu_master/PLAYWRIGHT_VERSION$"], - "matchStrings": ["^(?\\d+\\.\\d+\\.\\d+)$"], + "managerFilePatterns": ["edu_master/PLAYWRIGHT_VERSION"], + "matchStrings": ["^(?\\d+\\.\\d+\\.\\d+)(?:\\r?\\n)?$"], "datasourceTemplate": "pypi", "depNameTemplate": "playwright" }, + { + "customType": "regex", + "description": "singlesource: playwright Python client version pinned in Dockerfile ARG", + "managerFilePatterns": ["edu_master/webinar-checker/Dockerfile"], + "matchStrings": ["(?:^|\\n)ARG PLAYWRIGHT_VERSION=(?\\d+\\.\\d+\\.\\d+)(?:\\r?\\n|$)"], + "datasourceTemplate": "pypi", + "depNameTemplate": "playwright", + "versioningTemplate": "pep440" + }, { "customType": "regex", "description": "kube-prometheus-stack chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|prometheus-community/kube-prometheus-stack\\|prometheus\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "kube-prometheus-stack", @@ -43,7 +52,7 @@ { "customType": "regex", "description": "grafana/loki chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|grafana/loki\\|prometheus\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "loki", @@ -52,7 +61,7 @@ { "customType": "regex", "description": "grafana/alloy chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|grafana/alloy\\|prometheus\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "alloy", @@ -61,7 +70,7 @@ { "customType": "regex", "description": "actionlint version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)ACTIONLINT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "rhysd/actionlint" @@ -69,7 +78,7 @@ { "customType": "regex", "description": "shellcheck version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)SHELLCHECK_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "koalaman/shellcheck" @@ -77,7 +86,7 @@ { "customType": "regex", "description": "kubeconform version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)KUBECONFORM_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "yannh/kubeconform" @@ -85,7 +94,7 @@ { "customType": "regex", "description": "uv version used to build the pytest venv", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)UV_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "astral-sh/uv" @@ -93,7 +102,7 @@ { "customType": "regex", "description": "prettier version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)PRETTIER_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "npm", "depNameTemplate": "prettier" @@ -101,7 +110,7 @@ { "customType": "regex", "description": "ruff version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)RUFF_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "pypi", "depNameTemplate": "ruff" @@ -109,7 +118,7 @@ { "customType": "regex", "description": "pip-audit version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)PIP_AUDIT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "pypi", "depNameTemplate": "pip-audit" @@ -117,7 +126,7 @@ { "customType": "regex", "description": "yamllint version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)YAMLLINT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "pypi", "depNameTemplate": "yamllint" @@ -125,7 +134,7 @@ { "customType": "regex", "description": "hadolint version used by the ci workflow", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)HADOLINT_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "github-tags", "depNameTemplate": "hadolint/hadolint" @@ -133,7 +142,7 @@ { "customType": "regex", "description": "node version the ci workflow runs npm with", - "managerFilePatterns": ["^\\.gitea/workflows/tool-versions\\.env$"], + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], "matchStrings": ["(?:^|\\n)NODE_VERSION=\"(?[0-9.]+)\""], "datasourceTemplate": "node", "depNameTemplate": "node" @@ -141,7 +150,7 @@ { "customType": "regex", "description": "stakater/reloader chart version pinned in the deploy workflow", - "managerFilePatterns": ["^\\.gitea/workflows/deploy-lib\\.sh$"], + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], "matchStrings": ["\\|stakater/reloader\\|reloader\\|(?[0-9.]+)\\|"], "datasourceTemplate": "helm", "depNameTemplate": "reloader", @@ -150,7 +159,7 @@ ], "packageRules": [ { - "description": "Automerge digest and patch updates - safe by definition, review adds nothing, keeps the renovate queue and the deploy line short. Specific no-automerge rules below still override this for playwright, helm and majors.", + "description": "Automerge ordinary digest and patch updates after successful checks; specific manual-review rules below override this.", "matchUpdateTypes": ["digest", "patch"], "automerge": true }, @@ -161,6 +170,12 @@ "groupSlug": "all-minor", "automerge": false }, + { + "description": "Group ordinary patch updates; the specific groups and manual-review rules below take precedence", + "matchUpdateTypes": ["patch"], + "groupName": "all patch updates", + "groupSlug": "all-patch" + }, { "description": "Keep private homelab images unchanged", "matchDatasources": ["docker"], @@ -185,7 +200,7 @@ "automerge": false }, { - "description": "CI runs npm on the node the panel image is built from - the NODE_VERSION pin in tool-versions.env and node:22-alpine in the Dockerfile are the same dependency and move as one", + "description": "Keep CI Node runtime updates in a separate, manually reviewed group", "matchPackageNames": ["node"], "groupName": "node runtime", "groupSlug": "node", @@ -194,6 +209,8 @@ { "description": "Helm chart bumps change PVC fields and admission behaviour, keep them reviewable", "matchDatasources": ["helm"], + "groupName": "Helm chart {{depName}}", + "groupSlug": "helm-{{depName}}", "automerge": false }, { @@ -202,12 +219,6 @@ "dependencyDashboardApproval": true, "automerge": false }, - { - "description": "Group patch updates from all sources - automerge still applies via the digest/patch rule above (helm/playwright stay manual via their own rules)", - "matchUpdateTypes": ["patch"], - "groupName": "all patch updates", - "groupSlug": "all-patch" - }, { "description": "Python Y-bumps break compat (3.11->3.12->3.13->3.14) - keep the base image out of the shared minor/patch groups, review every bump separately. Placed last so its groupName wins.", "matchDatasources": ["docker"], From 552b22cb6681e18e0c6465d7aefb05917ccd8fbe Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 17:22:39 +0200 Subject: [PATCH 13/53] fix(renovate): include self-update Compose filename --- renovate/k8s/configmap.yaml | 3 +++ renovate/renovate.json | 3 +++ 2 files changed, 6 insertions(+) diff --git a/renovate/k8s/configmap.yaml b/renovate/k8s/configmap.yaml index ce589cd..b5a70a2 100644 --- a/renovate/k8s/configmap.yaml +++ b/renovate/k8s/configmap.yaml @@ -19,6 +19,9 @@ data: "dependencyDashboard": true, "prCreation": "immediate", "labels": ["dependencies", "automated"], + "docker-compose": { + "managerFilePatterns": ["renovate/renovate-compose.yaml"] + }, "helm-values": { "managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"] }, diff --git a/renovate/renovate.json b/renovate/renovate.json index 79df300..b65f8bd 100644 --- a/renovate/renovate.json +++ b/renovate/renovate.json @@ -8,6 +8,9 @@ "dependencyDashboard": true, "prCreation": "immediate", "labels": ["dependencies", "automated"], + "docker-compose": { + "managerFilePatterns": ["renovate/renovate-compose.yaml"] + }, "helm-values": { "managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"] }, From faac6febb8bce2dcefb944235c217e57fcca8c1d Mon Sep 17 00:00:00 2001 From: renovate-bot Date: Tue, 6 Oct 2026 10:19:09 +0000 Subject: [PATCH 14/53] chore(deps): update all patch updates --- termix/compose.yaml | 2 +- termix/k8s/termix.yaml | 2 +- vaultwarden/compose.yaml | 2 +- vaultwarden/k8s/vaultwarden.yaml | 2 +- 4 files changed, 4 insertions(+), 4 deletions(-) diff --git a/termix/compose.yaml b/termix/compose.yaml index 1e59453..a595b5b 100644 --- a/termix/compose.yaml +++ b/termix/compose.yaml @@ -1,6 +1,6 @@ services: termix: - image: ghcr.io/lukegus/termix:2.9.1 + image: ghcr.io/lukegus/termix:2.9.2 container_name: termix restart: unless-stopped # ports: diff --git a/termix/k8s/termix.yaml b/termix/k8s/termix.yaml index f0d13a0..f049568 100644 --- a/termix/k8s/termix.yaml +++ b/termix/k8s/termix.yaml @@ -31,7 +31,7 @@ spec: spec: containers: - name: termix - image: ghcr.io/lukegus/termix:2.9.1 + image: ghcr.io/lukegus/termix:2.9.2 envFrom: - configMapRef: name: termix-config diff --git a/vaultwarden/compose.yaml b/vaultwarden/compose.yaml index 70b52fa..b6d9ab1 100644 --- a/vaultwarden/compose.yaml +++ b/vaultwarden/compose.yaml @@ -1,7 +1,7 @@ services: server: container_name: vaultwarden-server - image: vaultwarden/server:1.37.3 + image: vaultwarden/server:1.37.4 restart: unless-stopped ports: - 9993:80 diff --git a/vaultwarden/k8s/vaultwarden.yaml b/vaultwarden/k8s/vaultwarden.yaml index b210a67..0540aac 100644 --- a/vaultwarden/k8s/vaultwarden.yaml +++ b/vaultwarden/k8s/vaultwarden.yaml @@ -31,7 +31,7 @@ spec: spec: containers: - name: vaultwarden - image: vaultwarden/server:1.37.3 + image: vaultwarden/server:1.37.4 ports: - containerPort: 80 envFrom: From 506c04e15c12bf5345d65fe1022fe5dae8db8efc Mon Sep 17 00:00:00 2001 From: renovate-bot Date: Tue, 6 Oct 2026 10:19:13 +0000 Subject: [PATCH 15/53] chore(deps): update all minor updates --- cloudflared/k8s/deployment.yaml | 2 +- edu_master/phpsessid-bot/Dockerfile | 2 +- edu_master/webinar-checker/Dockerfile | 2 +- homarr/compose.yaml | 2 +- homarr/k8s/homarr.yaml | 2 +- n8n/compose.yaml | 2 +- n8n/k8s/n8n.yaml | 2 +- netronome/compose.yaml | 2 +- netronome/k8s/netronome.yaml | 2 +- vpn/xui/k8s/xui.yaml | 2 +- 10 files changed, 10 insertions(+), 10 deletions(-) diff --git a/cloudflared/k8s/deployment.yaml b/cloudflared/k8s/deployment.yaml index 7be3a6c..03bc695 100644 --- a/cloudflared/k8s/deployment.yaml +++ b/cloudflared/k8s/deployment.yaml @@ -20,7 +20,7 @@ spec: spec: containers: - name: cloudflared - image: cloudflare/cloudflared:2026.9.3 + image: cloudflare/cloudflared:2026.10.0 imagePullPolicy: IfNotPresent args: - tunnel diff --git a/edu_master/phpsessid-bot/Dockerfile b/edu_master/phpsessid-bot/Dockerfile index 18a7a51..2c9f2c1 100644 --- a/edu_master/phpsessid-bot/Dockerfile +++ b/edu_master/phpsessid-bot/Dockerfile @@ -1,4 +1,4 @@ -FROM python:3.11-slim +FROM python:3.14-slim WORKDIR /app diff --git a/edu_master/webinar-checker/Dockerfile b/edu_master/webinar-checker/Dockerfile index 3f8d5fa..9edab9f 100644 --- a/edu_master/webinar-checker/Dockerfile +++ b/edu_master/webinar-checker/Dockerfile @@ -1,4 +1,4 @@ -FROM python:3.11-slim +FROM python:3.14-slim WORKDIR /app diff --git a/homarr/compose.yaml b/homarr/compose.yaml index 527b488..bf7983f 100644 --- a/homarr/compose.yaml +++ b/homarr/compose.yaml @@ -1,7 +1,7 @@ services: homarr: container_name: homarr - image: ghcr.io/homarr-labs/homarr:v2.1.2 + image: ghcr.io/homarr-labs/homarr:v2.2.0 restart: unless-stopped volumes: - ./appdata:/appdata diff --git a/homarr/k8s/homarr.yaml b/homarr/k8s/homarr.yaml index e1812b3..460527e 100644 --- a/homarr/k8s/homarr.yaml +++ b/homarr/k8s/homarr.yaml @@ -32,7 +32,7 @@ spec: serviceAccountName: homarr containers: - name: homarr - image: ghcr.io/homarr-labs/homarr:v2.1.2 + image: ghcr.io/homarr-labs/homarr:v2.2.0 envFrom: - configMapRef: name: homarr-config diff --git a/n8n/compose.yaml b/n8n/compose.yaml index 9f74296..d13787d 100644 --- a/n8n/compose.yaml +++ b/n8n/compose.yaml @@ -1,6 +1,6 @@ services: n8n: - image: docker.n8n.io/n8nio/n8n:2.42.3 + image: docker.n8n.io/n8nio/n8n:2.43.0 container_name: n8n restart: unless-stopped environment: diff --git a/n8n/k8s/n8n.yaml b/n8n/k8s/n8n.yaml index 54804ed..03da11e 100644 --- a/n8n/k8s/n8n.yaml +++ b/n8n/k8s/n8n.yaml @@ -31,7 +31,7 @@ spec: spec: containers: - name: n8n - image: docker.n8n.io/n8nio/n8n:2.42.3 + image: docker.n8n.io/n8nio/n8n:2.43.0 envFrom: - configMapRef: name: n8n-config diff --git a/netronome/compose.yaml b/netronome/compose.yaml index 2a47a5c..cd561b4 100644 --- a/netronome/compose.yaml +++ b/netronome/compose.yaml @@ -1,6 +1,6 @@ services: netronome: - image: ghcr.io/autobrr/netronome:v0.15.0 + image: ghcr.io/autobrr/netronome:v0.16.0 restart: unless-stopped container_name: netronome ports: diff --git a/netronome/k8s/netronome.yaml b/netronome/k8s/netronome.yaml index 5fc558f..6868753 100644 --- a/netronome/k8s/netronome.yaml +++ b/netronome/k8s/netronome.yaml @@ -34,7 +34,7 @@ spec: spec: containers: - name: netronome - image: ghcr.io/autobrr/netronome:v0.15.0 + image: ghcr.io/autobrr/netronome:v0.16.0 ports: - name: netronome-port protocol: TCP diff --git a/vpn/xui/k8s/xui.yaml b/vpn/xui/k8s/xui.yaml index 103e0c2..5eb55dc 100644 --- a/vpn/xui/k8s/xui.yaml +++ b/vpn/xui/k8s/xui.yaml @@ -38,7 +38,7 @@ spec: spec: containers: - name: xui - image: ghcr.io/mhsanaei/3x-ui:v3.8.5 + image: ghcr.io/mhsanaei/3x-ui:v3.9.0 envFrom: - configMapRef: name: xui-config From 2545312db1b3bedec99cde2b95b73f07a206c3b7 Mon Sep 17 00:00:00 2001 From: renovate-bot Date: Tue, 6 Oct 2026 10:19:21 +0000 Subject: [PATCH 16/53] chore(deps): update renovate/renovate docker tag to v44.139.0 --- renovate/k8s/cronjob.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/renovate/k8s/cronjob.yaml b/renovate/k8s/cronjob.yaml index f3b0b65..2992e0d 100644 --- a/renovate/k8s/cronjob.yaml +++ b/renovate/k8s/cronjob.yaml @@ -19,7 +19,7 @@ spec: restartPolicy: Never containers: - name: renovate - image: renovate/renovate:44.136.0 + image: renovate/renovate:44.139.0 env: - name: RENOVATE_PLATFORM value: gitea From 18c633c24299c6cae95b042e745cad0117fad3e7 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 17:45:44 +0200 Subject: [PATCH 17/53] feat(monitoring): add VictoriaMetrics trial stack --- prometheus-stack/k8s/grafana-values.yaml | 8 + prometheus-stack/k8s/ingress.yaml | 102 + prometheus-stack/k8s/victoria-scrape.yaml | 1758 +++++++++++++++++ .../k8s/victoria-secrets.yaml.example | 10 + prometheus-stack/k8s/victoria.yaml | 122 ++ prometheus-stack/k8s/vmalert.yaml | 70 + prometheus-stack/k8s/vmctl-backfill.yaml | 31 + templates/deployment.yaml | 543 +++++ templates/endpointslice.yaml | 63 + templates/gateway.yaml | 100 + templates/httproute.yaml | 181 ++ templates/ingressroute-tcp.yaml | 52 + templates/ingressroute-udp.yaml | 19 + templates/ingressroute.yaml | 93 + templates/service.yaml | 100 + templates/statefulset.yaml | 166 ++ templates/tcproute.yaml | 35 + templates/udproute.yaml | 29 + 18 files changed, 3482 insertions(+) create mode 100644 prometheus-stack/k8s/victoria-scrape.yaml create mode 100644 prometheus-stack/k8s/victoria-secrets.yaml.example create mode 100644 prometheus-stack/k8s/victoria.yaml create mode 100644 prometheus-stack/k8s/vmalert.yaml create mode 100644 prometheus-stack/k8s/vmctl-backfill.yaml create mode 100644 templates/deployment.yaml create mode 100644 templates/endpointslice.yaml create mode 100644 templates/gateway.yaml create mode 100644 templates/httproute.yaml create mode 100644 templates/ingressroute-tcp.yaml create mode 100644 templates/ingressroute-udp.yaml create mode 100644 templates/ingressroute.yaml create mode 100644 templates/service.yaml create mode 100644 templates/statefulset.yaml create mode 100644 templates/tcproute.yaml create mode 100644 templates/udproute.yaml diff --git a/prometheus-stack/k8s/grafana-values.yaml b/prometheus-stack/k8s/grafana-values.yaml index 0618ab0..5401346 100644 --- a/prometheus-stack/k8s/grafana-values.yaml +++ b/prometheus-stack/k8s/grafana-values.yaml @@ -38,6 +38,9 @@ grafana: # One block covers both the dashboards and datasources sidecars (p95 91M / 80M). sidecar: + datasources: + # Chart built-in Prometheus DS disabled: VictoriaMetrics (below) is the default. + defaultDatasourceEnabled: false resources: requests: memory: "96Mi" @@ -50,6 +53,11 @@ grafana: type: loki url: http://loki-gateway.prometheus.svc.cluster.local access: proxy + - name: VictoriaMetrics + type: prometheus + url: http://victoria-metrics.prometheus.svc.cluster.local:8428 + access: proxy + isDefault: true prometheus: prometheusSpec: diff --git a/prometheus-stack/k8s/ingress.yaml b/prometheus-stack/k8s/ingress.yaml index 0c48248..10b23ed 100644 --- a/prometheus-stack/k8s/ingress.yaml +++ b/prometheus-stack/k8s/ingress.yaml @@ -31,3 +31,105 @@ spec: port: 80 tls: secretName: internal-wildcard-tls +--- +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: prometheus-local + namespace: prometheus +spec: + entryPoints: + - websecure + routes: + - match: Host(`prom.workstation.internal`) || Host(`prom.gigaforust.internal`) + kind: Rule + services: + - name: prometheus-stack-kube-prom-prometheus + port: 9090 + tls: + secretName: internal-wildcard-tls +--- +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: alertmanager-local + namespace: prometheus +spec: + entryPoints: + - websecure + routes: + - match: Host(`am.workstation.internal`) || Host(`am.gigaforust.internal`) + kind: Rule + services: + - name: prometheus-stack-kube-prom-alertmanager + port: 9093 + tls: + secretName: internal-wildcard-tls +--- +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: loki-local + namespace: prometheus +spec: + entryPoints: + - websecure + routes: + - match: Host(`loki.workstation.internal`) || Host(`loki.gigaforust.internal`) + kind: Rule + services: + - name: loki-gateway + port: 80 + tls: + secretName: internal-wildcard-tls +--- +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: alloy-local + namespace: prometheus +spec: + entryPoints: + - websecure + routes: + - match: Host(`alloy.workstation.internal`) || Host(`alloy.gigaforust.internal`) + kind: Rule + services: + - name: alloy + port: 12345 + tls: + secretName: internal-wildcard-tls +--- +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: victoria-local + namespace: prometheus +spec: + entryPoints: + - websecure + routes: + - match: Host(`victoria.workstation.internal`) || Host(`victoria.gigaforust.internal`) + kind: Rule + services: + - name: victoria-metrics + port: 8428 + tls: + secretName: internal-wildcard-tls +--- +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: vmalert-local + namespace: prometheus +spec: + entryPoints: + - websecure + routes: + - match: Host(`vmalert.workstation.internal`) || Host(`vmalert.gigaforust.internal`) + kind: Rule + services: + - name: vmalert + port: 8880 + tls: + secretName: internal-wildcard-tls diff --git a/prometheus-stack/k8s/victoria-scrape.yaml b/prometheus-stack/k8s/victoria-scrape.yaml new file mode 100644 index 0000000..a510c48 --- /dev/null +++ b/prometheus-stack/k8s/victoria-scrape.yaml @@ -0,0 +1,1758 @@ +# Generated from the operator-rendered Prometheus config (secret +# prometheus-prometheus-stack-kube-prom-prometheus). Regenerate after +# ServiceMonitor changes; do not hand-edit scrape jobs below. +apiVersion: v1 +kind: ConfigMap +metadata: + name: victoria-scrape + namespace: prometheus +data: + scrape.yaml: | + global: + scrape_interval: 30s + scrape_timeout: 10s + external_labels: + scraper: victoria-trial + scrape_configs: + - job_name: serviceMonitor/cert-manager/cert-manager/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - cert-manager + scrape_interval: 60s + scrape_timeout: 30s + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (cainjector|cert-manager|webhook);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_instance + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance + regex: (cert-manager);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_component + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_component + regex: (cainjector|controller|webhook);true + - action: keep + source_labels: + - __meta_kubernetes_pod_container_port_name + regex: http-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: http-metrics + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/crowdsec/crowdsec-agent-service/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - crowdsec + attach_metadata: + node: true + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (crowdsec-agent-service);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: metrics + - source_labels: + - __meta_kubernetes_pod_node_name + target_label: machine + action: replace + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/crowdsec/crowdsec-service/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - crowdsec + attach_metadata: + node: true + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (crowdsec-service);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: metrics + - source_labels: + - __meta_kubernetes_pod_node_name + target_label: machine + action: replace + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/edu-master/webinar-checker/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - edu-master + scrape_interval: 30s + scrape_timeout: 10s + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (edu-master-webinar-checker);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: metrics + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/loki/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + scrape_interval: 15s + metrics_path: /metrics + scheme: http + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_instance + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance + regex: (loki);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (loki);true + - action: drop + source_labels: + - __meta_kubernetes_service_label_prometheus_io_service_monitor + - __meta_kubernetes_service_labelpresent_prometheus_io_service_monitor + regex: (false);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: http-metrics + - source_labels: + - job + target_label: job + replacement: prometheus/$1 + action: replace + - target_label: cluster + replacement: loki + action: replace + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-grafana/0 + honor_labels: true + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + metrics_path: /metrics + scheme: http + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_instance + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (grafana);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-web + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_prometheus_stack + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: http-web + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-alertmanager/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-alertmanager);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_self_monitor + - __meta_kubernetes_service_labelpresent_self_monitor + regex: (true);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-web + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: http-web + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-alertmanager/1 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-alertmanager);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_self_monitor + - __meta_kubernetes_service_labelpresent_self_monitor + regex: (true);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: reloader-web + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: reloader-web + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-apiserver/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - default + scheme: https + tls_config: + insecure_skip_verify: false + server_name: kubernetes + ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_component + - __meta_kubernetes_service_labelpresent_component + regex: (apiserver);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_provider + - __meta_kubernetes_service_labelpresent_provider + regex: (kubernetes);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: https + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_component + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: https + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + metric_relabel_configs: + - source_labels: + - __name__ + - le + regex: (etcd_request|apiserver_request_slo|apiserver_request_sli|apiserver_request)_duration_seconds_bucket;(0\.15|0\.2|0\.3|0\.35|0\.4|0\.45|0\.6|0\.7|0\.8|0\.9|1\.25|1\.5|1\.75|2|3|3\.5|4|4\.5|6|7|8|9|15|20|40|45|50)(\.0)? + action: drop + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-coredns/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - kube-system + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-coredns);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_jobLabel + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: http-metrics + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kube-proxy/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - kube-system + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-kube-proxy);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_jobLabel + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: http-metrics + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kubelet/0 + honor_labels: true + honor_timestamps: true + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - kube-system + attach_metadata: + node: false + scheme: https + tls_config: + insecure_skip_verify: true + ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (kubelet);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_k8s_app + - __meta_kubernetes_service_labelpresent_k8s_app + regex: (kubelet);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: https-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_k8s_app + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: https-metrics + - source_labels: + - __metrics_path__ + target_label: metrics_path + action: replace + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + metric_relabel_configs: + - source_labels: + - __name__ + - le + regex: (csi_operations|storage_operation_duration)_seconds_bucket;(0.25|2.5|15|25|120|600)(\.0)? + action: drop + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kubelet/1 + honor_labels: true + honor_timestamps: true + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - kube-system + attach_metadata: + node: false + scrape_interval: 10s + metrics_path: /metrics/cadvisor + scheme: https + tls_config: + insecure_skip_verify: true + ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (kubelet);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_k8s_app + - __meta_kubernetes_service_labelpresent_k8s_app + regex: (kubelet);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: https-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_k8s_app + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: https-metrics + - source_labels: + - __metrics_path__ + target_label: metrics_path + action: replace + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + metric_relabel_configs: + - source_labels: + - __name__ + regex: container_cpu_(cfs_throttled_seconds_total|load_average_10s|system_seconds_total|user_seconds_total) + action: drop + - source_labels: + - __name__ + regex: container_fs_(io_current|io_time_seconds_total|io_time_weighted_seconds_total|reads_merged_total|sector_reads_total|sector_writes_total|writes_merged_total) + action: drop + - source_labels: + - __name__ + regex: container_memory_(mapped_file|swap) + action: drop + - source_labels: + - __name__ + regex: container_(file_descriptors|tasks_state|threads_max) + action: drop + - source_labels: + - __name__ + - scope + regex: container_memory_failures_total;hierarchy + action: drop + - source_labels: + - __name__ + - interface + regex: container_network_.*;(cali|cilium|cni|lxc|nodelocaldns|tunl).* + action: drop + - source_labels: + - __name__ + regex: container_spec.* + action: drop + - source_labels: + - id + - pod + regex: .+; + action: drop + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kubelet/2 + honor_labels: true + honor_timestamps: true + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - kube-system + attach_metadata: + node: false + metrics_path: /metrics/probes + scheme: https + tls_config: + insecure_skip_verify: true + ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (kubelet);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_k8s_app + - __meta_kubernetes_service_labelpresent_k8s_app + regex: (kubelet);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: https-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_k8s_app + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: https-metrics + - source_labels: + - __metrics_path__ + target_label: metrics_path + action: replace + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-operator/0 + honor_labels: true + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + scheme: https + tls_config: + ca_file: /etc/prometheus/certs/0_prometheus_prometheus-stack-kube-prom-admission_ca + server_name: prometheus-stack-kube-prom-operator + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-operator);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: https + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: https + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-prometheus/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-prometheus);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_self_monitor + - __meta_kubernetes_service_labelpresent_self_monitor + regex: (true);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-web + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: http-web + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-prometheus/1 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (kube-prometheus-stack-prometheus);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_release + - __meta_kubernetes_service_labelpresent_release + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_self_monitor + - __meta_kubernetes_service_labelpresent_self_monitor + regex: (true);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: reloader-web + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: reloader-web + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-kube-state-metrics/0 + honor_labels: true + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_instance + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (kube-state-metrics);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: http + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/prometheus-stack-prometheus-node-exporter/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - prometheus + attach_metadata: + node: false + scheme: http + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_instance + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance + regex: (prometheus-stack);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (prometheus-node-exporter);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http-metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_jobLabel + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: http-metrics + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/prometheus/traefik/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - traefik + metrics_path: /metrics + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_instance + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance + regex: (traefik-traefik);true + - action: keep + source_labels: + - __meta_kubernetes_service_label_app_kubernetes_io_name + - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name + regex: (traefik);true + - action: keep + source_labels: + - __meta_kubernetes_pod_container_port_name + regex: metrics + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - source_labels: + - __meta_kubernetes_service_label_traefik + target_label: job + regex: (.+) + replacement: ${1} + - target_label: endpoint + replacement: metrics + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod + - job_name: serviceMonitor/uptime-kuma/uptime-kuma/0 + honor_labels: false + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: + - uptime-kuma + scrape_interval: 60s + scrape_timeout: 15s + metrics_path: /metrics + basic_auth: + username: uptime-kuma + password_file: /etc/vm/secrets/uptime-kuma-password + relabel_configs: + - source_labels: + - job + target_label: __tmp_prometheus_job_name + - action: keep + source_labels: + - __meta_kubernetes_service_label_app + - __meta_kubernetes_service_labelpresent_app + regex: (uptime-kuma);true + - action: keep + source_labels: + - __meta_kubernetes_endpoint_port_name + regex: http + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Node;(.*) + replacement: ${1} + target_label: node + - source_labels: + - __meta_kubernetes_endpoint_address_target_kind + - __meta_kubernetes_endpoint_address_target_name + separator: ; + regex: Pod;(.*) + replacement: ${1} + target_label: pod + - source_labels: + - __meta_kubernetes_namespace + target_label: namespace + - source_labels: + - __meta_kubernetes_service_name + target_label: service + - source_labels: + - __meta_kubernetes_pod_name + target_label: pod + - source_labels: + - __meta_kubernetes_pod_container_name + target_label: container + - action: drop + source_labels: + - __meta_kubernetes_pod_phase + regex: (Failed|Succeeded) + - source_labels: + - __meta_kubernetes_service_name + target_label: job + replacement: ${1} + - target_label: endpoint + replacement: http + - source_labels: + - __address__ + - __tmp_hash + target_label: __tmp_hash + regex: (.+); + replacement: $1 + action: replace + - source_labels: + - __tmp_hash + target_label: __tmp_hash + modulus: 1 + action: hashmod diff --git a/prometheus-stack/k8s/victoria-secrets.yaml.example b/prometheus-stack/k8s/victoria-secrets.yaml.example new file mode 100644 index 0000000..3ef93ad --- /dev/null +++ b/prometheus-stack/k8s/victoria-secrets.yaml.example @@ -0,0 +1,10 @@ +# Create the victoria-secrets Secret before deploying VictoriaMetrics. +# Copy the metrics password from uptime-kuma/k8s/secrets.yaml into the value below. +apiVersion: v1 +kind: Secret +metadata: + name: victoria-secrets + namespace: prometheus +type: Opaque +stringData: + uptime-kuma-password: "REPLACE_ME" diff --git a/prometheus-stack/k8s/victoria.yaml b/prometheus-stack/k8s/victoria.yaml new file mode 100644 index 0000000..d3adcdc --- /dev/null +++ b/prometheus-stack/k8s/victoria.yaml @@ -0,0 +1,122 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: victoria + namespace: prometheus +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: victoria +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: prometheus-stack-kube-prom-prometheus +subjects: + - kind: ServiceAccount + name: victoria + namespace: prometheus +--- +apiVersion: v1 +kind: Service +metadata: + name: victoria-metrics + namespace: prometheus +spec: + selector: + app: victoria-metrics + ports: + - port: 8428 + targetPort: 8428 +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: victoria-pvc + namespace: prometheus +spec: + resources: + requests: + storage: 10Gi + volumeMode: Filesystem + accessModes: + - ReadWriteOnce +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: victoria-deployment + namespace: prometheus +spec: + replicas: 1 + selector: + matchLabels: + app: victoria-metrics + strategy: + type: Recreate + template: + metadata: + labels: + app: victoria-metrics + spec: + serviceAccountName: victoria + containers: + - name: victoria + image: victoriametrics/victoria-metrics:v1.153.0-scratch + args: + - -storageDataPath=/vmdata + - -retentionPeriod=30d + - -httpListenAddr=:8428 + - -promscrape.config=/etc/vm/conf/scrape.yaml + - -promscrape.configCheckInterval=60s + ports: + - containerPort: 8428 + readinessProbe: + httpGet: + path: /health + port: 8428 + initialDelaySeconds: 15 + periodSeconds: 10 + failureThreshold: 6 + livenessProbe: + httpGet: + path: /health + port: 8428 + initialDelaySeconds: 60 + periodSeconds: 30 + failureThreshold: 3 + volumeMounts: + - name: vmdata + mountPath: /vmdata + - name: scrape-config + mountPath: /etc/vm/conf + readOnly: true + - name: vm-secrets + mountPath: /etc/vm/secrets + readOnly: true + - name: prom-admission-ca + mountPath: /etc/prometheus/certs + readOnly: true + resources: + requests: + cpu: "100m" + memory: "256Mi" + limits: + cpu: "1000m" + memory: "1Gi" + volumes: + - name: vmdata + persistentVolumeClaim: + claimName: victoria-pvc + - name: scrape-config + configMap: + name: victoria-scrape + - name: vm-secrets + secret: + secretName: victoria-secrets + - name: prom-admission-ca + secret: + secretName: prometheus-stack-kube-prom-admission + items: + - key: ca + path: 0_prometheus_prometheus-stack-kube-prom-admission_ca diff --git a/prometheus-stack/k8s/vmalert.yaml b/prometheus-stack/k8s/vmalert.yaml new file mode 100644 index 0000000..9351be0 --- /dev/null +++ b/prometheus-stack/k8s/vmalert.yaml @@ -0,0 +1,70 @@ +apiVersion: v1 +kind: Service +metadata: + name: vmalert + namespace: prometheus +spec: + selector: + app: vmalert + ports: + - port: 8880 + targetPort: 8880 +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: vmalert-deployment + namespace: prometheus +spec: + replicas: 1 + selector: + matchLabels: + app: vmalert + strategy: + type: Recreate + template: + metadata: + labels: + app: vmalert + spec: + containers: + - name: vmalert + image: victoriametrics/vmalert:v1.153.0 + args: + - -datasource.url=http://victoria-metrics.prometheus.svc.cluster.local:8428 + - -remoteWrite.url=http://victoria-metrics.prometheus.svc.cluster.local:8428 + - -notifier.url=http://prometheus-stack-kube-prom-alertmanager.prometheus.svc.cluster.local:9093 + - -rule=/etc/vm/rules/*.yaml + - -evaluationInterval=60s + - -httpListenAddr=:8880 + ports: + - containerPort: 8880 + readinessProbe: + httpGet: + path: /metrics + port: 8880 + initialDelaySeconds: 15 + periodSeconds: 10 + failureThreshold: 6 + livenessProbe: + httpGet: + path: /metrics + port: 8880 + initialDelaySeconds: 60 + periodSeconds: 30 + failureThreshold: 3 + volumeMounts: + - name: rules + mountPath: /etc/vm/rules + readOnly: true + resources: + requests: + cpu: "50m" + memory: "64Mi" + limits: + cpu: "200m" + memory: "256Mi" + volumes: + - name: rules + configMap: + name: prometheus-prometheus-stack-kube-prom-prometheus-rulefiles-0 diff --git a/prometheus-stack/k8s/vmctl-backfill.yaml b/prometheus-stack/k8s/vmctl-backfill.yaml new file mode 100644 index 0000000..269ee46 --- /dev/null +++ b/prometheus-stack/k8s/vmctl-backfill.yaml @@ -0,0 +1,31 @@ +# One-shot history backfill Prometheus -> VictoriaMetrics via vmctl remote-read. +# Safe to keep applied: a completed Job is a no-op on re-apply. +apiVersion: batch/v1 +kind: Job +metadata: + name: vmctl-backfill + namespace: prometheus +spec: + backoffLimit: 2 + ttlSecondsAfterFinished: 3600 + template: + spec: + restartPolicy: OnFailure + containers: + - name: vmctl + image: victoriametrics/vmctl:v1.153.0 + args: + - remote-read + - -s + - --disable-progress-bar + - --remote-read-src-addr=http://prometheus-stack-kube-prom-prometheus.prometheus.svc.cluster.local:9090 + - --remote-read-filter-time-start=2026-09-06T10:26:14Z + - --remote-read-step-interval=day + - --vm-addr=http://victoria-metrics.prometheus.svc.cluster.local:8428 + resources: + requests: + cpu: "200m" + memory: "512Mi" + limits: + cpu: "1000m" + memory: "1Gi" diff --git a/templates/deployment.yaml b/templates/deployment.yaml new file mode 100644 index 0000000..5e65504 --- /dev/null +++ b/templates/deployment.yaml @@ -0,0 +1,543 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + # name: unique within the namespace + name: nginx-deployment + # namespace: logical isolation + namespace: example + labels: + # free-form key/value tags used for selection and grouping. + app: nginx + app.kubernetes.io/name: nginx + app.kubernetes.io/instance: nginx-example + app.kubernetes.io/version: "1.27" + app.kubernetes.io/component: server + app.kubernetes.io/part-of: example + app.kubernetes.io/managed-by: kubectl + annotations: + # Key/value, but never used for selection + # Only for metadata (descriptions, owners, timestamps). + description: "deployment example" + owner: team-platform + # finalizers: identifiers that block deletion until some controller removes + # them after cleanup. RARE on workloads. + # finalizers: + # - example.com/cleanup +spec: + # replicas: how many pod copies to keep running. Default 1. + replicas: 3 + # revisionHistoryLimit: how many old ReplicaSets are kept so you can roll + # back. Default 10. + revisionHistoryLimit: 5 + # progressDeadlineSeconds: if a rollout makes no progress for this long it + # is marked ProgressDeadlineExceeded. Default 600. + progressDeadlineSeconds: 600 + # minReadySeconds: a new pod must stay Ready for this long before it counts + # as available. Protects against pods that flap right after start. + minReadySeconds: 10 + # paused: freezes the rollout controller. RARE - used to accumulate several + # changes and release them as a single rollout. + paused: false + # selector: defines which pods belong to this Deployment. Must match the + # pod template labels exactly. Immutable after creation. + selector: + matchLabels: + app: nginx + # matchExpressions: set-based selection (In, NotIn, Exists, DoesNotExist). + # RARE on Deployments. + # matchExpressions: + # - key: tier + # operator: In + # values: [frontend] + strategy: + # type: RollingUpdate (replace gradually, default) or Recreate (kill all + # old pods first). Recreate fits state that cannot have two writers at + # once (SQLite file, exclusive lock). + type: RollingUpdate + rollingUpdate: + # maxSurge: how many pods above `replicas` may exist mid-rollout. + # Number or percentage. + maxSurge: 1 + # maxUnavailable: how many pods may be simultaneously down mid-rollout. + # Number or percentage. + maxUnavailable: 1 + template: + metadata: + labels: + app: nginx + app.kubernetes.io/name: nginx + annotations: + description: "nginx pod" + spec: + # serviceAccountName: the identity pods use against the API server. + serviceAccountName: default + # automountServiceAccountToken: mount the API token into pods. Set false + # for pods that never call the API to shrink the escape blast radius. + automountServiceAccountToken: true + # schedulerName: which scheduler places the pod. The default scheduler + # handles virtually everything. + schedulerName: default-scheduler + # nodeName: pin the pod to one node, bypassing the scheduler. RARE and + # brittle - nodeSelector/affinity express intent better. + # nodeName: node-1 + # nodeSelector: hard requirement on node labels. + nodeSelector: + kubernetes.io/os: linux + # hostname/subdomain: give the pod a stable hostname and DNS entry + # ...svc.cluster.local. Mostly a + # StatefulSet concern (which gets this automatically). + hostname: nginx + subdomain: example-subdomain + # setHostnameAsFQDN: use the FQDN above as the hostname. Default false. + setHostnameAsFQDN: false + # priorityClassName: scheduling priority; higher values preempt lower. + # priorityClassName: high-priority + # preemptionPolicy: Never stops this pod from preempting others. + # Default PreemptLowerPriority. + preemptionPolicy: PreemptLowerPriority + # runtimeClassName: alternate container runtime (gVisor, Kata). + # Omit for the default runtime. + # runtimeClassName: gvisor + # enableServiceLinks: inject _SERVICE_HOST style env vars. True by + # default; false keeps the environment clean when you use DNS only. + enableServiceLinks: true + # hostAliases: extra /etc/hosts lines. RARE - usually means DNS should + # have been fixed instead. + hostAliases: + - ip: 192.168.1.10 + hostnames: + - legacy-db.example.com + # hostNetwork/hostPID/hostIPC: share the node's network/process/IPC + # namespaces. Needed for node-level agents; dangerous for apps. + hostNetwork: false + hostPID: false + hostIPC: false + # shareProcessNamespace: all containers in the pod see each other's + # processes. Handy for sidecar debuggers; off by default. + shareProcessNamespace: false + # dnsPolicy: ClusterFirst (default, .svc names resolve), Default + # (inherit the node's resolver), ClusterFirstWithHostNet (for + # hostNetwork pods), None (dnsConfig takes over completely). + dnsPolicy: ClusterFirst + dnsConfig: + nameservers: + - 1.1.1.1 + searches: + - example.com + options: + - name: ndots + value: "2" + # readinessGates: custom conditions (reported by external controllers) + # that must be true before the pod counts as Ready. RARE. + # readinessGates: + # - conditionType: example.com/lb-attached + # topologySpreadConstraints: spread pods across zones/hosts with skew + # control. The modern, expressive successor of bare podAntiAffinity. + topologySpreadConstraints: + - maxSkew: 1 + topologyKey: kubernetes.io/hostname + whenUnsatisfiable: ScheduleAnyway + labelSelector: + matchLabels: + app: nginx + affinity: + nodeAffinity: + # requiredDuringSchedulingIgnoredDuringExecution: hard node rule - + # pods that violate it are never scheduled there. + requiredDuringSchedulingIgnoredDuringExecution: + nodeSelectorTerms: + - matchExpressions: + - key: kubernetes.io/arch + operator: In + values: [amd64] + # preferredDuringSchedulingIgnoredDuringExecution: soft node rule - + # the scheduler tries, but schedules anyway if impossible. + preferredDuringSchedulingIgnoredDuringExecution: + - weight: 10 + preference: + matchExpressions: + - key: node-role.kubernetes.io/worker + operator: Exists + podAffinity: + # Attract to nodes already running matching pods (data locality). + preferredDuringSchedulingIgnoredDuringExecution: + - weight: 50 + podAffinityTerm: + labelSelector: + matchLabels: + app: cache + topologyKey: kubernetes.io/hostname + podAntiAffinity: + # Repel from nodes running matching pods (spread replicas). + preferredDuringSchedulingIgnoredDuringExecution: + - weight: 100 + podAffinityTerm: + labelSelector: + matchLabels: + app: nginx + topologyKey: kubernetes.io/hostname + tolerations: + # Tolerate a node taint to allow scheduling there. Without a matching + # toleration, a tainted node rejects the pod. + - key: dedicated + operator: Equal + value: "true" + effect: NoSchedule + # tolerationSeconds: with effect NoExecute, tolerate for this long + # before eviction. Omit for infinite tolerance. + tolerationSeconds: 3600 + # imagePullSecrets: credentials for private registries. + # imagePullSecrets: + # - name: registry-credentials + # securityContext (pod-level): defaults inherited by every container + # unless a container overrides them. + securityContext: + runAsUser: 101 + runAsGroup: 101 + # runAsNonRoot: refuse to start as uid 0. + runAsNonRoot: true + # fsGroup: group that owns mounted volumes; kubelet chowns on mount. + fsGroup: 101 + # fsGroupChangePolicy: Always (chown on every mount, slow on large + # volumes) or OnRootMismatch (chown only when needed). + fsGroupChangePolicy: OnRootMismatch + # supplementalGroups: extra groups granted for volume access. + supplementalGroups: [102] + # sysctls: namespaced kernel parameters. Unsafe ones need explicit + # kubelet opt-in. + sysctls: + - name: net.core.somaxconn + value: "1024" + # seLinuxOptions: SELinux user/role/type/level. RARE outside MLS. + seLinuxOptions: + level: s0:c123,c456 + # seccompProfile: syscall sandbox. RuntimeDefault is the sane + # baseline; Localhost loads a custom profile from the node. + seccompProfile: + type: RuntimeDefault + # windowsOptions: GMSA / runAsUserName for Windows nodes. + initContainers: + # Run strictly in order, each to completion, before app containers + # start. Used for migrations, permission fixes, dependency waits. + - name: init-permissions + image: busybox:1.36 + command: ["sh", "-c", "chown -R 101:101 /data"] + volumeMounts: + - name: html + mountPath: /data + resources: + requests: + cpu: "10m" + memory: "16Mi" + limits: + cpu: "50m" + memory: "64Mi" + # restartPolicy on a container (not the pod): Always turns it into + # a native sidecar that keeps running next to the app (1.28+). + # restartPolicy: Always + containers: + - name: nginx + # image: repository plus tag. Pin tags - `latest` moves under you. + image: nginx:1.27.3 + # imagePullPolicy: Always (re-pull even pinned tags), IfNotPresent + # (use cache, works offline), Never (cache only, fails otherwise). + imagePullPolicy: IfNotPresent + # command: overrides the image ENTRYPOINT. args: overrides CMD. + # command: ["nginx"] + # args: ["-g", "daemon off;"] + # workingDir: overrides the image WORKDIR. + workingDir: /usr/share/nginx/html + # stdin/stdinOnce/tty: interactive input. For debug shells and + # one-shot runs, never for servers. + stdin: false + stdinOnce: false + tty: false + ports: + - name: http + containerPort: 80 + protocol: TCP + # hostPort: expose straight on the node, bypassing Services. + # RARE - allows only one such pod per node per port. + # hostPort: 8080 + # hostIP: which node address hostPort binds to. + # hostIP: 127.0.0.1 + env: + - name: NGINX_PORT + value: "80" + - name: POD_NAME + valueFrom: + fieldRef: + # fieldPath exposes pod metadata: metadata.name, + # metadata.namespace, metadata.uid, spec.nodeName, + # spec.serviceAccountName, status.podIP(s), etc. + fieldPath: metadata.name + - name: NODE_NAME + valueFrom: + fieldRef: + fieldPath: spec.nodeName + - name: POD_MEMORY_LIMIT + valueFrom: + # resourceFieldRef exposes this container's own + # requests/limits. divisor formats the value. + resourceFieldRef: + resource: limits.memory + divisor: 1Mi + - name: API_PASSWORD + valueFrom: + secretKeyRef: + name: example-secrets + key: api-password + # optional: tolerate a missing key (variable stays unset). + optional: false + - name: LOG_LEVEL + valueFrom: + configMapKeyRef: + name: example-config + key: log-level + optional: false + envFrom: + # Bulk-inject every key of a ConfigMap/Secret as env vars. + - configMapRef: + name: example-config + optional: false + # prefix: prepended to every injected key, avoids collisions. + prefix: APP_ + - secretRef: + name: example-secrets + optional: false + resources: + # requests: guaranteed reservation used for scheduling. Set at + # measured idle/p95 - over-requesting starves neighboring pods. + requests: + cpu: "100m" + memory: "128Mi" + # limits: hard ceiling. Breaching memory kills the container + # (OOMKilled); breaching CPU only throttles it (slow, not dead). + limits: + cpu: "500m" + memory: "512Mi" + # claims: reference a ResourceClaim for dynamic resources + # (GPUs via DRA, 1.26+). RARE. + # claims: + # - name: gpu + # resizePolicy: what happens on in-place container resize (1.27+). + # NotRequired keeps running; RestartContainer restarts to apply. + resizePolicy: + - resourceName: cpu + restartPolicy: NotRequired + - resourceName: memory + restartPolicy: NotRequired + volumeMounts: + - name: html + mountPath: /usr/share/nginx/html + readOnly: false + # subPath: mount a single file/dir of the volume instead of + # its root. Typical for single-file config mounts. + # subPath: index.html + # subPathExpr: subPath assembled from env variables. + # subPathExpr: $(POD_NAME)/data + # mountPropagation: share mounts back with the host + # (HostToContainer, Bidirectional). Storage-driver territory. + mountPropagation: None + - name: tmp + mountPath: /tmp + # volumeDevices: raw block devices without a filesystem. RARE - + # databases on local PVs with volumeMode: Block. + # volumeDevices: + # - name: blockvol + # devicePath: /dev/xvda + livenessProbe: + # Exactly one handler per probe: httpGet, tcpSocket, exec, grpc. + httpGet: + path: /healthz + port: http + scheme: HTTP + # httpHeaders: extra headers sent with the probe request. + httpHeaders: + - name: Host + value: example.com + # initialDelaySeconds: wait after start before first probe. + initialDelaySeconds: 15 + # periodSeconds: interval between probes. + periodSeconds: 20 + # timeoutSeconds: when a single probe counts as failed. + timeoutSeconds: 5 + # successThreshold: consecutive successes to count as healthy. + # Keep 1. + successThreshold: 1 + # failureThreshold: consecutive failures to trigger the action. + failureThreshold: 3 + readinessProbe: + # Failing readiness removes the pod from Services (no traffic) + # without restarting it. Failing liveness restarts it. + httpGet: + path: /readyz + port: 80 + initialDelaySeconds: 5 + periodSeconds: 10 + timeoutSeconds: 3 + successThreshold: 1 + failureThreshold: 3 + startupProbe: + # Disables liveness/readiness until it first succeeds. Total + # budget = failureThreshold * periodSeconds (here 30 * 10s). + # The cure for slow-starting apps that otherwise get + # restart-looped before they finish booting. + tcpSocket: + port: 80 + # host: probe a different host than the pod IP. RARE. + failureThreshold: 30 + periodSeconds: 10 + timeoutSeconds: 5 + lifecycle: + # postStart runs right after start; preStop runs before SIGTERM. + # Slow hooks stall the pod transition - keep them fast. + postStart: + exec: + command: ["sh", "-c", "echo started > /tmp/started"] + # Classic preStop: sleep so endpoints are removed everywhere + # before the process receives SIGTERM. + preStop: + exec: + command: ["sh", "-c", "sleep 5"] + # terminationMessagePath: file whose content becomes the container's + # final status message. FallbackToLogsOnError appends log tail when + # the file is empty. + terminationMessagePath: /dev/termination-log + terminationMessagePolicy: File + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: false + runAsNonRoot: true + runAsUser: 101 + runAsGroup: 101 + capabilities: + add: + - NET_BIND_SERVICE + drop: + - ALL + seccompProfile: + type: RuntimeDefault + # ephemeralContainers: troubleshooting shells injected into a RUNNING + # pod with kubectl debug. Declared ad-hoc in practice, never committed. + # ephemeralContainers: + # - name: debugger + # image: busybox:1.36 + # command: ["sh"] + # stdin: true + # tty: true + # targetContainerName: nginx + volumes: + - name: html + persistentVolumeClaim: + claimName: example-pvc + # readOnly: mount the claim read-only in this pod. + readOnly: false + - name: tmp + emptyDir: + # medium "" (node disk) or Memory (tmpfs). sizeLimit evicts the + # pod when exceeded. Dies with the pod either way. + medium: "" + sizeLimit: 256Mi + - name: config-files + configMap: + # Each key becomes a file under the mount path. + name: example-config-files + defaultMode: 0644 + optional: false + items: + - key: nginx.conf + path: nginx.conf + mode: 0644 + - name: tls + secret: + # Secret volumes are tmpfs-backed, never touch node disk. + secretName: example-tls + defaultMode: 0644 + optional: false + items: + - key: tls.crt + path: tls.crt + mode: 0644 + - name: host-time + hostPath: + path: /etc/localtime + # type: DirectoryOrCreate, Directory, FileOrCreate, File, + # Socket, CharDevice, BlockDevice. Always set it: a missing + # path then fails loudly instead of silently creating the + # wrong filesystem object. + type: File + - name: podinfo + downwardAPI: + # Expose pod metadata as files. + items: + - path: labels + fieldRef: + fieldPath: metadata.labels + - path: cpu-request + resourceFieldRef: + resource: requests.cpu + containerName: nginx + divisor: 1m + - name: all-config + projected: + # Merge several sources into a single directory. + defaultMode: 0644 + sources: + - configMap: + name: example-config + items: + - key: log-level + path: log-level + - secret: + name: example-secrets + items: + - key: api-password + path: password + - downwardAPI: + items: + - path: podname + fieldRef: + fieldPath: metadata.name + # Further volume types share the same `name:` + type shape: + # nfs: { server, path }, csi: (storage drivers), persistentVolumeClaim + # shown above, plus legacy in-tree plugins (fc, iscsi, rbd, glusterfs) + # and gitRepo (deprecated - use initContainers + emptyDir instead). + # restartPolicy: only Always is valid for Deployments. (Jobs use + # OnFailure/Never; bare pods accept all three.) + restartPolicy: Always + terminationGracePeriodSeconds: 30 + # activeDeadlineSeconds: kill the pod after this long no matter what. + # A Job concern, not a server concern - shown only for completeness. + # activeDeadlineSeconds: 3600 +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: example-pvc + namespace: example +spec: + # accessModes: ReadWriteOnce (one node), ReadOnlyMany, ReadWriteMany + # (needs a shared filesystem), ReadWriteOncePod (single pod, strictest). + accessModes: + - ReadWriteOnce + # storageClassName: selects the provisioner. "" (empty) disables dynamic + # provisioning and binds a pre-created volume instead. + storageClassName: standard + # volumeMode: Filesystem (default) or Block (used with volumeDevices). + volumeMode: Filesystem + resources: + requests: + storage: 5Gi + # limits: accepted on paper, almost no provisioner enforces them. + # selector: bind a specific pre-created PV by labels. RARE with dynamic + # provisioning. + # selector: + # matchLabels: + # disk: ssd + # dataSource: clone this PVC from another PVC or a snapshot at creation. + # dataSource: + # name: example-snapshot + # kind: VolumeSnapshot + # apiGroup: snapshot.storage.k8s.io + # dataSourceRef: cross-namespace-capable successor of dataSource. diff --git a/templates/endpointslice.yaml b/templates/endpointslice.yaml new file mode 100644 index 0000000..6ac1892 --- /dev/null +++ b/templates/endpointslice.yaml @@ -0,0 +1,63 @@ +# Exhaustive EndpointSlice reference (discovery.k8s.io/v1). +# EndpointSlices are the address lists behind a Service: each entry says +# "IP X serves port Y, ready or not". Normally the endpoint controller writes +# them automatically from the Service selector. You write one by hand in a +# single case: a selector-less Service pointing OUTSIDE the cluster (a host +# daemon, a LAN appliance, an external database). The pair looks like: +# Service (no selector, same port names) + this EndpointSlice. +apiVersion: discovery.k8s.io/v1 +kind: EndpointSlice +metadata: + name: app-external-abc123 + namespace: example + labels: + # kubernetes.io/service-name: THE binding label. It must equal the + # selector-less Service name - this is what attaches the slice to it. + # The endpoint controller owns slices it creates; hand-written ones + # just need this label to be picked up. + kubernetes.io/service-name: app-external-service + app: app + annotations: + description: "hand-written slice for an off-cluster backend" +# addressType: IPv4, IPv6, or FQDN. All endpoints in one slice share it - +# mix families with one slice per family. +addressType: IPv4 +ports: + - name: http + # name: MUST match the Service port name it serves. + protocol: TCP + # port: the REAL backend port (may differ from the Service port - the + # Service port is the in-cluster alias, this is where packets go). + port: 8080 + # appProtocol: payload hint, mirrors the Service field. + appProtocol: http +endpoints: + # One entry per backend address. kube-proxy load-balances across the ones + # whose conditions say ready+serving+terminating=false. + - addresses: + # addresses: one or more IPs (or a single DNS name for FQDN slices). + - 192.168.1.50 + conditions: + # ready: backend accepts traffic. False removes it from rotation + # without deleting the entry (flap-friendly). + ready: true + # serving: the process is up. Differs from ready during shutdown: + # serving=false + terminating=true = draining. + serving: true + # terminating: the endpoint is going away. Draining traffic, not dead. + terminating: false + # hostname: DNS name published for this endpoint (headless Services). + # hostname: backend-1 + # targetRef: link back to the pod/node object (set automatically on + # controller-managed slices; omit on hand-written ones). + # targetRef: + # kind: Pod + # namespace: example + # name: app-67890abcde-fghij + # uid: 12345678-1234-1234-1234-123456789abc + # nodeName: the node hosting this endpoint (topology-aware routing). + # zone: override the endpoint zone (defaults from nodeName). + # hints: topology hints for zone-aware routing (PreferClose). + # hints: + # forZones: + # - name: zone-a diff --git a/templates/gateway.yaml b/templates/gateway.yaml new file mode 100644 index 0000000..bb36b7a --- /dev/null +++ b/templates/gateway.yaml @@ -0,0 +1,100 @@ +# Exhaustive Gateway reference (gateway.networking.k8s.io/v1). +# A Gateway is the entry door: it owns listener ports/protocols/hostnames and +# delegates actual routing to Route objects (HTTPRoute, TCPRoute, ...), which +# attach via parentRefs. One Gateway usually fronts many Routes. +apiVersion: gateway.networking.k8s.io/v1 +kind: Gateway +metadata: + name: example + namespace: example + labels: + app: example + annotations: + description: "exhaustive gateway example" +spec: + # gatewayClassName: which controller implements this Gateway + # (kubectl get gatewayclass). The controller only touches Gateways naming + # its own class; anything else stays Ignored. + gatewayClassName: example-class + # addresses: VIPs/hostnames to request for the Gateway. Most controllers + # (including cloud LBs) allocate and fill status.addresses automatically; + # setting this pins a static IP. Omit for auto-assignment. + # addresses: + # - type: IPAddress + # value: 203.0.113.10 + # - type: Hostname + # value: lb.example.com + # infrastructure: controller-specific settings for the provisioned data + # plane (labels/annotations propagated to it). RARE - most setups never + # need it. + # infrastructure: + # labels: + # environment: prod + # annotations: + # example.com/keep: "true" + listeners: + # Each listener = one port + protocol + optional hostname + TLS + which + # Routes may attach. Listener names are referenced by Route parentRefs + # via sectionName. + - name: http + # port: 1-65535. Must be free on the data plane (controllers often + # require 80/443 to match their own entrypoints, otherwise the + # listener is marked Invalid/Conflicted). + port: 80 + # protocol: HTTP, HTTPS, TLS, TCP, UDP. + protocol: HTTP + # hostname: restrict this listener to one DNS name. Omit to accept all + # (Routes then narrow via their own hostnames). Listener hostname and + # Route hostnames must intersect or the Route is rejected. + hostname: app.example.com + # allowedRoutes: which Routes may bind here. + allowedRoutes: + namespaces: + # from: Same (only this namespace), All (any namespace), or + # Selector (namespaces matching the selector below). + from: Same + # selector: used only with from: Selector. + # selector: + # matchLabels: + # shared-gateway-access: "true" + # kinds: restrict by Route kind. Default allows whatever the + # listener protocol supports (HTTPRoute on HTTP, etc.). + kinds: + - kind: HTTPRoute + - name: https + port: 443 + protocol: HTTPS + hostname: app.example.com + # tls: termination settings for HTTPS/TLS listeners. + tls: + # mode: Terminate (decrypt here, default) or Passthrough (forward + # encrypted bytes to the backend - the backend holds the key). + mode: Terminate + # certificateRefs: TLS Secrets (or other kinds) in the SAME namespace + # (cross-namespace needs a ReferenceGrant). SNI picks among them. + certificateRefs: + - name: app-prod-tls + # kind/group default to Secret / core. Other kinds (e.g. a + # cert-manager Certificate via a plugin) set kind + group. + kind: Secret + group: "" + # options: controller-specific TLS knobs, referenced by name + # (cipher suites, min version). RARE. + # options: + # name: tls-options + - name: tcp + port: 2222 + protocol: TCP + allowedRoutes: + namespaces: + from: Same + kinds: + - kind: TCPRoute + - name: udp + port: 3478 + protocol: UDP + allowedRoutes: + namespaces: + from: Same + kinds: + - kind: UDPRoute diff --git a/templates/httproute.yaml b/templates/httproute.yaml new file mode 100644 index 0000000..68fd0ad --- /dev/null +++ b/templates/httproute.yaml @@ -0,0 +1,181 @@ +# Exhaustive HTTPRoute reference (gateway.networking.k8s.io/v1). +# Attaches to a Gateway listener via parentRefs and routes HTTP(S) by +# hostname + path/method/headers/query. Rules are evaluated in order; the +# first matching rule wins. +apiVersion: gateway.networking.k8s.io/v1 +kind: HTTPRoute +metadata: + name: app + namespace: example + labels: + app: app +spec: + parentRefs: + # Every entry picks one listener to bind to. Omit sectionName/port to + # attach to ALL listeners of the Gateway (common for simple setups). + - name: example + # namespace: Gateway's namespace. Omit when same-namespace (the norm; + # cross-namespace needs the Gateway to allow it). + # namespace: example + # kind/group: default Gateway / gateway.networking.k8s.io. Set + # explicitly only for non-Gateway parents (mesh service parents). + kind: Gateway + group: gateway.networking.k8s.io + # sectionName: the listener name from the Gateway (http/https/...). + sectionName: https + # port: narrow further to one listener port. RARE when sectionName is + # already set. + port: 443 + # hostnames: which Host headers this Route serves. Must intersect the + # listener hostname; otherwise the Route is rejected as incompatible. + # Omit to match every hostname on the listener. + hostnames: + - app.example.com + - www.example.com + rules: + # Rule 1: API traffic with header manipulation and canary split. + - matches: + # All conditions inside one match are ANDed; several matches in the + # list are ORed. + - path: + # type: Exact (one URL), PathPrefix (subtree), RegularExpression. + type: PathPrefix + value: /api + method: POST + headers: + # type: Exact or RegularExpression. Names are case-insensitive + # per HTTP spec; values are case-sensitive. + - type: Exact + name: X-Api-Version + value: v2 + queryParams: + # Match on ?debug=true style parameters. + - type: Exact + name: debug + value: "true" + filters: + # Filters run in order and transform the request/response. + - type: RequestHeaderModifier + requestHeaderModifier: + # add: append even if present (duplicates allowed). set: + # overwrite-or-add. remove: delete by name. + add: + - name: X-Gateway + value: example + set: + - name: X-Forwarded-Proto + value: https + remove: + - X-Internal-Token + - type: ResponseHeaderModifier + responseHeaderModifier: + set: + - name: X-Frame-Options + value: DENY + remove: + - Server + # backendRefs below carry per-backend filters too; rule-level filters + # apply to every backend of this rule. + backendRefs: + - name: app-service + # port: the Service port (number). Required - unlike backendRef in + # Ingress, there is no default. + port: 80 + # group/kind: default Service / core (""). Other kinds (e.g. a + # ServiceImport for multi-cluster) set kind + group explicitly. + kind: Service + group: "" + # weight: traffic share. 90/10 below = canary: 90% stable, 10% new. + weight: 90 + filters: + # Per-backend filter: only this backend's requests get it. + - type: RequestHeaderModifier + requestHeaderModifier: + set: + - name: X-Backend + value: stable + - name: app-canary-service + port: 80 + weight: 10 + filters: + - type: RequestHeaderModifier + requestHeaderModifier: + set: + - name: X-Backend + value: canary + # timeouts: per-attempt deadlines. request = whole gateway-to-client + # exchange; backendRequest = single backend try. + timeouts: + request: 30s + backendRequest: 10s + # sessionPersistence: stick a client to one backend (cookie-based). + # Type Cookie or Header; absoluteTimeout caps the stickiness. + sessionPersistence: + sessionName: route-session + type: Cookie + absoluteTimeout: 1h + cookieConfig: + lifetimeType: Session + # Rule 2: redirect old path to a new URL. + - matches: + - path: + type: PathPrefix + value: /old-docs + filters: + - type: RequestRedirect + requestRedirect: + # Any combination: scheme/host/port/path/statusCode. Unset fields + # keep the original value. + scheme: https + hostname: docs.example.com + path: + # type: ReplaceFullPath or ReplacePrefixMatch (rewrite the + # matched prefix, keep the remainder). + type: ReplacePrefixMatch + replacePrefixMatch: /docs + port: 443 + # statusCode: 301 (permanent) or 302 (temporary). + statusCode: 301 + # Rule 3: rewrite the URL but still proxy (client sees no redirect). + - matches: + - path: + type: PathPrefix + value: /shop + filters: + - type: URLRewrite + urlRewrite: + path: + type: ReplacePrefixMatch + replacePrefixMatch: /store + # hostname: also rewrite the Host header sent upstream. + # hostname: store-internal.example.com + backendRefs: + - name: app-service + port: 80 + # Rule 4: mirror (shadow) traffic to a second backend for testing. + # The mirror gets a copy; its response is discarded. + - matches: + - path: + type: Exact + value: /checkout + filters: + - type: RequestMirror + requestMirror: + backendRef: + name: app-shadow-service + port: 80 + backendRefs: + - name: app-service + port: 80 + # Rule 5: delegate to an implementation-specific filter (auth plugin, + # rate limit, wasm). The controller documents the group/kind it honors. + # - filters: + # - type: ExtensionRef + # extensionRef: + # group: example.com + # kind: AuthPolicy + # name: app-auth + # Catch-all rule (no matches): everything not matched above lands here. + - backendRefs: + - name: app-service + port: 80 diff --git a/templates/ingressroute-tcp.yaml b/templates/ingressroute-tcp.yaml new file mode 100644 index 0000000..915d7e6 --- /dev/null +++ b/templates/ingressroute-tcp.yaml @@ -0,0 +1,52 @@ +# Exhaustive Traefik IngressRouteTCP reference (traefik.io/v1alpha1). +# Routes raw TCP: SSH, databases, or TLS-passthrough where Traefik never +# decrypts. Two TLS modes exist - termination (Traefik holds the cert) and +# passthrough (backend holds the cert) - and they are mutually exclusive. +apiVersion: traefik.io/v1alpha1 +kind: IngressRouteTCP +metadata: + name: app-ssh + namespace: example + labels: + app: app +spec: + entryPoints: + - ssh + routes: + # Plain TCP (SSH here): no TLS block at all, bytes flow as-is. + - match: HostSNI(`*`) + # HostSNI matches the TLS Server Name Indication. `*` accepts anything + # (required for non-TLS protocols like SSH that send no SNI). + # With TLS + a real hostname: HostSNI(`db.example.com`). + # middlewares: TCP middleware chain (IP allowlist, rate limit...). + # middlewares: + # - name: ssh-allowlist + # priority: same semantics as HTTP - higher wins. + # priority: 10 + services: + - name: app-service + port: 2222 + # weight: share of connections across backends. + # weight: 1 + # terminationDelay: linger after backend close to drain in-flight + # data. Default 100ms; raise for slow protocols. + terminationDelay: 100 + # proxyProtocol: PROXY header toward the backend (v1/v2) so it + # learns real client IPs. + # proxyProtocol: + # version: 2 + # TLS termination: Traefik decrypts with its own cert, forwards plaintext. + # - match: HostSNI(`db.example.com`) + # services: + # - name: app-service + # port: 5432 + # tls: enable TLS handling on this route. Omit entirely for plain TCP. + # tls: + # Either termination... + # secretName: app-tcp-tls + # options: + # name: modern-tls + # domains: + # - main: db.example.com + # ...or passthrough (Traefik never sees plaintext; needs SNI routing): + # passthrough: true diff --git a/templates/ingressroute-udp.yaml b/templates/ingressroute-udp.yaml new file mode 100644 index 0000000..4961ac6 --- /dev/null +++ b/templates/ingressroute-udp.yaml @@ -0,0 +1,19 @@ +# Exhaustive Traefik IngressRouteUDP reference (traefik.io/v1alpha1). +# Routes UDP datagrams (DNS, STUN/TURN, syslog...). No match rules exist - +# UDP has no hostname/SNI to route on, so one route per entrypoint simply +# forwards everything it receives. +apiVersion: traefik.io/v1alpha1 +kind: IngressRouteUDP +metadata: + name: app-stun + namespace: example + labels: + app: app +spec: + entryPoints: + - stun + services: + - name: app-service + port: 3478 + # weight: share of datagrams when several backends are listed. + weight: 1 diff --git a/templates/ingressroute.yaml b/templates/ingressroute.yaml new file mode 100644 index 0000000..dd32fbb --- /dev/null +++ b/templates/ingressroute.yaml @@ -0,0 +1,93 @@ +# Exhaustive Traefik IngressRoute reference (traefik.io/v1alpha1, HTTP). +# Routes are evaluated top to bottom by priority, then by rule length: the +# first matching route handles the request. Keep specific rules above the +# catch-all. +apiVersion: traefik.io/v1alpha1 +kind: IngressRoute +metadata: + name: app-prod + namespace: example + labels: + app: app + annotations: + description: "exhaustive traefik http route example" +spec: + # entryPoints: static entrypoints the route listens on (ports Traefik was + # started with: web=:80, websecure=:443, plus any custom ones). + entryPoints: + - websecure + routes: + # Rule 1: API subtree with middleware chain and weighted backends. + - match: Host(`app.example.com`) && PathPrefix(`/api`) + # kind: Rule (match traffic) or the same match used for redirections. + kind: Rule + # priority: explicit precedence. Higher wins regardless of position. + # Default is the rule length in characters - explicit numbers beat + # clever ordering. + priority: 100 + # middlewares: request pipeline, in order (auth, headers, rate limit, + # redirect, strip prefix...). Same-namespace by name; cross-namespace + # as name@namespace (never across providers without the suffix). + middlewares: + - name: app-auth + - name: security-headers + services: + - name: app-service + # port: Service port number or name. + port: 80 + # scheme: http (default), https (TLS backend), h2c (cleartext + # HTTP/2, e.g. gRPC without TLS). + scheme: http + # weight: traffic share for canary/blue-green splits. + weight: 90 + # serversTransport: ServersTransport CRD with TLS/forwarding + # tuning for this backend (rootCAs, insecureSkipVerify...). + # serversTransport: app-transport + # responseForwarding: + # flushInterval: 100ms + # passHostHeader: forward the original Host header (default true). + # passHostHeader: true + # proxyProtocol: speak PROXY protocol to the backend so it sees + # real client IPs. Version v1 or v2; backend must understand it. + # proxyProtocol: + # version: 2 + - name: app-canary-service + port: 80 + weight: 10 + # Rule 2: multiple hosts, regex path, external backend by URL. + - match: (Host(`app.example.com`) || Host(`www.example.com`)) && PathRegexp(`^/files/.*$`) + kind: Rule + priority: 50 + services: + # servers: bypass the Service and address backends directly. + # Only one of name/port (cluster Service) or servers (explicit + # URLs) may be set. + - name: app-service + port: 80 + # Catch-all rule: everything not matched above. + - match: Host(`app.example.com`) + kind: Rule + priority: 1 + services: + - name: app-service + port: 80 + # tls: terminate TLS on this route. Omit the whole block for plain HTTP. + tls: + # secretName: TLS Secret (tls.crt/tls.key) in THIS namespace. + secretName: app-prod-tls + # options: TLSOption CRD (minVersion, cipherSuites, sniStrict...). + # options: + # name: modern-tls + # certResolver: ACME resolver name (letsencrypt-style) that issues the + # certificate on demand. Use EITHER certResolver OR secretName, not both: + # resolver for auto-issued certs, secretName for pre-made ones. + # certResolver: letsencrypt + # store: custom TLSStore for the certificate. Default store otherwise. + # store: + # name: default + # domains: certificates to request/serve (main + SANs). With secretName + # this documents intent; with certResolver it drives issuance. + domains: + - main: app.example.com + sans: + - www.example.com diff --git a/templates/service.yaml b/templates/service.yaml new file mode 100644 index 0000000..e5181f5 --- /dev/null +++ b/templates/service.yaml @@ -0,0 +1,100 @@ +# Exhaustive Service reference (v1). +# A Service is a stable virtual endpoint in front of pods: one DNS name and +# one cluster IP that load-balances to the currently Ready pods behind it. +# Pods come and go; the Service name (app-service.example.svc.cluster.local) +# never changes. +apiVersion: v1 +kind: Service +metadata: + name: app-service + namespace: example + labels: + app: app + annotations: + description: "exhaustive service example" +spec: + # selector: pods carrying these labels receive traffic. Empty selector = + # no automatic endpoints (pair with a hand-written EndpointSlice to aim at + # an external address - see endpointslice.yaml). + selector: + app: app + ports: + - name: http + # protocol: TCP (default), UDP, or SCTP. Each port entry needs one. + protocol: TCP + # port: the port clients connect to on the Service IP. + port: 80 + # targetPort: the port on the pod. Number or container port NAME + # (names decouple the Service from container port renumbering). + # Omit when it equals `port`. + targetPort: http + # nodePort: fixed node port for type NodePort/LoadBalancer (range + # 30000-32767). Omit for auto-assignment. + # nodePort: 30080 + # appProtocol: protocol hint for the payload (http, https, grpc, h2c, + # ws...). Used by meshes and LBs, ignored by plain kube-proxy routing. + appProtocol: http + - name: metrics + protocol: TCP + port: 9090 + targetPort: 9090 + # type: ClusterIP (default, internal VIP), NodePort (also open a fixed port + # on every node), LoadBalancer (NodePort + cloud LB in front), + # ExternalName (DNS alias, no proxying - see below). + type: ClusterIP + # clusterIP: pin the virtual IP. "None" makes the Service headless: no VIP, + # DNS returns pod IPs directly (required base for StatefulSets). + # clusterIP: None + # clusterIPs / ipFamilies / ipFamilyPolicy: dual-stack control. + # ipFamilies: [IPv4] (default), [IPv6], or [IPv4, IPv6]. + # ipFamilyPolicy: SingleStack (default), PreferDualStack, RequireDualStack. + # ipFamilies: + # - IPv4 + # ipFamilyPolicy: SingleStack + # sessionAffinity: None (default, spread every connection) or ClientIP + # (same client IP sticks to one pod). + sessionAffinity: None + # sessionAffinityConfig: stickiness TTL for ClientIP affinity. + # sessionAffinityConfig: + # clientIP: + # timeoutSeconds: 10800 + # publishNotReadyAddresses: send traffic to not-Ready pods too. Needed for + # peer discovery where members must find each other before anyone is Ready. + publishNotReadyAddresses: false + # internalTrafficPolicy: Cluster (default, route to pods on any node) or + # Local (only pods on the receiving node - preserves source IP, drops + # traffic on nodes without local pods). + internalTrafficPolicy: Cluster + # --- NodePort / LoadBalancer extras (ignored by pure ClusterIP) --- + # externalTrafficPolicy: Cluster (default) or Local. Local preserves the + # client source IP but drops traffic arriving on nodes with no local pod. + # externalTrafficPolicy: Cluster + # healthCheckNodePort: fixed node port for the LB health check (Local + # policy). Omit for auto-assignment. + # healthCheckNodePort: 32111 + # allocateLoadBalancerNodePorts: set false to skip node-port allocation on + # a LoadBalancer (when the LB routes straight to pods). Default true. + # allocateLoadBalancerNodePorts: true + # loadBalancerIP: request a specific IP from the cloud provider. Provider + # support varies; most now prefer annotations or IP pools. + # loadBalancerIP: 203.0.113.10 + # loadBalancerSourceRanges: client CIDRs allowed through the cloud LB. + # Unset = world-open. This is the cloud firewall in front of the Service. + # loadBalancerSourceRanges: + # - 203.0.113.0/24 + # loadBalancerClass: use a custom LB implementation instead of the cloud + # default (e.g. MetalLB speaker). The named controller must be installed. + # loadBalancerClass: example.com/custom-lb + # trafficDistribution: hint how to prefer endpoints (PreferClose = same + # zone first). Best-effort, kube-proxy dependent. + # trafficDistribution: PreferClose + # --- other types --- + # externalIPs: extra IPs (already routed to nodes) that also serve this + # Service. Traffic arriving there is proxied like ClusterIP traffic. + # externalIPs: + # - 203.0.113.20 + # ExternalName type: no proxying at all - DNS CNAME to the target. + # Ports are informational. Used to reference outside names under a stable + # in-cluster name. + # type: ExternalName + # externalName: db.example.com diff --git a/templates/statefulset.yaml b/templates/statefulset.yaml new file mode 100644 index 0000000..b542726 --- /dev/null +++ b/templates/statefulset.yaml @@ -0,0 +1,166 @@ +# Exhaustive StatefulSet reference (apps/v1), shown with its headless Service. +# StatefulSets give each pod a stable name and stable storage: +# web-0, web-1, ... each reattached to its own volume after rescheduling. +# The classic use is databases and anything with an identity (postgres, +# redis, kafka). For the full container/pod field catalog see +# deployment.yaml - only StatefulSet-specific fields are expanded here. +apiVersion: v1 +kind: Service +metadata: + name: example-sts + namespace: example + labels: + app: example-sts +spec: + # clusterIP: None makes the Service headless: no virtual IP, DNS returns + # the pod IPs directly (web-0.example-sts...). Required for StatefulSets - + # it is how stable network identity works. + clusterIP: None + # publishNotReadyAddresses: include not-Ready pods in DNS. Needed for + # peer discovery when members must find each other before anyone is Ready + # (etcd, clustered databases). + publishNotReadyAddresses: false + selector: + app: example-sts + ports: + - name: db + port: 5432 + targetPort: db + # sessionAffinity: None (default, spread connections) or ClientIP (sticky + # sessions to one pod). Stateful apps sometimes want ClientIP. + sessionAffinity: None +--- +apiVersion: apps/v1 +kind: StatefulSet +metadata: + name: example-sts + namespace: example + labels: + app: example-sts +spec: + # serviceName: the headless Service above. Must exist; governs the pods' + # DNS domain. Changing it requires recreating the StatefulSet. + serviceName: example-sts + # replicas: pod count. Pods start in order 0..N-1 and stop in reverse. + replicas: 3 + # revisionHistoryLimit: old ControllerRevisions kept for rollback. + revisionHistoryLimit: 5 + # minReadySeconds: pod must stay Ready this long to count as available. + minReadySeconds: 10 + # podManagementPolicy: OrderedReady (default - strict 0,1,2 startup order, + # each predecessor must be Ready) or Parallel (start/stop all at once, + # faster for stateless-ish sets that only want stable names). + podManagementPolicy: OrderedReady + # persistentVolumeClaimRetentionPolicy: what happens to per-pod PVCs on + # scale-down (whenDeleted) and StatefulSet deletion (whenScaled). Retain + # keeps data (safe default); Delete wipes it. Set explicitly - the default + # Retain surprises people who expected cleanup. + persistentVolumeClaimRetentionPolicy: + whenDeleted: Retain + whenScaled: Retain + # ordinals: first ordinal (default 0). RARE - used when migrating an + # existing cluster whose numbering starts elsewhere. + # ordinals: + # start: 0 + selector: + matchLabels: + app: example-sts + updateStrategy: + # type: RollingUpdate (default) or OnDelete (new pods only replace old + # ones when you delete them manually - full control for databases). + type: RollingUpdate + rollingUpdate: + # partition: only ordinals >= partition are updated. Lets you canary: + # partition 2 updates web-2 first, then lower to 0 for the rest. + partition: 0 + # maxUnavailable: how many pods may be down during the update (1.25+). + # StatefulSets traditionally allowed exactly 1; now configurable. + maxUnavailable: 1 + template: + metadata: + labels: + app: example-sts + spec: + serviceAccountName: default + automountServiceAccountToken: true + terminationGracePeriodSeconds: 60 + # Databases want a long grace: 30s kills a checkpointing postmaster + # mid-write. 60-120 is typical for postgres. + containers: + - name: db + image: postgres:17 + imagePullPolicy: IfNotPresent + ports: + - name: db + containerPort: 5432 + env: + - name: POSTGRES_USER + value: example + - name: POSTGRES_DB + value: example + - name: POSTGRES_PASSWORD + valueFrom: + secretKeyRef: + name: example-secrets + key: db-password + # Probes for a database: pg_isready via exec is the standard. + # Budgets are generous - killing a recovering database only buys + # another full replay. + startupProbe: + exec: + command: ["sh", "-c", "pg_isready -U example -d example"] + # failureThreshold * periodSeconds = total startup budget + # (60 * 10s = 10 minutes here). + failureThreshold: 60 + periodSeconds: 10 + timeoutSeconds: 5 + readinessProbe: + exec: + command: ["sh", "-c", "pg_isready -U example -d example"] + periodSeconds: 10 + timeoutSeconds: 5 + livenessProbe: + exec: + command: ["sh", "-c", "pg_isready -U example -d example"] + initialDelaySeconds: 30 + periodSeconds: 30 + timeoutSeconds: 5 + failureThreshold: 3 + resources: + requests: + cpu: "100m" + memory: "256Mi" + limits: + cpu: "1000m" + memory: "1Gi" + volumeMounts: + - name: data + mountPath: /var/lib/postgresql/data + # subPath: mounting the volume root directly breaks postgres + # (it wants an empty dir); subPath gives it a subdirectory. + subPath: pgdata + volumes: [] + # No shared volumes here: each pod gets its own PVC from the template + # below, mounted under the same `data` name. + volumeClaimTemplates: + # One template entry per volume. Pod web-N gets PVC data-web-N, + # created on first scheduling and (per the retention policy) kept after. + - metadata: + name: data + labels: + app: example-sts + annotations: + description: "per-pod database storage" + spec: + accessModes: + # Must be ReadWriteOnce: one pod owns its volume. (Shared RWX + # defeats the whole point of per-pod volumes.) + - ReadWriteOnce + storageClassName: standard + volumeMode: Filesystem + resources: + requests: + storage: 10Gi + # selector/dataSource/dataSourceRef: same semantics as a plain PVC + # (bind a specific PV, clone, restore from snapshot). See + # deployment.yaml. diff --git a/templates/tcproute.yaml b/templates/tcproute.yaml new file mode 100644 index 0000000..d073b74 --- /dev/null +++ b/templates/tcproute.yaml @@ -0,0 +1,35 @@ +# Exhaustive TCPRoute reference (gateway.networking.k8s.io/v1alpha2). +# Routes raw TCP (SSH, databases, any non-HTTP protocol) from a TCP listener +# to backends. No hostname/path matching exists at this layer - there is +# nothing but the destination port to route on. For TLS with SNI-based +# routing see TLSRoute; for HTTP see HTTPRoute. +apiVersion: gateway.networking.k8s.io/v1alpha2 +kind: TCPRoute +metadata: + name: app-ssh + namespace: example + labels: + app: app +spec: + parentRefs: + # Bind to the TCP listener of the Gateway. Same fields as HTTPRoute + # parentRefs: name/kind/group plus sectionName and/or port narrowing. + - name: example + kind: Gateway + group: gateway.networking.k8s.io + sectionName: tcp + port: 2222 + rules: + - backendRefs: + - name: app-service + # port: backend Service port. Required. + port: 2222 + kind: Service + group: "" + # weight: share of connections when several backends are listed. + weight: 1 + # Second backend: every new connection goes 3:1 here. Weights are + # the only traffic-shaping knob TCP routing has. + - name: app-replica-service + port: 2222 + weight: 3 diff --git a/templates/udproute.yaml b/templates/udproute.yaml new file mode 100644 index 0000000..2da833f --- /dev/null +++ b/templates/udproute.yaml @@ -0,0 +1,29 @@ +# Exhaustive UDPRoute reference (gateway.networking.k8s.io/v1alpha2). +# Routes raw UDP datagrams (DNS, STUN/TURN, game servers, QUIC-before-TLS) +# from a UDP listener to backends. Same minimal shape as TCPRoute: UDP has +# no sessions or headers to match on, so backendRefs carry the whole rule. +apiVersion: gateway.networking.k8s.io/v1alpha2 +kind: UDPRoute +metadata: + name: app-stun + namespace: example + labels: + app: app +spec: + parentRefs: + # Bind to the UDP listener of the Gateway. Same fields as HTTPRoute + # parentRefs: name/kind/group plus sectionName and/or port narrowing. + - name: example + kind: Gateway + group: gateway.networking.k8s.io + sectionName: udp + port: 3478 + rules: + - backendRefs: + - name: app-service + # port: backend Service port. Required. + port: 3478 + kind: Service + group: "" + # weight: share of traffic when several backends are listed. + weight: 1 From 78cd15f12c2c0268433a7bdc7e39730c9b7f3c48 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 17:53:24 +0200 Subject: [PATCH 18/53] chore(monitoring): remove vmctl backfill job One-shot Prometheus history backfill is complete; drop the Job and clean up related comments. --- prometheus-stack/k8s/grafana-values.yaml | 1 - .../k8s/victoria-secrets.yaml.example | 2 -- prometheus-stack/k8s/vmctl-backfill.yaml | 31 ------------------- 3 files changed, 34 deletions(-) delete mode 100644 prometheus-stack/k8s/vmctl-backfill.yaml diff --git a/prometheus-stack/k8s/grafana-values.yaml b/prometheus-stack/k8s/grafana-values.yaml index 5401346..2a2f436 100644 --- a/prometheus-stack/k8s/grafana-values.yaml +++ b/prometheus-stack/k8s/grafana-values.yaml @@ -39,7 +39,6 @@ grafana: # One block covers both the dashboards and datasources sidecars (p95 91M / 80M). sidecar: datasources: - # Chart built-in Prometheus DS disabled: VictoriaMetrics (below) is the default. defaultDatasourceEnabled: false resources: requests: diff --git a/prometheus-stack/k8s/victoria-secrets.yaml.example b/prometheus-stack/k8s/victoria-secrets.yaml.example index 3ef93ad..0d90180 100644 --- a/prometheus-stack/k8s/victoria-secrets.yaml.example +++ b/prometheus-stack/k8s/victoria-secrets.yaml.example @@ -1,5 +1,3 @@ -# Create the victoria-secrets Secret before deploying VictoriaMetrics. -# Copy the metrics password from uptime-kuma/k8s/secrets.yaml into the value below. apiVersion: v1 kind: Secret metadata: diff --git a/prometheus-stack/k8s/vmctl-backfill.yaml b/prometheus-stack/k8s/vmctl-backfill.yaml deleted file mode 100644 index 269ee46..0000000 --- a/prometheus-stack/k8s/vmctl-backfill.yaml +++ /dev/null @@ -1,31 +0,0 @@ -# One-shot history backfill Prometheus -> VictoriaMetrics via vmctl remote-read. -# Safe to keep applied: a completed Job is a no-op on re-apply. -apiVersion: batch/v1 -kind: Job -metadata: - name: vmctl-backfill - namespace: prometheus -spec: - backoffLimit: 2 - ttlSecondsAfterFinished: 3600 - template: - spec: - restartPolicy: OnFailure - containers: - - name: vmctl - image: victoriametrics/vmctl:v1.153.0 - args: - - remote-read - - -s - - --disable-progress-bar - - --remote-read-src-addr=http://prometheus-stack-kube-prom-prometheus.prometheus.svc.cluster.local:9090 - - --remote-read-filter-time-start=2026-09-06T10:26:14Z - - --remote-read-step-interval=day - - --vm-addr=http://victoria-metrics.prometheus.svc.cluster.local:8428 - resources: - requests: - cpu: "200m" - memory: "512Mi" - limits: - cpu: "1000m" - memory: "1Gi" From 8c0e36a5c0f848979461d1f7f5c29f2bc446b441 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 18:03:59 +0200 Subject: [PATCH 19/53] feat(monitoring): replace scrape dump with vmagent --- .gitea/workflows/ci.yaml | 5 + .gitea/workflows/deploy-lib.sh | 2 + prometheus-stack/k8s/README.md | 17 + .../k8s/victoria-operator-values.yaml | 12 + prometheus-stack/k8s/victoria-scrape.yaml | 1758 ----------------- .../k8s/victoria-secrets.yaml.example | 8 - prometheus-stack/k8s/victoria.yaml | 43 - prometheus-stack/k8s/vmagent.yaml | 29 + renovate/renovate.json | 9 + templates/deployment.yaml | 543 ----- templates/endpointslice.yaml | 63 - templates/gateway.yaml | 100 - templates/httproute.yaml | 181 -- templates/ingressroute-tcp.yaml | 52 - templates/ingressroute-udp.yaml | 19 - templates/ingressroute.yaml | 93 - templates/service.yaml | 100 - templates/statefulset.yaml | 166 -- templates/tcproute.yaml | 35 - templates/udproute.yaml | 29 - 20 files changed, 74 insertions(+), 3190 deletions(-) create mode 100644 prometheus-stack/k8s/README.md create mode 100644 prometheus-stack/k8s/victoria-operator-values.yaml delete mode 100644 prometheus-stack/k8s/victoria-scrape.yaml delete mode 100644 prometheus-stack/k8s/victoria-secrets.yaml.example create mode 100644 prometheus-stack/k8s/vmagent.yaml delete mode 100644 templates/deployment.yaml delete mode 100644 templates/endpointslice.yaml delete mode 100644 templates/gateway.yaml delete mode 100644 templates/httproute.yaml delete mode 100644 templates/ingressroute-tcp.yaml delete mode 100644 templates/ingressroute-udp.yaml delete mode 100644 templates/ingressroute.yaml delete mode 100644 templates/service.yaml delete mode 100644 templates/statefulset.yaml delete mode 100644 templates/tcproute.yaml delete mode 100644 templates/udproute.yaml diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 5552327..938696e 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -294,6 +294,11 @@ jobs: echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps" failed=0 for m in ${manifests[@]+"${manifests[@]}"}; do + if [[ "$m" == "prometheus-stack/k8s/vmagent.yaml" ]] \ + && ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then + echo "skip server-side dry-run until the VictoriaMetrics Operator CRD is installed: $m" + continue + fi if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then failed=1 echo "::error file=${m}::$(printf '%s' "$out" | head -1)" diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index d3cb30a..9429adc 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -546,6 +546,7 @@ rollback_workloads() { # have to be declared as custom.regex managers in renovate/renovate.json. HELM_RELEASES=( "prometheus-stack|prometheus-community/kube-prometheus-stack|prometheus|86.2.3|prometheus-stack/k8s/grafana-values.yaml|prometheus-stack/k8s/active" + "victoria-operator|victoriametrics/victoria-metrics-operator|prometheus|0.68.1|prometheus-stack/k8s/victoria-operator-values.yaml|prometheus-stack/k8s/active" "loki|grafana/loki|prometheus|7.3.0|loki/k8s/loki-values.yaml|loki/k8s/active" "alloy|grafana/alloy|prometheus|1.12.1|loki/k8s/alloy-values.yaml|loki/k8s/active" "reloader|stakater/reloader|reloader|2.2.17|reloader/k8s/reloader-values.yaml|reloader/k8s/active" @@ -557,6 +558,7 @@ helm_repo_for() { prometheus-community/*) echo "prometheus-community https://prometheus-community.github.io/helm-charts" ;; grafana/*) echo "grafana https://grafana.github.io/helm-charts" ;; stakater/*) echo "stakater https://stakater.github.io/stakater-charts" ;; + victoriametrics/*) echo "victoriametrics https://victoriametrics.github.io/helm-charts" ;; esac } diff --git a/prometheus-stack/k8s/README.md b/prometheus-stack/k8s/README.md new file mode 100644 index 0000000..494726c --- /dev/null +++ b/prometheus-stack/k8s/README.md @@ -0,0 +1,17 @@ +# VictoriaMetrics + +The `victoria-operator` Helm release converts Prometheus Operator +`ServiceMonitor` resources into owned `VMServiceScrape` resources. The +`VMAgent` selects converted scrapes labeled `release: prometheus-stack` in all +namespaces and writes them to the existing single-node VictoriaMetrics +instance. Changes to selected `ServiceMonitor` resources are reconciled +automatically; there is no copied Prometheus scrape-config blob to regenerate. + +The agent drops targets for the Prometheus server service to avoid duplicating +its self-scrape. `scraper: victoria` identifies the samples ingested by this +VMAgent. + +The VictoriaMetrics Operator chart and its CRDs are installed before the +Kubernetes manifests by the normal deploy workflow. On a cluster where the +operator CRDs are not installed yet, CI skips the server-side dry-run of the +`VMAgent` resource; the deploy installs the chart before applying that resource. diff --git a/prometheus-stack/k8s/victoria-operator-values.yaml b/prometheus-stack/k8s/victoria-operator-values.yaml new file mode 100644 index 0000000..9e5639a --- /dev/null +++ b/prometheus-stack/k8s/victoria-operator-values.yaml @@ -0,0 +1,12 @@ +nameOverride: victoria-operator + +operator: + enable_converter_ownership: true + +resources: + requests: + cpu: 50m + memory: 96Mi + limits: + cpu: 200m + memory: 256Mi diff --git a/prometheus-stack/k8s/victoria-scrape.yaml b/prometheus-stack/k8s/victoria-scrape.yaml deleted file mode 100644 index a510c48..0000000 --- a/prometheus-stack/k8s/victoria-scrape.yaml +++ /dev/null @@ -1,1758 +0,0 @@ -# Generated from the operator-rendered Prometheus config (secret -# prometheus-prometheus-stack-kube-prom-prometheus). Regenerate after -# ServiceMonitor changes; do not hand-edit scrape jobs below. -apiVersion: v1 -kind: ConfigMap -metadata: - name: victoria-scrape - namespace: prometheus -data: - scrape.yaml: | - global: - scrape_interval: 30s - scrape_timeout: 10s - external_labels: - scraper: victoria-trial - scrape_configs: - - job_name: serviceMonitor/cert-manager/cert-manager/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - cert-manager - scrape_interval: 60s - scrape_timeout: 30s - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (cainjector|cert-manager|webhook);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_instance - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance - regex: (cert-manager);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_component - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_component - regex: (cainjector|controller|webhook);true - - action: keep - source_labels: - - __meta_kubernetes_pod_container_port_name - regex: http-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: http-metrics - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/crowdsec/crowdsec-agent-service/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - crowdsec - attach_metadata: - node: true - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (crowdsec-agent-service);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: metrics - - source_labels: - - __meta_kubernetes_pod_node_name - target_label: machine - action: replace - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/crowdsec/crowdsec-service/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - crowdsec - attach_metadata: - node: true - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (crowdsec-service);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: metrics - - source_labels: - - __meta_kubernetes_pod_node_name - target_label: machine - action: replace - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/edu-master/webinar-checker/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - edu-master - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (edu-master-webinar-checker);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: metrics - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/loki/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - scrape_interval: 15s - metrics_path: /metrics - scheme: http - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_instance - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance - regex: (loki);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (loki);true - - action: drop - source_labels: - - __meta_kubernetes_service_label_prometheus_io_service_monitor - - __meta_kubernetes_service_labelpresent_prometheus_io_service_monitor - regex: (false);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: http-metrics - - source_labels: - - job - target_label: job - replacement: prometheus/$1 - action: replace - - target_label: cluster - replacement: loki - action: replace - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-grafana/0 - honor_labels: true - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - metrics_path: /metrics - scheme: http - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_instance - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (grafana);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-web - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_prometheus_stack - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: http-web - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-alertmanager/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-alertmanager);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_self_monitor - - __meta_kubernetes_service_labelpresent_self_monitor - regex: (true);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-web - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: http-web - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-alertmanager/1 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-alertmanager);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_self_monitor - - __meta_kubernetes_service_labelpresent_self_monitor - regex: (true);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: reloader-web - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: reloader-web - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-apiserver/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - default - scheme: https - tls_config: - insecure_skip_verify: false - server_name: kubernetes - ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt - bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_component - - __meta_kubernetes_service_labelpresent_component - regex: (apiserver);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_provider - - __meta_kubernetes_service_labelpresent_provider - regex: (kubernetes);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: https - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_component - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: https - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - metric_relabel_configs: - - source_labels: - - __name__ - - le - regex: (etcd_request|apiserver_request_slo|apiserver_request_sli|apiserver_request)_duration_seconds_bucket;(0\.15|0\.2|0\.3|0\.35|0\.4|0\.45|0\.6|0\.7|0\.8|0\.9|1\.25|1\.5|1\.75|2|3|3\.5|4|4\.5|6|7|8|9|15|20|40|45|50)(\.0)? - action: drop - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-coredns/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - kube-system - bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-coredns);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_jobLabel - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: http-metrics - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kube-proxy/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - kube-system - bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-kube-proxy);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_jobLabel - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: http-metrics - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kubelet/0 - honor_labels: true - honor_timestamps: true - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - kube-system - attach_metadata: - node: false - scheme: https - tls_config: - insecure_skip_verify: true - ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt - bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (kubelet);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_k8s_app - - __meta_kubernetes_service_labelpresent_k8s_app - regex: (kubelet);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: https-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_k8s_app - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: https-metrics - - source_labels: - - __metrics_path__ - target_label: metrics_path - action: replace - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - metric_relabel_configs: - - source_labels: - - __name__ - - le - regex: (csi_operations|storage_operation_duration)_seconds_bucket;(0.25|2.5|15|25|120|600)(\.0)? - action: drop - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kubelet/1 - honor_labels: true - honor_timestamps: true - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - kube-system - attach_metadata: - node: false - scrape_interval: 10s - metrics_path: /metrics/cadvisor - scheme: https - tls_config: - insecure_skip_verify: true - ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt - bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (kubelet);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_k8s_app - - __meta_kubernetes_service_labelpresent_k8s_app - regex: (kubelet);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: https-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_k8s_app - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: https-metrics - - source_labels: - - __metrics_path__ - target_label: metrics_path - action: replace - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - metric_relabel_configs: - - source_labels: - - __name__ - regex: container_cpu_(cfs_throttled_seconds_total|load_average_10s|system_seconds_total|user_seconds_total) - action: drop - - source_labels: - - __name__ - regex: container_fs_(io_current|io_time_seconds_total|io_time_weighted_seconds_total|reads_merged_total|sector_reads_total|sector_writes_total|writes_merged_total) - action: drop - - source_labels: - - __name__ - regex: container_memory_(mapped_file|swap) - action: drop - - source_labels: - - __name__ - regex: container_(file_descriptors|tasks_state|threads_max) - action: drop - - source_labels: - - __name__ - - scope - regex: container_memory_failures_total;hierarchy - action: drop - - source_labels: - - __name__ - - interface - regex: container_network_.*;(cali|cilium|cni|lxc|nodelocaldns|tunl).* - action: drop - - source_labels: - - __name__ - regex: container_spec.* - action: drop - - source_labels: - - id - - pod - regex: .+; - action: drop - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-kubelet/2 - honor_labels: true - honor_timestamps: true - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - kube-system - attach_metadata: - node: false - metrics_path: /metrics/probes - scheme: https - tls_config: - insecure_skip_verify: true - ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt - bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (kubelet);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_k8s_app - - __meta_kubernetes_service_labelpresent_k8s_app - regex: (kubelet);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: https-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_k8s_app - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: https-metrics - - source_labels: - - __metrics_path__ - target_label: metrics_path - action: replace - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-operator/0 - honor_labels: true - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - scheme: https - tls_config: - ca_file: /etc/prometheus/certs/0_prometheus_prometheus-stack-kube-prom-admission_ca - server_name: prometheus-stack-kube-prom-operator - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-operator);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: https - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: https - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-prometheus/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-prometheus);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_self_monitor - - __meta_kubernetes_service_labelpresent_self_monitor - regex: (true);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-web - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: http-web - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-prom-prometheus/1 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (kube-prometheus-stack-prometheus);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_release - - __meta_kubernetes_service_labelpresent_release - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_self_monitor - - __meta_kubernetes_service_labelpresent_self_monitor - regex: (true);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: reloader-web - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: reloader-web - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-kube-state-metrics/0 - honor_labels: true - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_instance - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (kube-state-metrics);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: http - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/prometheus-stack-prometheus-node-exporter/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - prometheus - attach_metadata: - node: false - scheme: http - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_instance - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance - regex: (prometheus-stack);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (prometheus-node-exporter);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http-metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_jobLabel - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: http-metrics - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/prometheus/traefik/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - traefik - metrics_path: /metrics - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_instance - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_instance - regex: (traefik-traefik);true - - action: keep - source_labels: - - __meta_kubernetes_service_label_app_kubernetes_io_name - - __meta_kubernetes_service_labelpresent_app_kubernetes_io_name - regex: (traefik);true - - action: keep - source_labels: - - __meta_kubernetes_pod_container_port_name - regex: metrics - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - source_labels: - - __meta_kubernetes_service_label_traefik - target_label: job - regex: (.+) - replacement: ${1} - - target_label: endpoint - replacement: metrics - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod - - job_name: serviceMonitor/uptime-kuma/uptime-kuma/0 - honor_labels: false - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - uptime-kuma - scrape_interval: 60s - scrape_timeout: 15s - metrics_path: /metrics - basic_auth: - username: uptime-kuma - password_file: /etc/vm/secrets/uptime-kuma-password - relabel_configs: - - source_labels: - - job - target_label: __tmp_prometheus_job_name - - action: keep - source_labels: - - __meta_kubernetes_service_label_app - - __meta_kubernetes_service_labelpresent_app - regex: (uptime-kuma);true - - action: keep - source_labels: - - __meta_kubernetes_endpoint_port_name - regex: http - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Node;(.*) - replacement: ${1} - target_label: node - - source_labels: - - __meta_kubernetes_endpoint_address_target_kind - - __meta_kubernetes_endpoint_address_target_name - separator: ; - regex: Pod;(.*) - replacement: ${1} - target_label: pod - - source_labels: - - __meta_kubernetes_namespace - target_label: namespace - - source_labels: - - __meta_kubernetes_service_name - target_label: service - - source_labels: - - __meta_kubernetes_pod_name - target_label: pod - - source_labels: - - __meta_kubernetes_pod_container_name - target_label: container - - action: drop - source_labels: - - __meta_kubernetes_pod_phase - regex: (Failed|Succeeded) - - source_labels: - - __meta_kubernetes_service_name - target_label: job - replacement: ${1} - - target_label: endpoint - replacement: http - - source_labels: - - __address__ - - __tmp_hash - target_label: __tmp_hash - regex: (.+); - replacement: $1 - action: replace - - source_labels: - - __tmp_hash - target_label: __tmp_hash - modulus: 1 - action: hashmod diff --git a/prometheus-stack/k8s/victoria-secrets.yaml.example b/prometheus-stack/k8s/victoria-secrets.yaml.example deleted file mode 100644 index 0d90180..0000000 --- a/prometheus-stack/k8s/victoria-secrets.yaml.example +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: v1 -kind: Secret -metadata: - name: victoria-secrets - namespace: prometheus -type: Opaque -stringData: - uptime-kuma-password: "REPLACE_ME" diff --git a/prometheus-stack/k8s/victoria.yaml b/prometheus-stack/k8s/victoria.yaml index d3adcdc..0fce89e 100644 --- a/prometheus-stack/k8s/victoria.yaml +++ b/prometheus-stack/k8s/victoria.yaml @@ -1,23 +1,4 @@ apiVersion: v1 -kind: ServiceAccount -metadata: - name: victoria - namespace: prometheus ---- -apiVersion: rbac.authorization.k8s.io/v1 -kind: ClusterRoleBinding -metadata: - name: victoria -roleRef: - apiGroup: rbac.authorization.k8s.io - kind: ClusterRole - name: prometheus-stack-kube-prom-prometheus -subjects: - - kind: ServiceAccount - name: victoria - namespace: prometheus ---- -apiVersion: v1 kind: Service metadata: name: victoria-metrics @@ -59,7 +40,6 @@ spec: labels: app: victoria-metrics spec: - serviceAccountName: victoria containers: - name: victoria image: victoriametrics/victoria-metrics:v1.153.0-scratch @@ -67,8 +47,6 @@ spec: - -storageDataPath=/vmdata - -retentionPeriod=30d - -httpListenAddr=:8428 - - -promscrape.config=/etc/vm/conf/scrape.yaml - - -promscrape.configCheckInterval=60s ports: - containerPort: 8428 readinessProbe: @@ -88,15 +66,6 @@ spec: volumeMounts: - name: vmdata mountPath: /vmdata - - name: scrape-config - mountPath: /etc/vm/conf - readOnly: true - - name: vm-secrets - mountPath: /etc/vm/secrets - readOnly: true - - name: prom-admission-ca - mountPath: /etc/prometheus/certs - readOnly: true resources: requests: cpu: "100m" @@ -108,15 +77,3 @@ spec: - name: vmdata persistentVolumeClaim: claimName: victoria-pvc - - name: scrape-config - configMap: - name: victoria-scrape - - name: vm-secrets - secret: - secretName: victoria-secrets - - name: prom-admission-ca - secret: - secretName: prometheus-stack-kube-prom-admission - items: - - key: ca - path: 0_prometheus_prometheus-stack-kube-prom-admission_ca diff --git a/prometheus-stack/k8s/vmagent.yaml b/prometheus-stack/k8s/vmagent.yaml new file mode 100644 index 0000000..aed8964 --- /dev/null +++ b/prometheus-stack/k8s/vmagent.yaml @@ -0,0 +1,29 @@ +apiVersion: operator.victoriametrics.com/v1beta1 +kind: VMAgent +metadata: + name: vmagent + namespace: prometheus +spec: + image: + tag: v1.153.0 + scrapeInterval: 30s + externalLabels: + scraper: victoria + serviceScrapeNamespaceSelector: {} + serviceScrapeSelector: + matchLabels: + release: prometheus-stack + globalScrapeRelabelConfigs: + - action: drop + source_labels: + - __meta_kubernetes_service_name + regex: prometheus-stack-kube-prom-prometheus + remoteWrite: + - url: http://victoria-metrics.prometheus.svc.cluster.local:8428/api/v1/write + resources: + requests: + cpu: 100m + memory: 256Mi + limits: + cpu: "1000m" + memory: 1Gi diff --git a/renovate/renovate.json b/renovate/renovate.json index b65f8bd..e9eae92 100644 --- a/renovate/renovate.json +++ b/renovate/renovate.json @@ -52,6 +52,15 @@ "depNameTemplate": "kube-prometheus-stack", "registryUrlTemplate": "https://prometheus-community.github.io/helm-charts" }, + { + "customType": "regex", + "description": "VictoriaMetrics Operator chart version pinned in the deploy workflow", + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], + "matchStrings": ["\\|victoriametrics/victoria-metrics-operator\\|prometheus\\|(?[0-9.]+)\\|"], + "datasourceTemplate": "helm", + "depNameTemplate": "victoria-metrics-operator", + "registryUrlTemplate": "https://victoriametrics.github.io/helm-charts" + }, { "customType": "regex", "description": "grafana/loki chart version pinned in the deploy workflow", diff --git a/templates/deployment.yaml b/templates/deployment.yaml deleted file mode 100644 index 5e65504..0000000 --- a/templates/deployment.yaml +++ /dev/null @@ -1,543 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - # name: unique within the namespace - name: nginx-deployment - # namespace: logical isolation - namespace: example - labels: - # free-form key/value tags used for selection and grouping. - app: nginx - app.kubernetes.io/name: nginx - app.kubernetes.io/instance: nginx-example - app.kubernetes.io/version: "1.27" - app.kubernetes.io/component: server - app.kubernetes.io/part-of: example - app.kubernetes.io/managed-by: kubectl - annotations: - # Key/value, but never used for selection - # Only for metadata (descriptions, owners, timestamps). - description: "deployment example" - owner: team-platform - # finalizers: identifiers that block deletion until some controller removes - # them after cleanup. RARE on workloads. - # finalizers: - # - example.com/cleanup -spec: - # replicas: how many pod copies to keep running. Default 1. - replicas: 3 - # revisionHistoryLimit: how many old ReplicaSets are kept so you can roll - # back. Default 10. - revisionHistoryLimit: 5 - # progressDeadlineSeconds: if a rollout makes no progress for this long it - # is marked ProgressDeadlineExceeded. Default 600. - progressDeadlineSeconds: 600 - # minReadySeconds: a new pod must stay Ready for this long before it counts - # as available. Protects against pods that flap right after start. - minReadySeconds: 10 - # paused: freezes the rollout controller. RARE - used to accumulate several - # changes and release them as a single rollout. - paused: false - # selector: defines which pods belong to this Deployment. Must match the - # pod template labels exactly. Immutable after creation. - selector: - matchLabels: - app: nginx - # matchExpressions: set-based selection (In, NotIn, Exists, DoesNotExist). - # RARE on Deployments. - # matchExpressions: - # - key: tier - # operator: In - # values: [frontend] - strategy: - # type: RollingUpdate (replace gradually, default) or Recreate (kill all - # old pods first). Recreate fits state that cannot have two writers at - # once (SQLite file, exclusive lock). - type: RollingUpdate - rollingUpdate: - # maxSurge: how many pods above `replicas` may exist mid-rollout. - # Number or percentage. - maxSurge: 1 - # maxUnavailable: how many pods may be simultaneously down mid-rollout. - # Number or percentage. - maxUnavailable: 1 - template: - metadata: - labels: - app: nginx - app.kubernetes.io/name: nginx - annotations: - description: "nginx pod" - spec: - # serviceAccountName: the identity pods use against the API server. - serviceAccountName: default - # automountServiceAccountToken: mount the API token into pods. Set false - # for pods that never call the API to shrink the escape blast radius. - automountServiceAccountToken: true - # schedulerName: which scheduler places the pod. The default scheduler - # handles virtually everything. - schedulerName: default-scheduler - # nodeName: pin the pod to one node, bypassing the scheduler. RARE and - # brittle - nodeSelector/affinity express intent better. - # nodeName: node-1 - # nodeSelector: hard requirement on node labels. - nodeSelector: - kubernetes.io/os: linux - # hostname/subdomain: give the pod a stable hostname and DNS entry - # ...svc.cluster.local. Mostly a - # StatefulSet concern (which gets this automatically). - hostname: nginx - subdomain: example-subdomain - # setHostnameAsFQDN: use the FQDN above as the hostname. Default false. - setHostnameAsFQDN: false - # priorityClassName: scheduling priority; higher values preempt lower. - # priorityClassName: high-priority - # preemptionPolicy: Never stops this pod from preempting others. - # Default PreemptLowerPriority. - preemptionPolicy: PreemptLowerPriority - # runtimeClassName: alternate container runtime (gVisor, Kata). - # Omit for the default runtime. - # runtimeClassName: gvisor - # enableServiceLinks: inject _SERVICE_HOST style env vars. True by - # default; false keeps the environment clean when you use DNS only. - enableServiceLinks: true - # hostAliases: extra /etc/hosts lines. RARE - usually means DNS should - # have been fixed instead. - hostAliases: - - ip: 192.168.1.10 - hostnames: - - legacy-db.example.com - # hostNetwork/hostPID/hostIPC: share the node's network/process/IPC - # namespaces. Needed for node-level agents; dangerous for apps. - hostNetwork: false - hostPID: false - hostIPC: false - # shareProcessNamespace: all containers in the pod see each other's - # processes. Handy for sidecar debuggers; off by default. - shareProcessNamespace: false - # dnsPolicy: ClusterFirst (default, .svc names resolve), Default - # (inherit the node's resolver), ClusterFirstWithHostNet (for - # hostNetwork pods), None (dnsConfig takes over completely). - dnsPolicy: ClusterFirst - dnsConfig: - nameservers: - - 1.1.1.1 - searches: - - example.com - options: - - name: ndots - value: "2" - # readinessGates: custom conditions (reported by external controllers) - # that must be true before the pod counts as Ready. RARE. - # readinessGates: - # - conditionType: example.com/lb-attached - # topologySpreadConstraints: spread pods across zones/hosts with skew - # control. The modern, expressive successor of bare podAntiAffinity. - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - app: nginx - affinity: - nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: hard node rule - - # pods that violate it are never scheduled there. - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: kubernetes.io/arch - operator: In - values: [amd64] - # preferredDuringSchedulingIgnoredDuringExecution: soft node rule - - # the scheduler tries, but schedules anyway if impossible. - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 10 - preference: - matchExpressions: - - key: node-role.kubernetes.io/worker - operator: Exists - podAffinity: - # Attract to nodes already running matching pods (data locality). - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 50 - podAffinityTerm: - labelSelector: - matchLabels: - app: cache - topologyKey: kubernetes.io/hostname - podAntiAffinity: - # Repel from nodes running matching pods (spread replicas). - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 100 - podAffinityTerm: - labelSelector: - matchLabels: - app: nginx - topologyKey: kubernetes.io/hostname - tolerations: - # Tolerate a node taint to allow scheduling there. Without a matching - # toleration, a tainted node rejects the pod. - - key: dedicated - operator: Equal - value: "true" - effect: NoSchedule - # tolerationSeconds: with effect NoExecute, tolerate for this long - # before eviction. Omit for infinite tolerance. - tolerationSeconds: 3600 - # imagePullSecrets: credentials for private registries. - # imagePullSecrets: - # - name: registry-credentials - # securityContext (pod-level): defaults inherited by every container - # unless a container overrides them. - securityContext: - runAsUser: 101 - runAsGroup: 101 - # runAsNonRoot: refuse to start as uid 0. - runAsNonRoot: true - # fsGroup: group that owns mounted volumes; kubelet chowns on mount. - fsGroup: 101 - # fsGroupChangePolicy: Always (chown on every mount, slow on large - # volumes) or OnRootMismatch (chown only when needed). - fsGroupChangePolicy: OnRootMismatch - # supplementalGroups: extra groups granted for volume access. - supplementalGroups: [102] - # sysctls: namespaced kernel parameters. Unsafe ones need explicit - # kubelet opt-in. - sysctls: - - name: net.core.somaxconn - value: "1024" - # seLinuxOptions: SELinux user/role/type/level. RARE outside MLS. - seLinuxOptions: - level: s0:c123,c456 - # seccompProfile: syscall sandbox. RuntimeDefault is the sane - # baseline; Localhost loads a custom profile from the node. - seccompProfile: - type: RuntimeDefault - # windowsOptions: GMSA / runAsUserName for Windows nodes. - initContainers: - # Run strictly in order, each to completion, before app containers - # start. Used for migrations, permission fixes, dependency waits. - - name: init-permissions - image: busybox:1.36 - command: ["sh", "-c", "chown -R 101:101 /data"] - volumeMounts: - - name: html - mountPath: /data - resources: - requests: - cpu: "10m" - memory: "16Mi" - limits: - cpu: "50m" - memory: "64Mi" - # restartPolicy on a container (not the pod): Always turns it into - # a native sidecar that keeps running next to the app (1.28+). - # restartPolicy: Always - containers: - - name: nginx - # image: repository plus tag. Pin tags - `latest` moves under you. - image: nginx:1.27.3 - # imagePullPolicy: Always (re-pull even pinned tags), IfNotPresent - # (use cache, works offline), Never (cache only, fails otherwise). - imagePullPolicy: IfNotPresent - # command: overrides the image ENTRYPOINT. args: overrides CMD. - # command: ["nginx"] - # args: ["-g", "daemon off;"] - # workingDir: overrides the image WORKDIR. - workingDir: /usr/share/nginx/html - # stdin/stdinOnce/tty: interactive input. For debug shells and - # one-shot runs, never for servers. - stdin: false - stdinOnce: false - tty: false - ports: - - name: http - containerPort: 80 - protocol: TCP - # hostPort: expose straight on the node, bypassing Services. - # RARE - allows only one such pod per node per port. - # hostPort: 8080 - # hostIP: which node address hostPort binds to. - # hostIP: 127.0.0.1 - env: - - name: NGINX_PORT - value: "80" - - name: POD_NAME - valueFrom: - fieldRef: - # fieldPath exposes pod metadata: metadata.name, - # metadata.namespace, metadata.uid, spec.nodeName, - # spec.serviceAccountName, status.podIP(s), etc. - fieldPath: metadata.name - - name: NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_MEMORY_LIMIT - valueFrom: - # resourceFieldRef exposes this container's own - # requests/limits. divisor formats the value. - resourceFieldRef: - resource: limits.memory - divisor: 1Mi - - name: API_PASSWORD - valueFrom: - secretKeyRef: - name: example-secrets - key: api-password - # optional: tolerate a missing key (variable stays unset). - optional: false - - name: LOG_LEVEL - valueFrom: - configMapKeyRef: - name: example-config - key: log-level - optional: false - envFrom: - # Bulk-inject every key of a ConfigMap/Secret as env vars. - - configMapRef: - name: example-config - optional: false - # prefix: prepended to every injected key, avoids collisions. - prefix: APP_ - - secretRef: - name: example-secrets - optional: false - resources: - # requests: guaranteed reservation used for scheduling. Set at - # measured idle/p95 - over-requesting starves neighboring pods. - requests: - cpu: "100m" - memory: "128Mi" - # limits: hard ceiling. Breaching memory kills the container - # (OOMKilled); breaching CPU only throttles it (slow, not dead). - limits: - cpu: "500m" - memory: "512Mi" - # claims: reference a ResourceClaim for dynamic resources - # (GPUs via DRA, 1.26+). RARE. - # claims: - # - name: gpu - # resizePolicy: what happens on in-place container resize (1.27+). - # NotRequired keeps running; RestartContainer restarts to apply. - resizePolicy: - - resourceName: cpu - restartPolicy: NotRequired - - resourceName: memory - restartPolicy: NotRequired - volumeMounts: - - name: html - mountPath: /usr/share/nginx/html - readOnly: false - # subPath: mount a single file/dir of the volume instead of - # its root. Typical for single-file config mounts. - # subPath: index.html - # subPathExpr: subPath assembled from env variables. - # subPathExpr: $(POD_NAME)/data - # mountPropagation: share mounts back with the host - # (HostToContainer, Bidirectional). Storage-driver territory. - mountPropagation: None - - name: tmp - mountPath: /tmp - # volumeDevices: raw block devices without a filesystem. RARE - - # databases on local PVs with volumeMode: Block. - # volumeDevices: - # - name: blockvol - # devicePath: /dev/xvda - livenessProbe: - # Exactly one handler per probe: httpGet, tcpSocket, exec, grpc. - httpGet: - path: /healthz - port: http - scheme: HTTP - # httpHeaders: extra headers sent with the probe request. - httpHeaders: - - name: Host - value: example.com - # initialDelaySeconds: wait after start before first probe. - initialDelaySeconds: 15 - # periodSeconds: interval between probes. - periodSeconds: 20 - # timeoutSeconds: when a single probe counts as failed. - timeoutSeconds: 5 - # successThreshold: consecutive successes to count as healthy. - # Keep 1. - successThreshold: 1 - # failureThreshold: consecutive failures to trigger the action. - failureThreshold: 3 - readinessProbe: - # Failing readiness removes the pod from Services (no traffic) - # without restarting it. Failing liveness restarts it. - httpGet: - path: /readyz - port: 80 - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 3 - successThreshold: 1 - failureThreshold: 3 - startupProbe: - # Disables liveness/readiness until it first succeeds. Total - # budget = failureThreshold * periodSeconds (here 30 * 10s). - # The cure for slow-starting apps that otherwise get - # restart-looped before they finish booting. - tcpSocket: - port: 80 - # host: probe a different host than the pod IP. RARE. - failureThreshold: 30 - periodSeconds: 10 - timeoutSeconds: 5 - lifecycle: - # postStart runs right after start; preStop runs before SIGTERM. - # Slow hooks stall the pod transition - keep them fast. - postStart: - exec: - command: ["sh", "-c", "echo started > /tmp/started"] - # Classic preStop: sleep so endpoints are removed everywhere - # before the process receives SIGTERM. - preStop: - exec: - command: ["sh", "-c", "sleep 5"] - # terminationMessagePath: file whose content becomes the container's - # final status message. FallbackToLogsOnError appends log tail when - # the file is empty. - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - securityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 101 - runAsGroup: 101 - capabilities: - add: - - NET_BIND_SERVICE - drop: - - ALL - seccompProfile: - type: RuntimeDefault - # ephemeralContainers: troubleshooting shells injected into a RUNNING - # pod with kubectl debug. Declared ad-hoc in practice, never committed. - # ephemeralContainers: - # - name: debugger - # image: busybox:1.36 - # command: ["sh"] - # stdin: true - # tty: true - # targetContainerName: nginx - volumes: - - name: html - persistentVolumeClaim: - claimName: example-pvc - # readOnly: mount the claim read-only in this pod. - readOnly: false - - name: tmp - emptyDir: - # medium "" (node disk) or Memory (tmpfs). sizeLimit evicts the - # pod when exceeded. Dies with the pod either way. - medium: "" - sizeLimit: 256Mi - - name: config-files - configMap: - # Each key becomes a file under the mount path. - name: example-config-files - defaultMode: 0644 - optional: false - items: - - key: nginx.conf - path: nginx.conf - mode: 0644 - - name: tls - secret: - # Secret volumes are tmpfs-backed, never touch node disk. - secretName: example-tls - defaultMode: 0644 - optional: false - items: - - key: tls.crt - path: tls.crt - mode: 0644 - - name: host-time - hostPath: - path: /etc/localtime - # type: DirectoryOrCreate, Directory, FileOrCreate, File, - # Socket, CharDevice, BlockDevice. Always set it: a missing - # path then fails loudly instead of silently creating the - # wrong filesystem object. - type: File - - name: podinfo - downwardAPI: - # Expose pod metadata as files. - items: - - path: labels - fieldRef: - fieldPath: metadata.labels - - path: cpu-request - resourceFieldRef: - resource: requests.cpu - containerName: nginx - divisor: 1m - - name: all-config - projected: - # Merge several sources into a single directory. - defaultMode: 0644 - sources: - - configMap: - name: example-config - items: - - key: log-level - path: log-level - - secret: - name: example-secrets - items: - - key: api-password - path: password - - downwardAPI: - items: - - path: podname - fieldRef: - fieldPath: metadata.name - # Further volume types share the same `name:` + type shape: - # nfs: { server, path }, csi: (storage drivers), persistentVolumeClaim - # shown above, plus legacy in-tree plugins (fc, iscsi, rbd, glusterfs) - # and gitRepo (deprecated - use initContainers + emptyDir instead). - # restartPolicy: only Always is valid for Deployments. (Jobs use - # OnFailure/Never; bare pods accept all three.) - restartPolicy: Always - terminationGracePeriodSeconds: 30 - # activeDeadlineSeconds: kill the pod after this long no matter what. - # A Job concern, not a server concern - shown only for completeness. - # activeDeadlineSeconds: 3600 ---- -apiVersion: v1 -kind: PersistentVolumeClaim -metadata: - name: example-pvc - namespace: example -spec: - # accessModes: ReadWriteOnce (one node), ReadOnlyMany, ReadWriteMany - # (needs a shared filesystem), ReadWriteOncePod (single pod, strictest). - accessModes: - - ReadWriteOnce - # storageClassName: selects the provisioner. "" (empty) disables dynamic - # provisioning and binds a pre-created volume instead. - storageClassName: standard - # volumeMode: Filesystem (default) or Block (used with volumeDevices). - volumeMode: Filesystem - resources: - requests: - storage: 5Gi - # limits: accepted on paper, almost no provisioner enforces them. - # selector: bind a specific pre-created PV by labels. RARE with dynamic - # provisioning. - # selector: - # matchLabels: - # disk: ssd - # dataSource: clone this PVC from another PVC or a snapshot at creation. - # dataSource: - # name: example-snapshot - # kind: VolumeSnapshot - # apiGroup: snapshot.storage.k8s.io - # dataSourceRef: cross-namespace-capable successor of dataSource. diff --git a/templates/endpointslice.yaml b/templates/endpointslice.yaml deleted file mode 100644 index 6ac1892..0000000 --- a/templates/endpointslice.yaml +++ /dev/null @@ -1,63 +0,0 @@ -# Exhaustive EndpointSlice reference (discovery.k8s.io/v1). -# EndpointSlices are the address lists behind a Service: each entry says -# "IP X serves port Y, ready or not". Normally the endpoint controller writes -# them automatically from the Service selector. You write one by hand in a -# single case: a selector-less Service pointing OUTSIDE the cluster (a host -# daemon, a LAN appliance, an external database). The pair looks like: -# Service (no selector, same port names) + this EndpointSlice. -apiVersion: discovery.k8s.io/v1 -kind: EndpointSlice -metadata: - name: app-external-abc123 - namespace: example - labels: - # kubernetes.io/service-name: THE binding label. It must equal the - # selector-less Service name - this is what attaches the slice to it. - # The endpoint controller owns slices it creates; hand-written ones - # just need this label to be picked up. - kubernetes.io/service-name: app-external-service - app: app - annotations: - description: "hand-written slice for an off-cluster backend" -# addressType: IPv4, IPv6, or FQDN. All endpoints in one slice share it - -# mix families with one slice per family. -addressType: IPv4 -ports: - - name: http - # name: MUST match the Service port name it serves. - protocol: TCP - # port: the REAL backend port (may differ from the Service port - the - # Service port is the in-cluster alias, this is where packets go). - port: 8080 - # appProtocol: payload hint, mirrors the Service field. - appProtocol: http -endpoints: - # One entry per backend address. kube-proxy load-balances across the ones - # whose conditions say ready+serving+terminating=false. - - addresses: - # addresses: one or more IPs (or a single DNS name for FQDN slices). - - 192.168.1.50 - conditions: - # ready: backend accepts traffic. False removes it from rotation - # without deleting the entry (flap-friendly). - ready: true - # serving: the process is up. Differs from ready during shutdown: - # serving=false + terminating=true = draining. - serving: true - # terminating: the endpoint is going away. Draining traffic, not dead. - terminating: false - # hostname: DNS name published for this endpoint (headless Services). - # hostname: backend-1 - # targetRef: link back to the pod/node object (set automatically on - # controller-managed slices; omit on hand-written ones). - # targetRef: - # kind: Pod - # namespace: example - # name: app-67890abcde-fghij - # uid: 12345678-1234-1234-1234-123456789abc - # nodeName: the node hosting this endpoint (topology-aware routing). - # zone: override the endpoint zone (defaults from nodeName). - # hints: topology hints for zone-aware routing (PreferClose). - # hints: - # forZones: - # - name: zone-a diff --git a/templates/gateway.yaml b/templates/gateway.yaml deleted file mode 100644 index bb36b7a..0000000 --- a/templates/gateway.yaml +++ /dev/null @@ -1,100 +0,0 @@ -# Exhaustive Gateway reference (gateway.networking.k8s.io/v1). -# A Gateway is the entry door: it owns listener ports/protocols/hostnames and -# delegates actual routing to Route objects (HTTPRoute, TCPRoute, ...), which -# attach via parentRefs. One Gateway usually fronts many Routes. -apiVersion: gateway.networking.k8s.io/v1 -kind: Gateway -metadata: - name: example - namespace: example - labels: - app: example - annotations: - description: "exhaustive gateway example" -spec: - # gatewayClassName: which controller implements this Gateway - # (kubectl get gatewayclass). The controller only touches Gateways naming - # its own class; anything else stays Ignored. - gatewayClassName: example-class - # addresses: VIPs/hostnames to request for the Gateway. Most controllers - # (including cloud LBs) allocate and fill status.addresses automatically; - # setting this pins a static IP. Omit for auto-assignment. - # addresses: - # - type: IPAddress - # value: 203.0.113.10 - # - type: Hostname - # value: lb.example.com - # infrastructure: controller-specific settings for the provisioned data - # plane (labels/annotations propagated to it). RARE - most setups never - # need it. - # infrastructure: - # labels: - # environment: prod - # annotations: - # example.com/keep: "true" - listeners: - # Each listener = one port + protocol + optional hostname + TLS + which - # Routes may attach. Listener names are referenced by Route parentRefs - # via sectionName. - - name: http - # port: 1-65535. Must be free on the data plane (controllers often - # require 80/443 to match their own entrypoints, otherwise the - # listener is marked Invalid/Conflicted). - port: 80 - # protocol: HTTP, HTTPS, TLS, TCP, UDP. - protocol: HTTP - # hostname: restrict this listener to one DNS name. Omit to accept all - # (Routes then narrow via their own hostnames). Listener hostname and - # Route hostnames must intersect or the Route is rejected. - hostname: app.example.com - # allowedRoutes: which Routes may bind here. - allowedRoutes: - namespaces: - # from: Same (only this namespace), All (any namespace), or - # Selector (namespaces matching the selector below). - from: Same - # selector: used only with from: Selector. - # selector: - # matchLabels: - # shared-gateway-access: "true" - # kinds: restrict by Route kind. Default allows whatever the - # listener protocol supports (HTTPRoute on HTTP, etc.). - kinds: - - kind: HTTPRoute - - name: https - port: 443 - protocol: HTTPS - hostname: app.example.com - # tls: termination settings for HTTPS/TLS listeners. - tls: - # mode: Terminate (decrypt here, default) or Passthrough (forward - # encrypted bytes to the backend - the backend holds the key). - mode: Terminate - # certificateRefs: TLS Secrets (or other kinds) in the SAME namespace - # (cross-namespace needs a ReferenceGrant). SNI picks among them. - certificateRefs: - - name: app-prod-tls - # kind/group default to Secret / core. Other kinds (e.g. a - # cert-manager Certificate via a plugin) set kind + group. - kind: Secret - group: "" - # options: controller-specific TLS knobs, referenced by name - # (cipher suites, min version). RARE. - # options: - # name: tls-options - - name: tcp - port: 2222 - protocol: TCP - allowedRoutes: - namespaces: - from: Same - kinds: - - kind: TCPRoute - - name: udp - port: 3478 - protocol: UDP - allowedRoutes: - namespaces: - from: Same - kinds: - - kind: UDPRoute diff --git a/templates/httproute.yaml b/templates/httproute.yaml deleted file mode 100644 index 68fd0ad..0000000 --- a/templates/httproute.yaml +++ /dev/null @@ -1,181 +0,0 @@ -# Exhaustive HTTPRoute reference (gateway.networking.k8s.io/v1). -# Attaches to a Gateway listener via parentRefs and routes HTTP(S) by -# hostname + path/method/headers/query. Rules are evaluated in order; the -# first matching rule wins. -apiVersion: gateway.networking.k8s.io/v1 -kind: HTTPRoute -metadata: - name: app - namespace: example - labels: - app: app -spec: - parentRefs: - # Every entry picks one listener to bind to. Omit sectionName/port to - # attach to ALL listeners of the Gateway (common for simple setups). - - name: example - # namespace: Gateway's namespace. Omit when same-namespace (the norm; - # cross-namespace needs the Gateway to allow it). - # namespace: example - # kind/group: default Gateway / gateway.networking.k8s.io. Set - # explicitly only for non-Gateway parents (mesh service parents). - kind: Gateway - group: gateway.networking.k8s.io - # sectionName: the listener name from the Gateway (http/https/...). - sectionName: https - # port: narrow further to one listener port. RARE when sectionName is - # already set. - port: 443 - # hostnames: which Host headers this Route serves. Must intersect the - # listener hostname; otherwise the Route is rejected as incompatible. - # Omit to match every hostname on the listener. - hostnames: - - app.example.com - - www.example.com - rules: - # Rule 1: API traffic with header manipulation and canary split. - - matches: - # All conditions inside one match are ANDed; several matches in the - # list are ORed. - - path: - # type: Exact (one URL), PathPrefix (subtree), RegularExpression. - type: PathPrefix - value: /api - method: POST - headers: - # type: Exact or RegularExpression. Names are case-insensitive - # per HTTP spec; values are case-sensitive. - - type: Exact - name: X-Api-Version - value: v2 - queryParams: - # Match on ?debug=true style parameters. - - type: Exact - name: debug - value: "true" - filters: - # Filters run in order and transform the request/response. - - type: RequestHeaderModifier - requestHeaderModifier: - # add: append even if present (duplicates allowed). set: - # overwrite-or-add. remove: delete by name. - add: - - name: X-Gateway - value: example - set: - - name: X-Forwarded-Proto - value: https - remove: - - X-Internal-Token - - type: ResponseHeaderModifier - responseHeaderModifier: - set: - - name: X-Frame-Options - value: DENY - remove: - - Server - # backendRefs below carry per-backend filters too; rule-level filters - # apply to every backend of this rule. - backendRefs: - - name: app-service - # port: the Service port (number). Required - unlike backendRef in - # Ingress, there is no default. - port: 80 - # group/kind: default Service / core (""). Other kinds (e.g. a - # ServiceImport for multi-cluster) set kind + group explicitly. - kind: Service - group: "" - # weight: traffic share. 90/10 below = canary: 90% stable, 10% new. - weight: 90 - filters: - # Per-backend filter: only this backend's requests get it. - - type: RequestHeaderModifier - requestHeaderModifier: - set: - - name: X-Backend - value: stable - - name: app-canary-service - port: 80 - weight: 10 - filters: - - type: RequestHeaderModifier - requestHeaderModifier: - set: - - name: X-Backend - value: canary - # timeouts: per-attempt deadlines. request = whole gateway-to-client - # exchange; backendRequest = single backend try. - timeouts: - request: 30s - backendRequest: 10s - # sessionPersistence: stick a client to one backend (cookie-based). - # Type Cookie or Header; absoluteTimeout caps the stickiness. - sessionPersistence: - sessionName: route-session - type: Cookie - absoluteTimeout: 1h - cookieConfig: - lifetimeType: Session - # Rule 2: redirect old path to a new URL. - - matches: - - path: - type: PathPrefix - value: /old-docs - filters: - - type: RequestRedirect - requestRedirect: - # Any combination: scheme/host/port/path/statusCode. Unset fields - # keep the original value. - scheme: https - hostname: docs.example.com - path: - # type: ReplaceFullPath or ReplacePrefixMatch (rewrite the - # matched prefix, keep the remainder). - type: ReplacePrefixMatch - replacePrefixMatch: /docs - port: 443 - # statusCode: 301 (permanent) or 302 (temporary). - statusCode: 301 - # Rule 3: rewrite the URL but still proxy (client sees no redirect). - - matches: - - path: - type: PathPrefix - value: /shop - filters: - - type: URLRewrite - urlRewrite: - path: - type: ReplacePrefixMatch - replacePrefixMatch: /store - # hostname: also rewrite the Host header sent upstream. - # hostname: store-internal.example.com - backendRefs: - - name: app-service - port: 80 - # Rule 4: mirror (shadow) traffic to a second backend for testing. - # The mirror gets a copy; its response is discarded. - - matches: - - path: - type: Exact - value: /checkout - filters: - - type: RequestMirror - requestMirror: - backendRef: - name: app-shadow-service - port: 80 - backendRefs: - - name: app-service - port: 80 - # Rule 5: delegate to an implementation-specific filter (auth plugin, - # rate limit, wasm). The controller documents the group/kind it honors. - # - filters: - # - type: ExtensionRef - # extensionRef: - # group: example.com - # kind: AuthPolicy - # name: app-auth - # Catch-all rule (no matches): everything not matched above lands here. - - backendRefs: - - name: app-service - port: 80 diff --git a/templates/ingressroute-tcp.yaml b/templates/ingressroute-tcp.yaml deleted file mode 100644 index 915d7e6..0000000 --- a/templates/ingressroute-tcp.yaml +++ /dev/null @@ -1,52 +0,0 @@ -# Exhaustive Traefik IngressRouteTCP reference (traefik.io/v1alpha1). -# Routes raw TCP: SSH, databases, or TLS-passthrough where Traefik never -# decrypts. Two TLS modes exist - termination (Traefik holds the cert) and -# passthrough (backend holds the cert) - and they are mutually exclusive. -apiVersion: traefik.io/v1alpha1 -kind: IngressRouteTCP -metadata: - name: app-ssh - namespace: example - labels: - app: app -spec: - entryPoints: - - ssh - routes: - # Plain TCP (SSH here): no TLS block at all, bytes flow as-is. - - match: HostSNI(`*`) - # HostSNI matches the TLS Server Name Indication. `*` accepts anything - # (required for non-TLS protocols like SSH that send no SNI). - # With TLS + a real hostname: HostSNI(`db.example.com`). - # middlewares: TCP middleware chain (IP allowlist, rate limit...). - # middlewares: - # - name: ssh-allowlist - # priority: same semantics as HTTP - higher wins. - # priority: 10 - services: - - name: app-service - port: 2222 - # weight: share of connections across backends. - # weight: 1 - # terminationDelay: linger after backend close to drain in-flight - # data. Default 100ms; raise for slow protocols. - terminationDelay: 100 - # proxyProtocol: PROXY header toward the backend (v1/v2) so it - # learns real client IPs. - # proxyProtocol: - # version: 2 - # TLS termination: Traefik decrypts with its own cert, forwards plaintext. - # - match: HostSNI(`db.example.com`) - # services: - # - name: app-service - # port: 5432 - # tls: enable TLS handling on this route. Omit entirely for plain TCP. - # tls: - # Either termination... - # secretName: app-tcp-tls - # options: - # name: modern-tls - # domains: - # - main: db.example.com - # ...or passthrough (Traefik never sees plaintext; needs SNI routing): - # passthrough: true diff --git a/templates/ingressroute-udp.yaml b/templates/ingressroute-udp.yaml deleted file mode 100644 index 4961ac6..0000000 --- a/templates/ingressroute-udp.yaml +++ /dev/null @@ -1,19 +0,0 @@ -# Exhaustive Traefik IngressRouteUDP reference (traefik.io/v1alpha1). -# Routes UDP datagrams (DNS, STUN/TURN, syslog...). No match rules exist - -# UDP has no hostname/SNI to route on, so one route per entrypoint simply -# forwards everything it receives. -apiVersion: traefik.io/v1alpha1 -kind: IngressRouteUDP -metadata: - name: app-stun - namespace: example - labels: - app: app -spec: - entryPoints: - - stun - services: - - name: app-service - port: 3478 - # weight: share of datagrams when several backends are listed. - weight: 1 diff --git a/templates/ingressroute.yaml b/templates/ingressroute.yaml deleted file mode 100644 index dd32fbb..0000000 --- a/templates/ingressroute.yaml +++ /dev/null @@ -1,93 +0,0 @@ -# Exhaustive Traefik IngressRoute reference (traefik.io/v1alpha1, HTTP). -# Routes are evaluated top to bottom by priority, then by rule length: the -# first matching route handles the request. Keep specific rules above the -# catch-all. -apiVersion: traefik.io/v1alpha1 -kind: IngressRoute -metadata: - name: app-prod - namespace: example - labels: - app: app - annotations: - description: "exhaustive traefik http route example" -spec: - # entryPoints: static entrypoints the route listens on (ports Traefik was - # started with: web=:80, websecure=:443, plus any custom ones). - entryPoints: - - websecure - routes: - # Rule 1: API subtree with middleware chain and weighted backends. - - match: Host(`app.example.com`) && PathPrefix(`/api`) - # kind: Rule (match traffic) or the same match used for redirections. - kind: Rule - # priority: explicit precedence. Higher wins regardless of position. - # Default is the rule length in characters - explicit numbers beat - # clever ordering. - priority: 100 - # middlewares: request pipeline, in order (auth, headers, rate limit, - # redirect, strip prefix...). Same-namespace by name; cross-namespace - # as name@namespace (never across providers without the suffix). - middlewares: - - name: app-auth - - name: security-headers - services: - - name: app-service - # port: Service port number or name. - port: 80 - # scheme: http (default), https (TLS backend), h2c (cleartext - # HTTP/2, e.g. gRPC without TLS). - scheme: http - # weight: traffic share for canary/blue-green splits. - weight: 90 - # serversTransport: ServersTransport CRD with TLS/forwarding - # tuning for this backend (rootCAs, insecureSkipVerify...). - # serversTransport: app-transport - # responseForwarding: - # flushInterval: 100ms - # passHostHeader: forward the original Host header (default true). - # passHostHeader: true - # proxyProtocol: speak PROXY protocol to the backend so it sees - # real client IPs. Version v1 or v2; backend must understand it. - # proxyProtocol: - # version: 2 - - name: app-canary-service - port: 80 - weight: 10 - # Rule 2: multiple hosts, regex path, external backend by URL. - - match: (Host(`app.example.com`) || Host(`www.example.com`)) && PathRegexp(`^/files/.*$`) - kind: Rule - priority: 50 - services: - # servers: bypass the Service and address backends directly. - # Only one of name/port (cluster Service) or servers (explicit - # URLs) may be set. - - name: app-service - port: 80 - # Catch-all rule: everything not matched above. - - match: Host(`app.example.com`) - kind: Rule - priority: 1 - services: - - name: app-service - port: 80 - # tls: terminate TLS on this route. Omit the whole block for plain HTTP. - tls: - # secretName: TLS Secret (tls.crt/tls.key) in THIS namespace. - secretName: app-prod-tls - # options: TLSOption CRD (minVersion, cipherSuites, sniStrict...). - # options: - # name: modern-tls - # certResolver: ACME resolver name (letsencrypt-style) that issues the - # certificate on demand. Use EITHER certResolver OR secretName, not both: - # resolver for auto-issued certs, secretName for pre-made ones. - # certResolver: letsencrypt - # store: custom TLSStore for the certificate. Default store otherwise. - # store: - # name: default - # domains: certificates to request/serve (main + SANs). With secretName - # this documents intent; with certResolver it drives issuance. - domains: - - main: app.example.com - sans: - - www.example.com diff --git a/templates/service.yaml b/templates/service.yaml deleted file mode 100644 index e5181f5..0000000 --- a/templates/service.yaml +++ /dev/null @@ -1,100 +0,0 @@ -# Exhaustive Service reference (v1). -# A Service is a stable virtual endpoint in front of pods: one DNS name and -# one cluster IP that load-balances to the currently Ready pods behind it. -# Pods come and go; the Service name (app-service.example.svc.cluster.local) -# never changes. -apiVersion: v1 -kind: Service -metadata: - name: app-service - namespace: example - labels: - app: app - annotations: - description: "exhaustive service example" -spec: - # selector: pods carrying these labels receive traffic. Empty selector = - # no automatic endpoints (pair with a hand-written EndpointSlice to aim at - # an external address - see endpointslice.yaml). - selector: - app: app - ports: - - name: http - # protocol: TCP (default), UDP, or SCTP. Each port entry needs one. - protocol: TCP - # port: the port clients connect to on the Service IP. - port: 80 - # targetPort: the port on the pod. Number or container port NAME - # (names decouple the Service from container port renumbering). - # Omit when it equals `port`. - targetPort: http - # nodePort: fixed node port for type NodePort/LoadBalancer (range - # 30000-32767). Omit for auto-assignment. - # nodePort: 30080 - # appProtocol: protocol hint for the payload (http, https, grpc, h2c, - # ws...). Used by meshes and LBs, ignored by plain kube-proxy routing. - appProtocol: http - - name: metrics - protocol: TCP - port: 9090 - targetPort: 9090 - # type: ClusterIP (default, internal VIP), NodePort (also open a fixed port - # on every node), LoadBalancer (NodePort + cloud LB in front), - # ExternalName (DNS alias, no proxying - see below). - type: ClusterIP - # clusterIP: pin the virtual IP. "None" makes the Service headless: no VIP, - # DNS returns pod IPs directly (required base for StatefulSets). - # clusterIP: None - # clusterIPs / ipFamilies / ipFamilyPolicy: dual-stack control. - # ipFamilies: [IPv4] (default), [IPv6], or [IPv4, IPv6]. - # ipFamilyPolicy: SingleStack (default), PreferDualStack, RequireDualStack. - # ipFamilies: - # - IPv4 - # ipFamilyPolicy: SingleStack - # sessionAffinity: None (default, spread every connection) or ClientIP - # (same client IP sticks to one pod). - sessionAffinity: None - # sessionAffinityConfig: stickiness TTL for ClientIP affinity. - # sessionAffinityConfig: - # clientIP: - # timeoutSeconds: 10800 - # publishNotReadyAddresses: send traffic to not-Ready pods too. Needed for - # peer discovery where members must find each other before anyone is Ready. - publishNotReadyAddresses: false - # internalTrafficPolicy: Cluster (default, route to pods on any node) or - # Local (only pods on the receiving node - preserves source IP, drops - # traffic on nodes without local pods). - internalTrafficPolicy: Cluster - # --- NodePort / LoadBalancer extras (ignored by pure ClusterIP) --- - # externalTrafficPolicy: Cluster (default) or Local. Local preserves the - # client source IP but drops traffic arriving on nodes with no local pod. - # externalTrafficPolicy: Cluster - # healthCheckNodePort: fixed node port for the LB health check (Local - # policy). Omit for auto-assignment. - # healthCheckNodePort: 32111 - # allocateLoadBalancerNodePorts: set false to skip node-port allocation on - # a LoadBalancer (when the LB routes straight to pods). Default true. - # allocateLoadBalancerNodePorts: true - # loadBalancerIP: request a specific IP from the cloud provider. Provider - # support varies; most now prefer annotations or IP pools. - # loadBalancerIP: 203.0.113.10 - # loadBalancerSourceRanges: client CIDRs allowed through the cloud LB. - # Unset = world-open. This is the cloud firewall in front of the Service. - # loadBalancerSourceRanges: - # - 203.0.113.0/24 - # loadBalancerClass: use a custom LB implementation instead of the cloud - # default (e.g. MetalLB speaker). The named controller must be installed. - # loadBalancerClass: example.com/custom-lb - # trafficDistribution: hint how to prefer endpoints (PreferClose = same - # zone first). Best-effort, kube-proxy dependent. - # trafficDistribution: PreferClose - # --- other types --- - # externalIPs: extra IPs (already routed to nodes) that also serve this - # Service. Traffic arriving there is proxied like ClusterIP traffic. - # externalIPs: - # - 203.0.113.20 - # ExternalName type: no proxying at all - DNS CNAME to the target. - # Ports are informational. Used to reference outside names under a stable - # in-cluster name. - # type: ExternalName - # externalName: db.example.com diff --git a/templates/statefulset.yaml b/templates/statefulset.yaml deleted file mode 100644 index b542726..0000000 --- a/templates/statefulset.yaml +++ /dev/null @@ -1,166 +0,0 @@ -# Exhaustive StatefulSet reference (apps/v1), shown with its headless Service. -# StatefulSets give each pod a stable name and stable storage: -# web-0, web-1, ... each reattached to its own volume after rescheduling. -# The classic use is databases and anything with an identity (postgres, -# redis, kafka). For the full container/pod field catalog see -# deployment.yaml - only StatefulSet-specific fields are expanded here. -apiVersion: v1 -kind: Service -metadata: - name: example-sts - namespace: example - labels: - app: example-sts -spec: - # clusterIP: None makes the Service headless: no virtual IP, DNS returns - # the pod IPs directly (web-0.example-sts...). Required for StatefulSets - - # it is how stable network identity works. - clusterIP: None - # publishNotReadyAddresses: include not-Ready pods in DNS. Needed for - # peer discovery when members must find each other before anyone is Ready - # (etcd, clustered databases). - publishNotReadyAddresses: false - selector: - app: example-sts - ports: - - name: db - port: 5432 - targetPort: db - # sessionAffinity: None (default, spread connections) or ClientIP (sticky - # sessions to one pod). Stateful apps sometimes want ClientIP. - sessionAffinity: None ---- -apiVersion: apps/v1 -kind: StatefulSet -metadata: - name: example-sts - namespace: example - labels: - app: example-sts -spec: - # serviceName: the headless Service above. Must exist; governs the pods' - # DNS domain. Changing it requires recreating the StatefulSet. - serviceName: example-sts - # replicas: pod count. Pods start in order 0..N-1 and stop in reverse. - replicas: 3 - # revisionHistoryLimit: old ControllerRevisions kept for rollback. - revisionHistoryLimit: 5 - # minReadySeconds: pod must stay Ready this long to count as available. - minReadySeconds: 10 - # podManagementPolicy: OrderedReady (default - strict 0,1,2 startup order, - # each predecessor must be Ready) or Parallel (start/stop all at once, - # faster for stateless-ish sets that only want stable names). - podManagementPolicy: OrderedReady - # persistentVolumeClaimRetentionPolicy: what happens to per-pod PVCs on - # scale-down (whenDeleted) and StatefulSet deletion (whenScaled). Retain - # keeps data (safe default); Delete wipes it. Set explicitly - the default - # Retain surprises people who expected cleanup. - persistentVolumeClaimRetentionPolicy: - whenDeleted: Retain - whenScaled: Retain - # ordinals: first ordinal (default 0). RARE - used when migrating an - # existing cluster whose numbering starts elsewhere. - # ordinals: - # start: 0 - selector: - matchLabels: - app: example-sts - updateStrategy: - # type: RollingUpdate (default) or OnDelete (new pods only replace old - # ones when you delete them manually - full control for databases). - type: RollingUpdate - rollingUpdate: - # partition: only ordinals >= partition are updated. Lets you canary: - # partition 2 updates web-2 first, then lower to 0 for the rest. - partition: 0 - # maxUnavailable: how many pods may be down during the update (1.25+). - # StatefulSets traditionally allowed exactly 1; now configurable. - maxUnavailable: 1 - template: - metadata: - labels: - app: example-sts - spec: - serviceAccountName: default - automountServiceAccountToken: true - terminationGracePeriodSeconds: 60 - # Databases want a long grace: 30s kills a checkpointing postmaster - # mid-write. 60-120 is typical for postgres. - containers: - - name: db - image: postgres:17 - imagePullPolicy: IfNotPresent - ports: - - name: db - containerPort: 5432 - env: - - name: POSTGRES_USER - value: example - - name: POSTGRES_DB - value: example - - name: POSTGRES_PASSWORD - valueFrom: - secretKeyRef: - name: example-secrets - key: db-password - # Probes for a database: pg_isready via exec is the standard. - # Budgets are generous - killing a recovering database only buys - # another full replay. - startupProbe: - exec: - command: ["sh", "-c", "pg_isready -U example -d example"] - # failureThreshold * periodSeconds = total startup budget - # (60 * 10s = 10 minutes here). - failureThreshold: 60 - periodSeconds: 10 - timeoutSeconds: 5 - readinessProbe: - exec: - command: ["sh", "-c", "pg_isready -U example -d example"] - periodSeconds: 10 - timeoutSeconds: 5 - livenessProbe: - exec: - command: ["sh", "-c", "pg_isready -U example -d example"] - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - resources: - requests: - cpu: "100m" - memory: "256Mi" - limits: - cpu: "1000m" - memory: "1Gi" - volumeMounts: - - name: data - mountPath: /var/lib/postgresql/data - # subPath: mounting the volume root directly breaks postgres - # (it wants an empty dir); subPath gives it a subdirectory. - subPath: pgdata - volumes: [] - # No shared volumes here: each pod gets its own PVC from the template - # below, mounted under the same `data` name. - volumeClaimTemplates: - # One template entry per volume. Pod web-N gets PVC data-web-N, - # created on first scheduling and (per the retention policy) kept after. - - metadata: - name: data - labels: - app: example-sts - annotations: - description: "per-pod database storage" - spec: - accessModes: - # Must be ReadWriteOnce: one pod owns its volume. (Shared RWX - # defeats the whole point of per-pod volumes.) - - ReadWriteOnce - storageClassName: standard - volumeMode: Filesystem - resources: - requests: - storage: 10Gi - # selector/dataSource/dataSourceRef: same semantics as a plain PVC - # (bind a specific PV, clone, restore from snapshot). See - # deployment.yaml. diff --git a/templates/tcproute.yaml b/templates/tcproute.yaml deleted file mode 100644 index d073b74..0000000 --- a/templates/tcproute.yaml +++ /dev/null @@ -1,35 +0,0 @@ -# Exhaustive TCPRoute reference (gateway.networking.k8s.io/v1alpha2). -# Routes raw TCP (SSH, databases, any non-HTTP protocol) from a TCP listener -# to backends. No hostname/path matching exists at this layer - there is -# nothing but the destination port to route on. For TLS with SNI-based -# routing see TLSRoute; for HTTP see HTTPRoute. -apiVersion: gateway.networking.k8s.io/v1alpha2 -kind: TCPRoute -metadata: - name: app-ssh - namespace: example - labels: - app: app -spec: - parentRefs: - # Bind to the TCP listener of the Gateway. Same fields as HTTPRoute - # parentRefs: name/kind/group plus sectionName and/or port narrowing. - - name: example - kind: Gateway - group: gateway.networking.k8s.io - sectionName: tcp - port: 2222 - rules: - - backendRefs: - - name: app-service - # port: backend Service port. Required. - port: 2222 - kind: Service - group: "" - # weight: share of connections when several backends are listed. - weight: 1 - # Second backend: every new connection goes 3:1 here. Weights are - # the only traffic-shaping knob TCP routing has. - - name: app-replica-service - port: 2222 - weight: 3 diff --git a/templates/udproute.yaml b/templates/udproute.yaml deleted file mode 100644 index 2da833f..0000000 --- a/templates/udproute.yaml +++ /dev/null @@ -1,29 +0,0 @@ -# Exhaustive UDPRoute reference (gateway.networking.k8s.io/v1alpha2). -# Routes raw UDP datagrams (DNS, STUN/TURN, game servers, QUIC-before-TLS) -# from a UDP listener to backends. Same minimal shape as TCPRoute: UDP has -# no sessions or headers to match on, so backendRefs carry the whole rule. -apiVersion: gateway.networking.k8s.io/v1alpha2 -kind: UDPRoute -metadata: - name: app-stun - namespace: example - labels: - app: app -spec: - parentRefs: - # Bind to the UDP listener of the Gateway. Same fields as HTTPRoute - # parentRefs: name/kind/group plus sectionName and/or port narrowing. - - name: example - kind: Gateway - group: gateway.networking.k8s.io - sectionName: udp - port: 3478 - rules: - - backendRefs: - - name: app-service - # port: backend Service port. Required. - port: 3478 - kind: Service - group: "" - # weight: share of traffic when several backends are listed. - weight: 1 From 6057734a4f0d27aad2f3f41a01faa0683e6e5aac Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 18:08:17 +0200 Subject: [PATCH 20/53] fix(renovate): sync generated configmap --- renovate/k8s/configmap.yaml | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/renovate/k8s/configmap.yaml b/renovate/k8s/configmap.yaml index b5a70a2..76baa8b 100644 --- a/renovate/k8s/configmap.yaml +++ b/renovate/k8s/configmap.yaml @@ -63,6 +63,15 @@ data: "depNameTemplate": "kube-prometheus-stack", "registryUrlTemplate": "https://prometheus-community.github.io/helm-charts" }, + { + "customType": "regex", + "description": "VictoriaMetrics Operator chart version pinned in the deploy workflow", + "managerFilePatterns": [".gitea/workflows/deploy-lib.sh"], + "matchStrings": ["\\|victoriametrics/victoria-metrics-operator\\|prometheus\\|(?[0-9.]+)\\|"], + "datasourceTemplate": "helm", + "depNameTemplate": "victoria-metrics-operator", + "registryUrlTemplate": "https://victoriametrics.github.io/helm-charts" + }, { "customType": "regex", "description": "grafana/loki chart version pinned in the deploy workflow", From 3f2b4e9acf9cf61db9abc203c1810b13435be088 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 18:18:25 +0200 Subject: [PATCH 21/53] fix(deploy): skip VMAgent preflight before CRD install --- .gitea/tests/deploy-validation.sh | 14 ++++++++++++++ .gitea/workflows/deploy-lib.sh | 18 ++++++++++++++++++ 2 files changed, 32 insertions(+) diff --git a/.gitea/tests/deploy-validation.sh b/.gitea/tests/deploy-validation.sh index 94e6480..0727dce 100755 --- a/.gitea/tests/deploy-validation.sh +++ b/.gitea/tests/deploy-validation.sh @@ -73,6 +73,20 @@ fi grep -q 'MISSING OR UNREADABLE: app/credentials' "$scratch/secrets.log" # API/rendering errors must not produce an empty reference list and pass. kubectl() { return 1; } +if ! skip_uninstalled_vmagent_crd "$REPO/prometheus-stack/k8s/vmagent.yaml"; then + echo 'VMAgent preflight did not skip an uninstalled CRD' >&2 + exit 1 +fi +kubectl() { return 0; } +if skip_uninstalled_vmagent_crd "$REPO/prometheus-stack/k8s/vmagent.yaml"; then + echo 'VMAgent preflight skipped an installed CRD' >&2 + exit 1 +fi +if skip_uninstalled_vmagent_crd "$REPO/prometheus-stack/k8s/victoria.yaml"; then + echo 'VMAgent preflight skipped an unrelated manifest' >&2 + exit 1 +fi +kubectl() { return 1; } if check_referenced_secrets >"$scratch/secrets.log"; then echo 'Secret check accepted a failed manifest render' >&2 exit 1 diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 9429adc..0f6ee53 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -731,6 +731,18 @@ check_referenced_secrets() { fi } +# The VMAgent CRD is installed by the VictoriaMetrics Operator Helm release in +# stage_apply_k8s, after this preflight stage. Skip only its dry-run until then. +skip_uninstalled_vmagent_crd() { + local manifest="$1" + if [[ "$manifest" == "$REPO/prometheus-stack/k8s/vmagent.yaml" ]] \ + && ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then + echo " skip: VMAgent CRD is installed by Helm during apply: ${manifest#"$REPO"/}" + return 0 + fi + return 1 +} + stage_validate() { check_prune_mode || return 1 cd "$REPO" @@ -748,6 +760,9 @@ stage_validate() { done log "Validate k8s manifests (kubectl dry-run=client)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do + if skip_uninstalled_vmagent_crd "$m"; then + continue + fi kubectl apply --dry-run=client -f "$m" >/dev/null done for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do @@ -755,6 +770,9 @@ stage_validate() { done log "Validate k8s manifests (kubectl dry-run=server)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do + if skip_uninstalled_vmagent_crd "$m"; then + continue + fi kubectl apply --dry-run=server -f "$m" >/dev/null done for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do From 1b70a5530008406aaefb8a26bd9503d752917a99 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 18:21:19 +0200 Subject: [PATCH 22/53] fix(ci): handle malformed push before SHA --- .gitea/workflows/ci.yaml | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 938696e..c7ed6cc 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -338,11 +338,20 @@ jobs: - name: Detect changed docker-built services id: services shell: bash + env: + PUSH_BEFORE: ${{ github.event.before }} run: | set -euo pipefail - base="${{ github.event.before }}" - if [ -z "$base" ] || [ "$base" = "0000000000000000000000000000000000000000" ]; then - base="$(git rev-list --max-parents=0 HEAD)" + base="${PUSH_BEFORE:-}" + empty_tree="$(git hash-object -t tree /dev/null)" + if [[ "$base" =~ ^0{40}$ ]]; then + base="$empty_tree" + elif [[ ! "$base" =~ ^[0-9a-fA-F]{40}$ ]] || ! git cat-file -e "${base}^{commit}" 2>/dev/null; then + # Some Gitea push payloads expose `before` as multiple root commits + # joined by newlines. It is not a usable diff base; use this push's + # first parent so image changes in the current commit are still built. + base="$(git rev-parse "${GITHUB_SHA}^" 2>/dev/null || printf '%s' "$empty_tree")" + echo "::warning::invalid push-before value; comparing against ${base}" fi # A failed diff used to leave changed_files empty, which reads exactly From cef499de732269d251ab61968c164b46d6ed78cd Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 19:19:47 +0200 Subject: [PATCH 23/53] fix(deploy): skip VMAgent in secrets check before CRD install check_referenced_secrets ran kubectl create on vmagent.yaml even when the VMAgent CRD is not installed yet, failing validate with 'no matches for kind VMAgent'. Apply the same skip_uninstalled_vmagent_crd guard used by both dry-run loops. --- .gitea/workflows/deploy-lib.sh | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 0f6ee53..b25c4cd 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -705,6 +705,9 @@ check_referenced_secrets() { local missing=() refs="" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do + if skip_uninstalled_vmagent_crd "$m"; then + continue + fi objects="$(kubectl create --dry-run=client --validate=false -f "$m" -o json)" || return 1 extracted="$(printf '%s' "$objects" | jq -r -f "$REPO/.gitea/workflows/secret-references.jq")" || return 1 refs+="$extracted"$'\n' From 7fdeacffb89e48de3739ddc99ae280a58890eb0c Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 19:29:30 +0200 Subject: [PATCH 24/53] feat(monitoring): stand down Prometheus server during VM trial vmagent scrapes and remote-writes to VictoriaMetrics, so the Prometheus server scales to 0. Encoded as prometheusSpec.replicas in values instead of a kubectl patch, so helm keeps owning spec.replicas and Helm 4 server-side apply stops conflicting with the kubectl-patch field manager. --- prometheus-stack/k8s/grafana-values.yaml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/prometheus-stack/k8s/grafana-values.yaml b/prometheus-stack/k8s/grafana-values.yaml index 2a2f436..56acb32 100644 --- a/prometheus-stack/k8s/grafana-values.yaml +++ b/prometheus-stack/k8s/grafana-values.yaml @@ -60,6 +60,10 @@ grafana: prometheus: prometheusSpec: + # VM trial: vmagent scrapes and remote-writes to VictoriaMetrics, so the + # Prometheus server itself stands down. Encoded here (not a kubectl patch) + # so helm keeps owning spec.replicas and upgrades do not conflict on it. + replicas: 0 retention: 60d retentionSize: 32GB storageSpec: From 9be8fb6c699b26c064cc5df5c3e23c2e3f999ec7 Mon Sep 17 00:00:00 2001 From: renovate-bot Date: Tue, 6 Oct 2026 16:19:34 +0000 Subject: [PATCH 25/53] chore(deps): update lscr.io/linuxserver/qbittorrent docker tag to v20 --- streaming/compose.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/streaming/compose.yaml b/streaming/compose.yaml index 87724d2..9c009f3 100644 --- a/streaming/compose.yaml +++ b/streaming/compose.yaml @@ -19,7 +19,7 @@ services: - streaming qbittorrent: - image: lscr.io/linuxserver/qbittorrent:5.2.4 + image: lscr.io/linuxserver/qbittorrent:20.04.1 container_name: qbittorrent restart: unless-stopped environment: From 2ac2a94bb4594cee4de51431b09059a3215265f9 Mon Sep 17 00:00:00 2001 From: renovate-bot Date: Tue, 6 Oct 2026 16:19:22 +0000 Subject: [PATCH 26/53] chore(deps): update renovate/renovate docker tag to v44.140.0 --- renovate/k8s/cronjob.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/renovate/k8s/cronjob.yaml b/renovate/k8s/cronjob.yaml index 2992e0d..0bd9ff8 100644 --- a/renovate/k8s/cronjob.yaml +++ b/renovate/k8s/cronjob.yaml @@ -19,7 +19,7 @@ spec: restartPolicy: Never containers: - name: renovate - image: renovate/renovate:44.139.0 + image: renovate/renovate:44.140.0 env: - name: RENOVATE_PLATFORM value: gitea From 5f9354b9a8e7fda819f46045c6d993d9dc461ba5 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 22:55:38 +0200 Subject: [PATCH 27/53] fix(edu): deploy reviewed application release with Redis authentication --- .gitea/workflows/ci.yaml | 34 +++---------------- edu_master/README.md | 14 ++++++++ edu_master/k8s/alerts.yaml | 43 ++++++++++++++++++------- edu_master/k8s/redis-networkpolicy.yaml | 22 +++++++++++++ edu_master/k8s/redis.yaml | 21 ++++++++++++ edu_master/k8s/session-keeper.yaml | 16 ++++++++- edu_master/k8s/webinar-checker.yaml | 26 ++++++++++----- 7 files changed, 126 insertions(+), 50 deletions(-) create mode 100644 edu_master/README.md create mode 100644 edu_master/k8s/redis-networkpolicy.yaml diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index c7ed6cc..70bde19 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -387,9 +387,6 @@ jobs: homepages/*) add_service homepages ;; - edu_master/phpsessid-bot/*|edu_master/webinar-checker/*|edu_master/compose.yaml) - add_service edu_master - ;; esac done @@ -496,32 +493,6 @@ jobs: done done ;; - edu_master) - for variant in session-keeper webinar-checker; do - case "$variant" in - session-keeper) - context="edu_master/phpsessid-bot" - image="${REGISTRY}/forust/session-keeper" - ;; - webinar-checker) - context="edu_master/webinar-checker" - image="${REGISTRY}/forust/webinar-checker" - ;; - esac - set_tags - build_args=() - for tag in "${tags[@]}"; do - build_args+=(-t "${image}:${tag}") - done - docker build \ - --cache-from "type=registry,ref=${image}:buildcache" \ - --cache-to "type=registry,ref=${image}:buildcache,mode=max" \ - "${build_args[@]}" "$context" - for tag in "${tags[@]}"; do - docker push "${image}:${tag}" - done - done - ;; esac done @@ -552,6 +523,11 @@ jobs: fi echo "pinning ${#repos[@]} image(s) to $commit_tag" for repo in "${repos[@]}"; do + # EDU images are released by the application repository and pinned + # directly by digest in edu_master manifests. Never retag them here. + case "$repo" in + */session-keeper|*/webinar-checker) continue ;; + esac if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then echo " already built by this push: ${repo##*/}" continue diff --git a/edu_master/README.md b/edu_master/README.md new file mode 100644 index 0000000..093d47c --- /dev/null +++ b/edu_master/README.md @@ -0,0 +1,14 @@ +# EDU deployment ownership + +Application source and release builds: `forust/edu-master`. +The homelab pipeline deploys `edu_master/k8s` and preserves explicit image digests. +The application copies in this directory are legacy and are not build inputs. +Do not publish EDU `prod` images from homelab or resolve releases from moving tags. + +For an EDU release, validate both images, select their digests in the keeper and +checker manifests, and run the existing homelab validation/apply/verification +helpers against this service. Keep the existing Secret and Redis PVC. +Coordinate Redis authentication changes with both clients and all init/probes; +keep a pre-rollout Redis backup and both previous compatible image references. +The current HTTP checker does not depend on Playwright; check other consumers +before removing the separate browser service. diff --git a/edu_master/k8s/alerts.yaml b/edu_master/k8s/alerts.yaml index dce9760..82c62dc 100644 --- a/edu_master/k8s/alerts.yaml +++ b/edu_master/k8s/alerts.yaml @@ -50,7 +50,36 @@ spec: severity: critical annotations: summary: "Webinar checker failing consecutively" - description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).' + description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / http error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).' + + - alert: WebinarCheckerNeverStarted + expr: | + (time() - edu_process_start > 120) + and (webinar_check_last_run_timestamp_seconds == 0) + for: 2m + labels: + severity: critical + annotations: + summary: "Webinar checker job has not started" + description: "The process exposes metrics but its webinar job has never started." + + - alert: WebinarDeliveryPending + expr: edu_delivery_pending > 0 + for: 5m + labels: + severity: warning + annotations: + summary: "Webinar notifications await delivery" + description: "Telegram delivery has pending recipients. Check delivery failures and retry status." + + - alert: EduRedisUnavailable + expr: edu_redis_connected == 0 + for: 2m + labels: + severity: critical + annotations: + summary: "EDU checker cannot reach Redis" + description: "Redis health checks are failing; checker commands and delivery may be unavailable." # Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken. - alert: WebinarCheckerScrapeDown @@ -74,7 +103,7 @@ spec: summary: "EDU_PHPSESSID missing" description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials." - # Hard deps: checker and playwright deployments unavailable. + # Hard deps: checker deployment unavailable. - alert: WebinarCheckerDeploymentDown expr: | kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0 @@ -84,13 +113,3 @@ spec: annotations: summary: "Webinar checker deployment unavailable" description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m." - - - alert: PlaywrightServiceDown - expr: | - kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0 - for: 10m - labels: - severity: critical - annotations: - summary: "Playwright service unavailable" - description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it." diff --git a/edu_master/k8s/redis-networkpolicy.yaml b/edu_master/k8s/redis-networkpolicy.yaml new file mode 100644 index 0000000..e7e4ec1 --- /dev/null +++ b/edu_master/k8s/redis-networkpolicy.yaml @@ -0,0 +1,22 @@ +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: redis-clients-only + namespace: edu-master +spec: + podSelector: + matchLabels: + app: edu-master-redis + policyTypes: + - Ingress + ingress: + - from: + - podSelector: + matchLabels: + app: edu-master-session-keeper + - podSelector: + matchLabels: + app: edu-master-webinar-checker + ports: + - protocol: TCP + port: 6379 diff --git a/edu_master/k8s/redis.yaml b/edu_master/k8s/redis.yaml index e0fca76..e4f139c 100644 --- a/edu_master/k8s/redis.yaml +++ b/edu_master/k8s/redis.yaml @@ -20,6 +20,27 @@ spec: - name: redis image: redis:8.10.2-alpine imagePullPolicy: IfNotPresent + env: + - name: REDIS_PASSWORD + valueFrom: + secretKeyRef: + name: edu-master-secrets + key: REDIS_PASSWORD + - name: REDISCLI_AUTH + valueFrom: + secretKeyRef: + name: edu-master-secrets + key: REDIS_PASSWORD + command: + - /bin/sh + - -ec + - | + case "$REDIS_PASSWORD" in *[!0-9a-fA-F]*|'') echo 'REDIS_PASSWORD must be 64 hex characters' >&2; exit 1;; esac + [ "${#REDIS_PASSWORD}" -eq 64 ] || { echo 'REDIS_PASSWORD must be 64 hex characters' >&2; exit 1; } + umask 077 + printf 'requirepass "%s"\n' "$REDIS_PASSWORD" > /tmp/redis-auth.conf + chown redis:redis /tmp/redis-auth.conf + exec docker-entrypoint.sh redis-server /tmp/redis-auth.conf ports: - containerPort: 6379 volumeMounts: diff --git a/edu_master/k8s/session-keeper.yaml b/edu_master/k8s/session-keeper.yaml index f3c330d..65bfd97 100644 --- a/edu_master/k8s/session-keeper.yaml +++ b/edu_master/k8s/session-keeper.yaml @@ -16,12 +16,20 @@ spec: type: Recreate template: metadata: + annotations: + edu.forust.xyz/source-commit: "90829d6c8080b9928f9da23587678e640939e10a" labels: app: edu-master-session-keeper spec: initContainers: - name: wait-redis image: redis:8.10.2-alpine + env: + - name: REDISCLI_AUTH + valueFrom: + secretKeyRef: + name: edu-master-secrets + key: REDIS_PASSWORD command: - /bin/sh - -ec @@ -35,10 +43,16 @@ spec: echo "redis is ready" containers: - name: session-keeper - image: gcr.forust.xyz/forust/session-keeper:prod + image: gcr.forust.xyz/forust/session-keeper@sha256:49285e87cc5bc4cf4ffe190813d87927916c2df8a206daac0aeb7d227c636450 envFrom: - secretRef: name: edu-master-secrets + env: + - name: REDISCLI_AUTH + valueFrom: + secretKeyRef: + name: edu-master-secrets + key: REDIS_PASSWORD resources: requests: cpu: 25m diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml index 3c53074..6280de1 100644 --- a/edu_master/k8s/webinar-checker.yaml +++ b/edu_master/k8s/webinar-checker.yaml @@ -16,14 +16,22 @@ spec: type: Recreate template: metadata: + annotations: + edu.forust.xyz/source-commit: "90829d6c8080b9928f9da23587678e640939e10a" labels: app: edu-master-webinar-checker spec: # Enforces dependency order like compose depends_on: - # redis healthy -> session-keeper healthy (EXISTS EDU_PHPSESSID) -> playwright started + # redis healthy -> session-keeper healthy (EXISTS EDU_PHPSESSID) initContainers: - name: wait-deps image: redis:8.10.2-alpine + env: + - name: REDISCLI_AUTH + valueFrom: + secretKeyRef: + name: edu-master-secrets + key: REDIS_PASSWORD command: - /bin/sh - -ec @@ -41,15 +49,9 @@ spec: sleep 2 done echo "PHPSESSID ok" - until nc -z playwright-service 3000; do - i=$((i+1)) - [ "$i" -ge 300 ] && echo "TIMEOUT: playwright-service not reachable" && exit 1 - sleep 2 - done - echo "playwright ok" containers: - name: webinar-checker - image: gcr.forust.xyz/forust/webinar-checker:prod + image: gcr.forust.xyz/forust/webinar-checker@sha256:66c146f7b43cb9f0dc31ba9aa36d217e01df42ddafba5971b79c12ec215b2c01 ports: - name: metrics containerPort: 8000 @@ -62,6 +64,14 @@ spec: timeoutSeconds: 3 failureThreshold: 12 initialDelaySeconds: 10 + livenessProbe: + httpGet: + path: /live + port: metrics + initialDelaySeconds: 60 + periodSeconds: 15 + timeoutSeconds: 3 + failureThreshold: 4 envFrom: - secretRef: name: edu-master-secrets From 9a76529be80617c15ab5e73e00517f16066caec3 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 23:06:28 +0200 Subject: [PATCH 28/53] refactor(ci): use native runner and durable incremental deploys --- .gitea/deploy-dependencies.json | 3 + .gitea/runner/README.md | 122 +++++ .gitea/runner/buildkitd.toml | 11 + .gitea/runner/config.yaml | 10 + .gitea/runner/gitea-runner.service | 18 + .gitea/runner/homelab-deploy@.service | 12 + .gitea/runner/setup-runner.sh | 48 ++ .gitea/runner/setup-workstation.sh | 32 ++ .gitea/workflows/ci.yaml | 429 ++---------------- .gitea/workflows/compose-release.py | 93 ++++ .gitea/workflows/deploy-controller.py | 323 ++++++++++++++ .gitea/workflows/deploy-lib.sh | 619 +++++++++----------------- .gitea/workflows/deploy-plan.py | 124 ++++++ .gitea/workflows/deploy-stage.sh | 10 + .gitea/workflows/deploy.yaml | 232 +++------- .gitea/workflows/install-ci-tools.sh | 81 ++-- .gitea/workflows/release.py | 350 +++++++++++++++ .gitea/workflows/renovate-ci.yaml | 2 +- .gitea/workflows/renovate-run.yaml | 2 +- .gitea/workflows/ssh-run.sh | 111 ++--- .gitea/workflows/tool-versions.env | 3 + renovate/k8s/configmap.yaml | 7 + renovate/renovate.json | 7 + tests/test_cicd.py | 267 +++++++++++ tests/test_cicd_lifecycle.py | 237 ++++++++++ 25 files changed, 2090 insertions(+), 1063 deletions(-) create mode 100644 .gitea/deploy-dependencies.json create mode 100644 .gitea/runner/README.md create mode 100644 .gitea/runner/buildkitd.toml create mode 100644 .gitea/runner/config.yaml create mode 100644 .gitea/runner/gitea-runner.service create mode 100644 .gitea/runner/homelab-deploy@.service create mode 100755 .gitea/runner/setup-runner.sh create mode 100755 .gitea/runner/setup-workstation.sh create mode 100644 .gitea/workflows/compose-release.py create mode 100644 .gitea/workflows/deploy-controller.py create mode 100644 .gitea/workflows/deploy-plan.py create mode 100755 .gitea/workflows/deploy-stage.sh create mode 100644 .gitea/workflows/release.py create mode 100644 tests/test_cicd.py create mode 100644 tests/test_cicd_lifecycle.py diff --git a/.gitea/deploy-dependencies.json b/.gitea/deploy-dependencies.json new file mode 100644 index 0000000..7c5f3f8 --- /dev/null +++ b/.gitea/deploy-dependencies.json @@ -0,0 +1,3 @@ +{ + "postgres": ["authentik", "gitea", "immich", "n8n", "netbox", "netronome"] +} diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md new file mode 100644 index 0000000..803d303 --- /dev/null +++ b/.gitea/runner/README.md @@ -0,0 +1,122 @@ +# Homelab CI/CD + +The native Gitea runner runs on **vps**; production runs on **workstation**. +Jobs run on `homelab:host`, one at a time. No job images or Kubernetes credentials +are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows. + +## Runner installation + +Install Docker Engine with Compose and Buildx, Git, Python 3.11+, Bash, curl, +GNU tar/xz, flock and systemd using the host's package manager. Keep the existing +Gitea runner 3.0.2 binary at `/usr/local/bin/gitea-runner`. + +From this checkout on the VPS: + +```sh +sudo bash .gitea/runner/setup-runner.sh +``` + +The installer reuses `/var/lib/gitea-runner/.runner` and the existing service. +For a new host, install the same runner binary and register as `gitea-runner` +using the registration token interactively, label `homelab:host`, and working +directory `/var/lib/gitea-runner`; then rerun the installer. Tokens never belong +in this repository or command-line examples. + +Pinned tools live in the runner user's `~/.cache/homelab-ci`; CI repairs version +drift there. Installations are locked. Buildx uses only the `homelab-ci` builder, +pushes directly to the registry, and caps retained local cache at 1 GiB with a +2 GiB free-space target. This is not a hard limit on peak build disk usage. +Nothing runs `docker system prune`, removes unrelated images, or deletes volumes. + +## Workstation setup + +As the existing SSH deploy user on workstation: + +```sh +sudo loginctl enable-linger forust +bash .gitea/runner/setup-workstation.sh +``` + +The controller uses `/srv/homelab` as the persistent configuration tree and makes +a detached source worktree for each SHA. It never resets `/srv/homelab`, moves +local configuration, renames Compose projects, or changes volume names. +The installer records the current Kubernetes context and cluster UID in +`~/.config/homelab-deploy/environment`. Check these before installing. + +Configure Gitea Actions Variables: + +- `DEPLOY_HOST`, `DEPLOY_USER`, `DEPLOY_PORT`: the existing VPS-to-workstation SSH endpoint. +- `DEPLOY_KNOWN_HOSTS`: workstation's verified SSH host key entry for that endpoint. +- `AUTODEPLOY`: `false` initially; `true` enables deployment after successful main CI. + +Keep `DEPLOY_SSH_KEY`, `REGISTRY_USERNAME` and `REGISTRY_PASSWORD` in Actions +Secrets. Legacy endpoint secrets remain accepted during migration. The Actions +token must have repository read and Actions read access for release downloads. +The deploy user's existing Docker registry authentication remains necessary. + +## Releases and deployment + +CI publishes `release-` as a Gitea artifact with all three owned image +digests and build input fingerprints. Unchanged images are reused only from a +successful main CI artifact, never from `:prod`. EDU images remain pinned to the +digests released by their application repository. Expired artifacts cause CI to +rebuild images; they block deployment until CI is rerun. + +Run deploy from main with `deploy_ref=main` or a checked SHA: + +- `full`: required for the first baseline; reconcile all active components. +- `changed`: compare with the last fully successful production deploy. +- `plan`: validate configuration and show selection without changing production resources. +- `refresh_images=true`: explicitly refresh mutable third-party Compose tags. + +The manual and automatic paths both require successful CI, a successful build +job and the exact SHA's release artifact. PRs cannot publish images or deploy. +Removed resources are reported and require explicit removal; no automatic prune. +Service dependencies are listed in `.gitea/deploy-dependencies.json`. + +A workstation user systemd service holds the deploy lock across validation, +sequential apply, verification and smoke checks. SSH clients only submit/follow: +disconnecting or cancelling the Actions client does not kill production apply. +Retrying the same run ID does not start another apply. `ExecStopPost` recovers +interrupted runs before the unit finishes. Kubernetes rolls back to captured +revisions; configuration and persistent data are not reverted. + +## Status and recovery + +`--retry` repeats failed verification and smoke checks, never apply. Recovery +keeps a failed deploy out of the successful baseline, even after rollback. + +On workstation (replace the numeric ID with Actions run ID and attempt): + +```sh +python3 ~/.local/lib/homelab-deploy/controller.py status 123-1 +python3 ~/.local/lib/homelab-deploy/controller.py recover 123-1 --retry +journalctl --user -u homelab-deploy@123-1 +``` + +Runs live in `~/.local/state/homelab-deploy/runs`. Compose stores resolved configs +with restricted permissions; these may contain credentials and must never be +uploaded as CI artifacts. Stage logs print the exact manual recovery command +using `compose-before/.json`, the original project directory and project +name. Compose does not automatically roll back, and Nextcloud AIO's child +containers remain managed by AIO. Preserve its own backups for data recovery. + +The controller retains twenty successful/planned runs and preserves failures. +Update the workstation dispatcher only when no deploy is running. + +## Validation and migration rollback + +```sh +python3 -m unittest discover -s tests -v +bash .gitea/tests/deploy-validation.sh +``` + +Test on a separate namespace before the initial production `full` run. Check a +failed rollout, interrupted SSH and repeated run ID, and verify that an isolated +service change does not upgrade unrelated Helm releases or Compose stacks. + +To roll back the migration, disable autodeploy and finish or recover the remote +run first. Restore the runner config/unit from `.before-` backups, +reload systemd and restart the runner. Restore the prior workflows from Git. +Production data and persistent volumes stay where they were. Do not remove run +state or Compose recovery files until recovery is confirmed. diff --git a/.gitea/runner/buildkitd.toml b/.gitea/runner/buildkitd.toml new file mode 100644 index 0000000..9b23273 --- /dev/null +++ b/.gitea/runner/buildkitd.toml @@ -0,0 +1,11 @@ +[worker.oci] + gc = true + reservedSpace = "256MB" + maxUsedSpace = "1GB" + minFreeSpace = "2GB" + +[[worker.oci.gcpolicy]] + reservedSpace = "256MB" + maxUsedSpace = "1GB" + minFreeSpace = "2GB" + all = true diff --git a/.gitea/runner/config.yaml b/.gitea/runner/config.yaml new file mode 100644 index 0000000..77028e5 --- /dev/null +++ b/.gitea/runner/config.yaml @@ -0,0 +1,10 @@ +runner: + file: /var/lib/gitea-runner/.runner + capacity: 1 + timeout: 5h + labels: + - homelab:host +cache: + enabled: false +container: + docker_host: unix:///var/run/docker.sock diff --git a/.gitea/runner/gitea-runner.service b/.gitea/runner/gitea-runner.service new file mode 100644 index 0000000..8665bfc --- /dev/null +++ b/.gitea/runner/gitea-runner.service @@ -0,0 +1,18 @@ +[Unit] +Description=Gitea Actions runner +After=network-online.target docker.service +Wants=network-online.target + +[Service] +User=gitea-runner +Group=gitea-runner +SupplementaryGroups=docker +WorkingDirectory=/var/lib/gitea-runner +Environment=PATH=/var/lib/gitea-runner/.cache/homelab-ci/bin:/usr/local/bin:/usr/bin:/bin +ExecStart=/usr/local/bin/gitea-runner daemon --config /etc/gitea-runner/config.yaml +Restart=on-failure +RestartSec=5 +UMask=0077 + +[Install] +WantedBy=multi-user.target diff --git a/.gitea/runner/homelab-deploy@.service b/.gitea/runner/homelab-deploy@.service new file mode 100644 index 0000000..1e9ceab --- /dev/null +++ b/.gitea/runner/homelab-deploy@.service @@ -0,0 +1,12 @@ +[Unit] +Description=Homelab deploy %i + +[Service] +Type=exec +EnvironmentFile=%h/.config/homelab-deploy/environment +ExecStart=/usr/bin/python3 %h/.local/lib/homelab-deploy/controller.py execute %i +ExecStopPost=/usr/bin/python3 %h/.local/lib/homelab-deploy/controller.py recover %i +RuntimeMaxSec=5h +TimeoutStopSec=135min +KillMode=control-group +UMask=0077 diff --git a/.gitea/runner/setup-runner.sh b/.gitea/runner/setup-runner.sh new file mode 100755 index 0000000..bacd7e1 --- /dev/null +++ b/.gitea/runner/setup-runner.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +# Native host runner, with pinned user-space tools and no extra CI images. +set -euo pipefail +here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +[ "$(id -u)" -eq 0 ] || { echo 'Run with sudo on the runner host' >&2; exit 1; } +for tool in docker curl python3 git tar xz flock runuser systemctl; do + command -v "$tool" >/dev/null || { echo "Install missing prerequisite: $tool" >&2; exit 1; } +done +docker info >/dev/null +docker compose version >/dev/null +docker buildx version >/dev/null +id gitea-runner >/dev/null 2>&1 || useradd --system --create-home --home-dir /var/lib/gitea-runner --shell /usr/bin/bash gitea-runner +# Reuse the established service account and runner registration. +runner_home="$(getent passwd gitea-runner | cut -d: -f6)" +[ "$runner_home" = /var/lib/gitea-runner ] || { echo 'Unexpected runner home; inspect the existing service first' >&2; exit 1; } +runuser -u gitea-runner -- docker info >/dev/null || { echo "The runner user needs access to Docker before setup" >&2; exit 1; } +command -v gitea-runner >/dev/null || { echo 'Install gitea-runner 3.0.2 at /usr/local/bin/gitea-runner first' >&2; exit 1; } +mkdir -p /etc/gitea-runner +stamp="$(date -u +%Y%m%dT%H%M%SZ)" +for existing in /etc/gitea-runner/config.yaml /etc/systemd/system/gitea-runner.service; do + [ ! -f "$existing" ] || cp -p "$existing" "$existing.before-$stamp" +done +scratch="$(mktemp -d)" +trap 'rm -rf "$scratch"' EXIT +chmod 755 "$scratch" +install -m 0644 "$here/../workflows/install-ci-tools.sh" "$here/../workflows/tool-versions.env" "$scratch/" +runuser -u gitea-runner -- bash "$scratch/install-ci-tools.sh" +install -m 0644 "$here/config.yaml" /etc/gitea-runner/config.yaml +python3 - <<'PYLABELS' +import json +from pathlib import Path +registration = Path('/var/lib/gitea-runner/.runner') +if registration.exists(): + labels = json.loads(registration.read_text()).get('labels', []) + labels = [label for label in labels if isinstance(label, str) and label.split(':')[0] != 'homelab'] + labels.append('homelab:host') + config = Path('/etc/gitea-runner/config.yaml') + config.write_text(config.read_text().replace(' - homelab:host', '\n'.join(' - ' + json.dumps(label) for label in labels))) +PYLABELS +install -m 0644 "$here/gitea-runner.service" /etc/systemd/system/gitea-runner.service +if [ ! -f /var/lib/gitea-runner/.runner ]; then + echo 'Register once as gitea-runner with homelab:host before starting the service.' + exit 0 +fi +systemctl daemon-reload +systemctl enable --now gitea-runner.service +systemctl restart gitea-runner.service +echo "Runner ready. Configuration backups: *.before-$stamp" diff --git a/.gitea/runner/setup-workstation.sh b/.gitea/runner/setup-workstation.sh new file mode 100755 index 0000000..a6d5f9b --- /dev/null +++ b/.gitea/runner/setup-workstation.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# Run as the existing deploy user on workstation. Never resets the working tree. +set -euo pipefail +here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +repo="${HOMELAB_REPO:-/srv/homelab}" +for tool in python3 git kubectl helm docker flock timeout; do + command -v "$tool" >/dev/null || { echo "Install missing dependency: $tool" >&2; exit 1; } +done +[ -d "$repo/.git" ] || { echo "Missing deploy checkout: $repo" >&2; exit 1; } +[[ "$repo" =~ ^/[A-Za-z0-9_./-]+$ ]] || { echo 'Deploy path must be absolute and contain no whitespace' >&2; exit 1; } +if [ "$(loginctl show-user "$USER" -p Linger --value)" != yes ]; then + echo "Run once: sudo loginctl enable-linger $USER" >&2 + exit 1 +fi +config="${XDG_CONFIG_HOME:-$HOME/.config}/homelab-deploy" +mkdir -p "$config" "$HOME/.local/lib/homelab-deploy" "$HOME/.config/systemd/user" +chmod 700 "$config" +if [ ! -f "$config/environment" ]; then + context="$(kubectl config current-context)" + cluster_uid="$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" + printf 'HOMELAB_REPO=%s\nKUBE_CONTEXT=%s\nEXPECTED_CLUSTER_UID=%s\n' "$repo" "$context" "$cluster_uid" >"$config/environment" + chmod 600 "$config/environment" +fi +# Do not replace a dispatcher while an existing deploy uses it. +if systemctl --user list-units 'homelab-deploy@*' --state=running --no-legend | grep -q .; then + echo 'An existing deploy is running; wait before updating the controller' >&2 + exit 1 +fi +install -m 0755 "$here/../workflows/deploy-controller.py" "$HOME/.local/lib/homelab-deploy/controller.py" +install -m 0644 "$here/homelab-deploy@.service" "$HOME/.config/systemd/user/homelab-deploy@.service" +systemctl --user daemon-reload +echo 'Controller ready. Run a checked main SHA in full mode for the initial baseline.' diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 70bde19..6bb9910 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -1,38 +1,29 @@ name: ci - -on: +"on": push: branches: - main - pull_request: - workflow_dispatch: - -# Every job here is checkout plus local tools. The token needs to read the tree -# and nothing else, and saying so keeps a future step that reaches for the API -# from quietly holding a token that can write to the repository. + pull_request: null + workflow_dispatch: null permissions: contents: read - + actions: read concurrency: group: ci-${{ github.ref }} cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} - -env: - REGISTRY: gcr.forust.xyz - jobs: - lint-compose: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 + checks: + runs-on: homelab + timeout-minutes: 30 steps: - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - # Structure check for every committed Compose file, active or not. - # Interpolation, env-file and bind-mount resolution are all switched off, - # because inactive stacks have no .env here and would only fail on their - # ${VAR:?} guards. Active stacks get the full check with interpolation in - # the deploy workflow, where the real .env files live. + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Validate Compose files shell: bash run: | @@ -61,35 +52,15 @@ jobs: exit 1 fi echo "checked ${#files[@]} Compose file(s)" - - lint-actionlint: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint Gitea Actions workflows with actionlint shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint)" - export PATH="$tools_dir:$PATH" actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml - - lint-shellcheck: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint shell scripts with ShellCheck shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" - export PATH="$tools_dir:$PATH" mapfile -t scripts < <( git ls-files '*.sh' ':(glob)**/*.bash' ) @@ -99,20 +70,10 @@ jobs: fi shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" bash .gitea/tests/deploy-validation.sh - - lint-prettier: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Check formatting with Prettier shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)" - export PATH="$tools_dir:$PATH" mapfile -t prettier_files < <( git ls-files \ @@ -126,37 +87,17 @@ jobs: fi prettier --check --ignore-unknown "${prettier_files[@]}" - - lint-ruff: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint and format-check Python with Ruff shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff)" - export PATH="$tools_dir:$PATH" - ruff check . - ruff format --check . + ruff check . .gitea/workflows + ruff format --check . .gitea/workflows python3 -m unittest discover -s tests -v - - lint-yaml: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint YAML syntax shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)" - export PATH="$tools_dir:$PATH" mapfile -t yaml_files < <( git ls-files '*.yaml' '*.yml' \ @@ -170,20 +111,10 @@ jobs: fi yamllint -c .yamllint "${yaml_files[@]}" - - lint-dockerfiles: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 10 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Lint Dockerfiles shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)" - export PATH="$tools_dir:$PATH" mapfile -t dockerfiles < <( git ls-files ':(glob)**/Dockerfile' ':(glob)**/Dockerfile.*' @@ -195,20 +126,10 @@ jobs: fi hadolint -c .hadolint.yaml "${dockerfiles[@]}" - - validate: - runs-on: [self-hosted, linux, arch, homelab] - timeout-minutes: 20 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - name: Validate Kubernetes manifests against JSON schemas shell: bash run: | set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)" - export PATH="$tools_dir:$PATH" mapfile -t manifests < <( git ls-files ':(glob)**/k8s/**/*.yaml' ':(glob)**/k8s/**/*.yml' \ @@ -225,317 +146,27 @@ jobs: -ignore-missing-schemas \ -summary \ "${manifests[@]}" - - # kubeconform has no schemas for CRDs, so every IngressRoute, Certificate, - # PrometheusRule, Middleware, ServersTransport and ServiceMonitor is silently - # skipped above. The live API server knows the real CRD schemas (and runs the - # cert-manager / Traefik admission webhooks), so validate there too. - # - # Only services marked with a k8s/active marker are checked: server-side - # dry-run needs the target namespace to exist, and inactive services are not - # deployed. Services being enabled for the first time are still covered by - # the JSON-schema pass above. - # - # Main pushes only. `--dry-run=server` persists nothing, but it does execute - # the admission webhooks of the production API server, so anyone able to open - # a pull request would be able to run arbitrary manifest content through - # cert-manager and Traefik. A pull request has nothing to gain from it either: - # only main is ever deployed, and this job runs to completion before the - # deploy workflow is allowed to start, so a bad CRD is still caught before - # anything reaches the cluster -- just on the push rather than on the PR. - - name: Note the server-side check is not running here - if: github.event_name == 'pull_request' || github.ref != 'refs/heads/main' - shell: bash - run: | - echo "::notice::Skipping the server-side dry-run. It executes the cert-manager and" \ - "Traefik admission webhooks against the production API server, so it is limited" \ - "to pushes to main. CRDs are still schema-checked by kubeconform above, and the" \ - "server-side pass still runs on main before the deploy." - - - name: Validate active manifests against the live API server - if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main' - shell: bash - run: | - set -euo pipefail - - if ! kubectl get --raw='/readyz' --request-timeout=10s >/dev/null 2>&1; then - echo "::warning::Cluster unreachable — skipped server-side validation of CRDs (IngressRoute, Certificate, PrometheusRule). Review manifest changes manually." - exit 0 - fi - - mapfile -t k8s_dirs < <( - git ls-files '*.yaml' '*.yml' \ - | grep -E '(^|/)k8s/' \ - | sed -E 's#((^|.*/)k8s)/.*#\1#' \ - | sort -u - ) - - manifests=() - kustomize_apps=() - for dir in "${k8s_dirs[@]}"; do - if [ ! -f "${dir}/active" ]; then - echo "skip (no k8s/active): ${dir}" - continue - fi - if [ -f "${dir}/overlays/prod/kustomization.yaml" ]; then - kustomize_apps+=("${dir}/overlays/prod") - elif [ -f "${dir}/base/kustomization.yaml" ]; then - kustomize_apps+=("${dir}/base") - else - while IFS= read -r f; do - [ -n "$f" ] && manifests+=("$f") - done < <( - git ls-files "${dir}/*.yaml" "${dir}/*.yml" \ - | grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' - ) - fi - done - - echo "server-side dry-run: ${#manifests[@]} manifests, ${#kustomize_apps[@]} kustomize apps" - failed=0 - for m in ${manifests[@]+"${manifests[@]}"}; do - if [[ "$m" == "prometheus-stack/k8s/vmagent.yaml" ]] \ - && ! kubectl get crd vmagents.operator.victoriametrics.com >/dev/null 2>&1; then - echo "skip server-side dry-run until the VictoriaMetrics Operator CRD is installed: $m" - continue - fi - if ! out="$(kubectl apply --dry-run=server -f "$m" 2>&1)"; then - failed=1 - echo "::error file=${m}::$(printf '%s' "$out" | head -1)" - fi - done - for k in ${kustomize_apps[@]+"${kustomize_apps[@]}"}; do - if ! out="$(kubectl apply -k "$k" --dry-run=server 2>&1)"; then - failed=1 - echo "::error file=${k}::$(printf '%s' "$out" | head -1)" - fi - done - - if [ "$failed" -ne 0 ]; then - echo "Server-side validation failed. The API server (or an admission webhook) rejected these manifests." - exit 1 - fi - echo "server-side dry-run: all active manifests accepted by the API server" - build: needs: - # The panel's scan-deps/test-backend/test-frontend jobs gated here until - # userbot moved to its own repo; upstream's code is upstream's gate now. - # The rule is unchanged: publishing and passing the checks are the same - # gate, so a commit that fails any of these still cannot move :prod. - [lint-actionlint, lint-shellcheck, lint-compose, lint-prettier, lint-ruff, lint-yaml, lint-dockerfiles, validate] - if: github.event_name != 'pull_request' && (github.ref_name == 'main' || github.ref_name == 'dev') && !startsWith(github.ref_name, 'renovate/') - runs-on: [self-hosted, linux, arch, homelab] + - checks + if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main' + runs-on: homelab timeout-minutes: 60 - outputs: - services: ${{ steps.services.outputs.services }} steps: - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 with: fetch-depth: 0 - - - name: Detect changed docker-built services - id: services - shell: bash - env: - PUSH_BEFORE: ${{ github.event.before }} - run: | - set -euo pipefail - base="${PUSH_BEFORE:-}" - empty_tree="$(git hash-object -t tree /dev/null)" - if [[ "$base" =~ ^0{40}$ ]]; then - base="$empty_tree" - elif [[ ! "$base" =~ ^[0-9a-fA-F]{40}$ ]] || ! git cat-file -e "${base}^{commit}" 2>/dev/null; then - # Some Gitea push payloads expose `before` as multiple root commits - # joined by newlines. It is not a usable diff base; use this push's - # first parent so image changes in the current commit are still built. - base="$(git rev-parse "${GITHUB_SHA}^" 2>/dev/null || printf '%s' "$empty_tree")" - echo "::warning::invalid push-before value; comparing against ${base}" - fi - - # A failed diff used to leave changed_files empty, which reads exactly - # like "nothing to build": the job went green having built nothing and - # the tag never moved. The status is checked, not assumed. - if ! changed="$(git diff --name-only "$base" "${GITHUB_SHA}")"; then - echo "::error::cannot diff ${base}..${GITHUB_SHA}" - exit 1 - fi - mapfile -t changed_files <<<"$changed" - - services=() - - add_service() { - local name="$1" - local seen=0 - for existing in "${services[@]}"; do - if [ "$existing" = "$name" ]; then - seen=1 - break - fi - done - if [ "$seen" -eq 0 ]; then - services+=("$name") - fi - } - - for file in "${changed_files[@]}"; do - case "$file" in - errorpages/*) - add_service errorpages - ;; - homepages/*) - add_service homepages - ;; - esac - done - - if [ "${#services[@]}" -eq 0 ]; then - echo "No docker-built services changed." - echo "services=" >> "$GITHUB_OUTPUT" - exit 0 - fi - - printf '%s\n' "${services[@]}" | tee /tmp/services.txt - echo "services=$(paste -sd, /tmp/services.txt)" >> "$GITHUB_OUTPUT" - - - name: Log in to registry - # The pin step below also writes (manifest PUTs), and it runs on every - # main push — including manifest-only ones where services is empty. A - # stale persistent login on the old runner used to mask this; a clean - # runner pushes anonymously and gets 401. - if: steps.services.outputs.services != '' || github.ref_name == 'main' - shell: bash - # Through env, not by substitution into the script. A secret written - # into a run: block is pasted into the shell source before bash parses - # it, so a password containing a quote, a backtick or $(...) becomes - # code that runs. Masking the value in the log does not prevent that. + - name: Build changed images and write release env: + GITEA_TOKEN: ${{ github.token }} REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }} REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }} - run: | - set -euo pipefail - printf '%s' "$REGISTRY_PASSWORD" | docker login "${REGISTRY}" \ - -u "$REGISTRY_USERNAME" \ - --password-stdin - - - name: Build and push changed images - if: steps.services.outputs.services != '' - shell: bash - run: | - # This step was the one run: block in the workflow without it, and it - # is the one that cannot afford it: a docker push that failed partway - # through the loop used to be followed by more pushes, the loop's exit - # status came from the last one, and the job went green with half the - # images missing from the registry. - set -euo pipefail - IFS=, read -r -a services <<< "${{ steps.services.outputs.services }}" - - # Tags for this push. The commit-pinned name is the point of this - # step: the deploy resolves it in preference to :prod, so a deploy - # that sat in the queue behind a later push still gets the build of - # the commit CI validated, instead of whatever :prod points at by the - # time it runs. See render_pinned in deploy-lib.sh. - commit_tag="" - if [ "${GITHUB_REF_NAME}" = "main" ]; then - commit_tag="sha-${GITHUB_SHA:0:12}" - fi - - set_tags() { - tags=() - case "${GITHUB_REF_NAME}" in - main) tags+=("main" "prod") ;; - dev) tags+=("dev") ;; - esac - if [ -n "$commit_tag" ]; then - tags+=("$commit_tag") - fi - } - - for service in "${services[@]}"; do - case "$service" in - errorpages) - image="${REGISTRY}/forust/error-pages" - set_tags - build_args=() - for tag in "${tags[@]}"; do - build_args+=(-t "${image}:${tag}") - done - docker build \ - --cache-from "type=registry,ref=${image}:buildcache" \ - --cache-to "type=registry,ref=${image}:buildcache,mode=max" \ - "${build_args[@]}" errorpages - for tag in "${tags[@]}"; do - docker push "${image}:${tag}" - done - ;; - homepages) - for variant in forust xdfnx; do - case "$variant" in - forust) - image="${REGISTRY}/forust/forust-homepage" - ;; - xdfnx) - image="${REGISTRY}/forust/xdfnx-homepage" - ;; - esac - set_tags - build_args=() - for tag in "${tags[@]}"; do - build_args+=(-t "${image}:${tag}") - done - docker build \ - --cache-from "type=registry,ref=${image}:buildcache" \ - --cache-to "type=registry,ref=${image}:buildcache,mode=max" \ - "${build_args[@]}" -f "homepages/Dockerfile.${variant}" homepages - for tag in "${tags[@]}"; do - docker push "${image}:${tag}" - done - done - ;; - esac - done - - # Every image the tree names has to carry the commit-pinned name, not only - # the ones this push rebuilt. A push that touches nothing but manifests - # builds nothing, and its deploy would then find no commit-pinned tag to - # resolve and quietly fall back to the moving :prod - which is the whole - # failure the commit-pinned name exists to remove. - # - # Re-tagging copies the manifest list and transfers no layers, so pinning - # six images that already exist costs six registry writes. - # - # The list is derived from the tree rather than written out here, so an - # image added to a manifest is covered without a second place to update. - - name: Pin the commit name on the images this push did not rebuild - if: github.ref_name == 'main' - shell: bash - run: | - set -euo pipefail - commit_tag="sha-${GITHUB_SHA:0:12}" - mapfile -t repos < <( - git grep -hoE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+' -- '*.yaml' '*.yml' \ - | sort -u - ) - if [ "${#repos[@]}" -eq 0 ]; then - echo "No own images referenced by the tree." - exit 0 - fi - echo "pinning ${#repos[@]} image(s) to $commit_tag" - for repo in "${repos[@]}"; do - # EDU images are released by the application repository and pinned - # directly by digest in edu_master manifests. Never retag them here. - case "$repo" in - */session-keeper|*/webinar-checker) continue ;; - esac - if docker buildx imagetools inspect "$repo:$commit_tag" >/dev/null 2>&1; then - echo " already built by this push: ${repo##*/}" - continue - fi - if ! docker buildx imagetools inspect "$repo:prod" >/dev/null 2>&1; then - echo " WARNING: ${repo##*/} has no :prod to pin and no build produced it" - continue - fi - docker buildx imagetools create --tag "$repo:$commit_tag" "$repo:prod" - echo " pinned ${repo##*/}" - done + run: python3 .gitea/workflows/release.py build + - name: Store commit release + uses: actions/upload-artifact@c6a366c94c3e0affe28c06c8df20a878f24da3cf + with: + name: release-${{ github.sha }} + path: release.json + if-no-files-found: error + retention-days: 30 diff --git a/.gitea/workflows/compose-release.py b/.gitea/workflows/compose-release.py new file mode 100644 index 0000000..bad8d99 --- /dev/null +++ b/.gitea/workflows/compose-release.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Resolve Compose images without changing project names or local bind paths.""" + +import json +import os +import re +import subprocess +import sys +from pathlib import Path + + +def output(*args, **kwargs): + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603 + + +def resolve(reference): + if '@sha256:' in reference: + return reference + descriptor = json.loads( + output('docker', 'buildx', 'imagetools', 'inspect', reference, '--format', '{{json .Manifest}}') + ) + digest = descriptor['digest'] + if not re.fullmatch(r'sha256:[0-9a-f]{64}', digest): + raise ValueError(f'Invalid registry digest for {reference}') + # Strip tag only from the final path segment (registry ports are preserved). + repository = reference.rsplit('/', 1) + repository[-1] = repository[-1].split(':')[0] + return '/'.join(repository) + '@' + digest + + +def prepare(source_file): + config_repo = Path(os.environ['CONFIG_REPO']) + source_repo = Path(os.environ['REPO']) + directory = Path(os.environ['RUN_DIR']) + relative = source_file.relative_to(source_repo) + project_dir = config_repo / relative.parent + base = ['docker', 'compose', '--project-directory', str(project_dir), '-f', str(source_file)] + config = json.loads(output(*base, 'config', '--format', 'json', cwd=config_repo)) + project = config['name'] + previous_file = directory / 'previous.json' + previous = json.loads(previous_file.read_text()) if previous_file.exists() else {} + images_file = directory / 'compose-images.json' + locks = json.loads(images_file.read_text()) if images_file.exists() else previous.get('compose-images', {}) + release = json.loads((directory / 'release.json').read_text()) + before = json.loads(json.dumps(config)) + for service, settings in config['services'].items(): + reference = settings.get('image') + if not reference or settings.get('build'): + raise ValueError(f'{project}/{service}: Compose deploy requires a published image') + image_repo = reference.split('@')[0].rsplit('/', 1) + image_repo[-1] = image_repo[-1].split(':')[0] + image_repo = '/'.join(image_repo) + if image_repo in release['images']: + pinned = image_repo + '@' + release['images'][image_repo] + elif os.environ.get('REFRESH_IMAGES') != 'true' and reference in locks: + pinned = locks[reference] + else: + pinned = resolve(reference) + settings['image'] = pinned + locks[reference] = pinned + # Capture what is running, not the current value of its mutable tag. + ids = output( + 'docker', + 'ps', + '-aq', + '--filter', + f'label=com.docker.compose.project={project}', + '--filter', + f'label=com.docker.compose.service={service}', + ).splitlines() + actual = set() + for container in ids: + image_id = output('docker', 'inspect', container, '--format', '{{.Image}}') + digests = json.loads(output('docker', 'image', 'inspect', image_id, '--format', '{{json .RepoDigests}}')) + actual.add(next((d for d in digests or [] if d.split('@')[0] == image_repo), image_id)) + if len(actual) > 1: + raise ValueError(f'{project}/{service}: mixed running images, cannot capture one recovery config') + before['services'][service]['image'] = next(iter(actual)) if actual else reference + for name, data in (('compose', config), ('compose-before', before)): + folder = directory / name + folder.mkdir(mode=0o700, exist_ok=True) + destination = folder / f'{relative.parent.name}.json' + destination.write_text(json.dumps(data, indent=2) + '\n') + destination.chmod(0o600) + images_file.write_text(json.dumps(locks, indent=2) + '\n') + print(f'Compose {project}: images pinned; local paths preserved') + print( + f'Recovery: docker compose --project-directory {project_dir} -p {project} -f {directory}/compose-before/{relative.parent.name}.json up -d --pull never' + ) + + +if __name__ == '__main__': + prepare(Path(sys.argv[1])) diff --git a/.gitea/workflows/deploy-controller.py b/.gitea/workflows/deploy-controller.py new file mode 100644 index 0000000..432f672 --- /dev/null +++ b/.gitea/workflows/deploy-controller.py @@ -0,0 +1,323 @@ +#!/usr/bin/env python3 +"""Durable workstation deployment controller. Install with setup-workstation.sh.""" + +import argparse +import contextlib +import fcntl +import importlib.util +import json +import math +import os +import re +import shutil +import subprocess +import sys +import time +from pathlib import Path + +STATE = Path(os.environ.get('HOMELAB_STATE', Path.home() / '.local/state/homelab-deploy')) +CONFIG_REPO = Path(os.environ.get('HOMELAB_REPO', '/srv/homelab')) +RUN_ID = re.compile(r'[0-9]+-[0-9]+') + + +def command(*args, **kwargs): + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603, S607 + + +def atomic_json(path, data): + temporary = path.with_suffix('.tmp') + temporary.write_text(json.dumps(data, indent=2) + '\n') + temporary.chmod(0o600) + temporary.replace(path) + + +@contextlib.contextmanager +def lock(name): + STATE.mkdir(mode=0o700, parents=True, exist_ok=True) + with (STATE / name).open('a') as stream: + fcntl.flock(stream, fcntl.LOCK_EX) + yield + + +def load_module(name, path): + spec = importlib.util.spec_from_file_location(name, path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def run_directory(run_id): + if not RUN_ID.fullmatch(run_id): + raise ValueError('Run ID must be numeric workflow-id and attempt') + return STATE / 'runs' / run_id + + +def start(run_id): + payload = sys.stdin.buffer.read(256 * 1024 + 1) + if len(payload) > 256 * 1024: + raise ValueError('Deploy request exceeds 256 KiB') + request = json.loads(payload) + sha = request['release']['sha'] + if not re.fullmatch(r'[0-9a-f]{40}', sha) or request['mode'] not in ('changed', 'full', 'plan'): + raise ValueError('Invalid deploy SHA or mode') + if not isinstance(request['refresh_images'], bool): + raise ValueError('refresh_images must be boolean') + directory = run_directory(run_id) + with lock('prepare.lock'): + if (directory / 'request.json').exists(): + if json.loads((directory / 'request.json').read_text()) != request: + raise ValueError('Run ID already belongs to a different request') + else: + directory.mkdir(mode=0o700, parents=True, exist_ok=True) + command('git', '-C', str(CONFIG_REPO), 'fetch', '--quiet', 'origin', 'main') + command('git', '-C', str(CONFIG_REPO), 'merge-base', '--is-ancestor', sha, 'origin/main') + if not (directory / 'source').exists(): + command('git', '-C', str(CONFIG_REPO), 'worktree', 'add', '--detach', str(directory / 'source'), sha) + if command('git', '-C', str(directory / 'source'), 'rev-parse', 'HEAD') != sha: + raise ValueError('Prepared source does not match deploy SHA') + release_module = load_module('release', directory / 'source/.gitea/workflows/release.py') + release_module.validate_release(request['release'], sha) + atomic_json(directory / 'release.json', request['release']) + atomic_json(directory / 'request.json', request) + if not (directory / 'status.json').exists(): + atomic_json(directory / 'status.json', {'state': 'queued', 'stages': {}}) + # Starting an existing active or finished ID is idempotent; never re-apply it. + if json.loads((directory / 'status.json').read_text())['state'] == 'queued': + command('systemctl', '--user', 'start', '--no-block', f'homelab-deploy@{run_id}.service') + print(f'Accepted deploy {run_id} ({sha})') + + +def environment(directory): + request = json.loads((directory / 'request.json').read_text()) + return { + **os.environ, + 'REPO': str(directory / 'source'), + 'CONFIG_REPO': str(CONFIG_REPO), + 'RUN_DIR': str(directory), + 'DEPLOY_SHA': request['release']['sha'], + 'RELEASE_FILE': str(directory / 'release.json'), + 'DEPLOY_PLAN': str(directory / 'plan.json'), + 'DEPLOY_SNAPSHOT_DIR': str(directory / 'snapshot'), + 'REFRESH_IMAGES': str(request['refresh_images']).lower(), + 'ROLLOUT_PARALLELISM': '4', + } + + +def stage(directory, name, budget): + status = json.loads((directory / 'status.json').read_text()) + if name in status['stages'] and status['stages'][name].get('result') in ('success', 'failure'): + return status['stages'][name]['result'] == 'success' + started = time.time() + status['stages'][name] = {'result': 'running', 'started': started} + atomic_json(directory / 'status.json', status) + script = directory / 'source/.gitea/workflows/deploy-stage.sh' + with (directory / f'{name}.log').open('a') as log: + # timeout kills the whole stage process group, including children, before recovery. + result = subprocess.run( # noqa: S603, S607 + [ + shutil.which('timeout') or '/usr/bin/timeout', + '--signal=TERM', + '--kill-after=30s', + str(budget), + 'bash', + str(script), + name, + ], + env=environment(directory), + stdout=log, + stderr=subprocess.STDOUT, + check=False, + ).returncode + status = json.loads((directory / 'status.json').read_text()) + status['stages'][name].update( + result='success' if result == 0 else 'failure', exit_code=result, seconds=round(time.time() - started) + ) + atomic_json(directory / 'status.json', status) + return result == 0 + + +def make_plan(directory): + source = directory / 'source' + planner = load_module('deploy_plan', source / '.gitea/workflows/deploy-plan.py') + request = json.loads((directory / 'request.json').read_text()) + previous = json.loads((STATE / 'last-success.json').read_text()) if (STATE / 'last-success.json').exists() else None + helm = json.loads(command('helm', 'list', '--all', '-A', '-o', 'json')) + plan = planner.make_plan(source, CONFIG_REPO, request['release'], previous, request['mode'], helm) + if request['refresh_images']: + plan['selected']['compose'] = plan['active']['compose'] + atomic_json(directory / 'plan.json', plan) + if previous: + atomic_json(directory / 'previous.json', previous) + # Local config is deliberately separate from the immutable Git source. + return plan + + +def finish_success(directory, plan): + # Repeating finalization after a crash is safe while holding deploy.lock. + plan['run_id'] = directory.name + path = directory / 'compose-images.json' + previous = directory / 'previous.json' + plan['compose-images'] = ( + json.loads(path.read_text()) + if path.exists() + else json.loads(previous.read_text()).get('compose-images', {}) + if previous.exists() + else {} + ) + atomic_json(STATE / 'last-success.json', plan) + status = json.loads((directory / 'status.json').read_text()) + status['state'] = 'success' + atomic_json(directory / 'status.json', status) + try: + retain_completed(directory) + except (OSError, subprocess.CalledProcessError) as error: + print(f'Retention deferred: {error}', flush=True) + + +def recover(directory, retry=False): + status = json.loads((directory / 'status.json').read_text()) + if status['state'] in ('success', 'planned'): + return + completed = ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke') + if all(status['stages'].get(name, {}).get('result') == 'success' for name in completed): + finish_success(directory, json.loads((directory / 'plan.json').read_text())) + return + if retry: + for name in ('verify-k8s', 'smoke'): + if status['stages'].get(name, {}).get('result') == 'failure': + del status['stages'][name] + atomic_json(directory / 'status.json', status) + snapshot = directory / 'snapshot/current' + if snapshot.exists(): + stage(directory, 'verify-k8s', 7200) + stage(directory, 'smoke', 600) + status = json.loads((directory / 'status.json').read_text()) + status['state'] = 'failure' + atomic_json(directory / 'status.json', status) + + +def execute(run_id): + directory = run_directory(run_id) + with lock('deploy.lock'): + status = json.loads((directory / 'status.json').read_text()) + if status['state'] != 'queued': + return + # A crashed predecessor must be recovered before another apply begins. + for other in (STATE / 'runs').iterdir(): + if ( + other != directory + and (other / 'status.json').exists() + and json.loads((other / 'status.json').read_text())['state'] == 'running' + ): + raise ValueError(f'Interrupted deploy {other.name}; run recover first') + status['state'] = 'running' + atomic_json(directory / 'status.json', status) + try: + plan = make_plan(directory) + print( + json.dumps({'selected': plan['selected'], 'helm': plan['helm'], 'manual_removals': plan['removed']}), + flush=True, + ) + if not stage(directory, 'doctor', 600): + raise RuntimeError('Preflight failed') + if not stage(directory, 'validate', 1200): + raise RuntimeError('Validation failed') + if json.loads((directory / 'request.json').read_text())['mode'] == 'plan': + status = json.loads((directory / 'status.json').read_text()) + status['state'] = 'planned' + atomic_json(directory / 'status.json', status) + return + # Budget includes both rollout checks and rollback waves, plus API overhead. + count = int( + command( + 'bash', + str(directory / 'source/.gitea/workflows/deploy-stage.sh'), + 'workload-count', + env=environment(directory), + ) + ) + verify_budget = max(600, 2 * math.ceil(count / 4) * 300 + 120) + if verify_budget > 7200: + raise ValueError('More than two hours of recovery required; split this deploy') + k8s_ok = stage(directory, 'apply-k8s', 2700) + compose_ok = stage(directory, 'apply-compose', 1800) if k8s_ok else False + verify_ok = stage(directory, 'verify-k8s', verify_budget) + smoke_ok = stage(directory, 'smoke', 600) + if not all((k8s_ok, compose_ok, verify_ok, smoke_ok)): + raise RuntimeError('Deploy failed; inspect stage logs and recovery report') + finish_success(directory, plan) + except Exception as error: + with (directory / 'controller.log').open('a') as stream: + stream.write(f'{error}\n') + recover(directory) + raise + + +def retain_completed(current): + finished = [] + for directory in (STATE / 'runs').iterdir(): + status_file = directory / 'status.json' + if status_file.exists() and json.loads(status_file.read_text())['state'] in ('success', 'planned'): + finished.append(directory) + for directory in sorted(finished, key=lambda p: p.stat().st_mtime, reverse=True)[20:]: + if directory == current: + continue + command('git', '-C', str(CONFIG_REPO), 'worktree', 'remove', '--force', str(directory / 'source')) + shutil.rmtree(directory) + + +def follow(run_id, phase): + directory = run_directory(run_id) + groups = { + 'apply': ('doctor', 'validate', 'apply-k8s', 'apply-compose'), + 'verify': ('verify-k8s',), + 'smoke': ('smoke',), + } + names = groups[phase] + offsets = {} + while True: + status = json.loads((directory / 'status.json').read_text()) + for name in (*names, 'controller'): + path = directory / f'{name}.log' + if path.exists(): + with path.open() as stream: + stream.seek(offsets.get(name, 0)) + content = stream.read() + if content: + print(content, end='', flush=True) + offsets[name] = stream.tell() + stages = status['stages'] + if all(stages.get(name, {}).get('result') in ('success', 'failure') for name in names): + return all(stages[name]['result'] == 'success' for name in names) + if status['state'] in ('success', 'failure', 'planned'): + return status['state'] in ('success', 'planned') + time.sleep(3) + + +def main(): + os.umask(0o077) + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow')) + parser.add_argument('run_id') + parser.add_argument('phase', nargs='?', choices=('apply', 'verify', 'smoke')) + parser.add_argument('--retry', action='store_true', help='Retry failed recovery checks; never repeat apply') + args = parser.parse_args() + directory = run_directory(args.run_id) + if args.action == 'start': + start(args.run_id) + elif args.action == 'execute': + execute(args.run_id) + elif args.action == 'recover': + with lock('deploy.lock'): + recover(directory, retry=args.retry) + elif args.action == 'status': + print((directory / 'status.json').read_text()) + if (directory / 'plan.json').exists(): + plan = json.loads((directory / 'plan.json').read_text()) + print(json.dumps({k: plan[k] for k in ('sha', 'selected', 'helm', 'removed')}, indent=2)) + elif not follow(args.run_id, args.phase): + sys.exit(1) + + +if __name__ == '__main__': + main() diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index b25c4cd..8731b90 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Shared stages for the deploy workflow. Runs on the workstation, invoked as: +# Workstation deploy stages; invoked by the durable controller against pinned source. # REPO=/srv/homelab APPLY_PRUNE=false bash -se <<'EOF' # source "$REPO/.gitea/workflows/deploy-lib.sh" # run_stage "$STAGE" @@ -8,8 +8,8 @@ set -euo pipefail : "${REPO:?REPO must be set}" APPLY_PRUNE="${APPLY_PRUNE:-false}" -# Commit CI validated. Empty for a manual workflow_dispatch, which falls back to -# the current origin/main. +CONFIG_REPO="${CONFIG_REPO:-$REPO}" +# Exact SHA accepted by the CI gate for both manual and automatic deploys. DEPLOY_SHA="${DEPLOY_SHA:-}" # Handoff point between the apply stage (writes) and the verify stage (reads). # Under the deploy user's own XDG state directory rather than /var/backups: the @@ -20,7 +20,7 @@ DEPLOY_SNAPSHOT_DIR="${DEPLOY_SNAPSHOT_DIR:-${XDG_STATE_HOME:-$HOME/.local/state # Per-workload rollout budget and how many workloads to watch at once. The whole # apply job has its own timeout-minutes as a backstop. ROLLOUT_TIMEOUT="${ROLLOUT_TIMEOUT:-300}" -ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-8}" +ROLLOUT_PARALLELISM="${ROLLOUT_PARALLELISM:-4}" WORKLOAD_KINDS="deployments.apps,statefulsets.apps,daemonsets.apps" log() { @@ -60,6 +60,25 @@ kustomize_overlay() { fi } +selected_service() { + local kind="$1" service="$2" section=selected + [ -n "${DEPLOY_PLAN:-}" ] || return 0 + if [ "${DEPLOY_SMOKE_ALL:-false}" = true ]; then section=active; fi + jq -e --arg kind "$kind" --arg service "$service" --arg section "$section" \ + '.[$section][$kind] | index($service) != null' "$DEPLOY_PLAN" >/dev/null +} + +# Resolve .env and relative binds on the persistent workstation tree. Locked +# JSON configs keep the same Compose project name and volume names. +compose() { + local cf="$1" locked project_dir + project_dir="$CONFIG_REPO/$(basename "$(dirname "$cf")")" + shift + locked="${RUN_DIR:-/nonexistent}/compose/$(basename "$(dirname "$cf")").json" + if [ -f "$locked" ]; then cf="$locked"; fi + (cd "$CONFIG_REPO" && docker compose --project-directory "$project_dir" -f "$cf" "$@") +} + select_manifests() { K8S_MANIFESTS=() KUSTOMIZE_APPS=() @@ -67,6 +86,7 @@ select_manifests() { local kd_rel kd overlay cf_rel cf f while IFS= read -r kd_rel; do kd="$REPO/$kd_rel" + selected_service k8s "${kd_rel%/k8s}" || continue if [ ! -f "$kd/active" ]; then echo "skip (no k8s/active): $kd_rel" continue @@ -88,6 +108,7 @@ select_manifests() { ) while IFS= read -r cf_rel; do cf="$REPO/$cf_rel" + selected_service compose "$(dirname "$cf_rel")" || continue if [ -f "$(dirname "$cf")/active" ]; then echo "compose: $cf_rel" COMPOSE_STACKS+=("$cf") @@ -104,16 +125,10 @@ select_manifests() { # on failure roll them back to the revision that was running before, so a bad # push to main cannot leave a service crash-looping. # -# Verification lives in its own workflow job, not at the end of the apply stage. -# Inside a single process it is worthless exactly when it is needed most: a job -# killed by timeout-minutes or cancelled mid-apply never reaches the rollback -# code, and leaves a half-applied cluster behind. Split out, the apply job can -# die in any way and the verify job still runs. +# The workstation controller runs apply and verification as separate durable +# stages. Runner jobs only follow their logs. ExecStopPost recovers interrupted +# runs using the per-run snapshot, even when the SSH connection has gone away. # -# That split needs a handoff point on the workstation, because the two stages are -# separate processes on separate runner jobs: DEPLOY_SNAPSHOT_DIR/current, written -# before anything is applied, read by the verify stage afterwards. - # Creates this run's snapshot directory and publishes it as the handoff point for # the verify stage. Fails hard by design: a deploy that cannot record what it is # about to change must not start, because then nothing can be rolled back for it @@ -143,18 +158,35 @@ snapshot_dir() { } save_snapshot() { - local dir="$1" + local dir="$1" releases revision status log "Saving pre-apply snapshot to $dir" - workload_generations >"$dir/generations.before" 2>/dev/null \ - || warn "could not snapshot workload generations" - kubectl get "$WORKLOAD_KINDS" -A -o yaml >"$dir/workloads.yaml" 2>/dev/null \ - || warn "could not snapshot workloads" - for release in prometheus-stack loki alloy; do - if helm status "$release" -n prometheus >/dev/null 2>&1; then - { - echo "revision: $(helm history "$release" -n prometheus -o json 2>/dev/null)" - helm get values "$release" -n prometheus --all 2>/dev/null - } >"$dir/helm-$release.txt" + workload_generations >"$dir/generations.before" || return 1 + kubectl get "$WORKLOAD_KINDS" -A -o json >"$dir/workloads.json" || return 1 + kubectl get controllerrevisions.apps -A -o json >"$dir/controller-revisions.json" || return 1 + jq --slurpfile revisions "$dir/controller-revisions.json" ' + [.items[] | . as $w | { + kind: (.kind | ascii_downcase), namespace: .metadata.namespace, name: .metadata.name, uid: .metadata.uid, + revision: (if .kind == "Deployment" then (.metadata.annotations["deployment.kubernetes.io/revision"] // "0" | tonumber) + else ([$revisions[0].items[] | select(.metadata.namespace == $w.metadata.namespace) + | select(any(.metadata.ownerReferences[]?; .uid == $w.metadata.uid)) + | select($w.kind != "StatefulSet" or .metadata.name == $w.status.currentRevision) | .revision] | max // 0) end) + }]' "$dir/workloads.json" >"$dir/revisions.json" || return 1 + releases="$(helm list --all -A -o json)" || return 1 + for entry in "${HELM_RELEASES[@]}"; do + IFS='|' read -r release _ namespace _ _ _ <<<"$entry" + if ! jq -e --arg r "$release" --arg n "$namespace" \ + 'any(.[]; .name == $r and .namespace == $n)' <<<"$releases" >/dev/null; then + continue + fi + helm status "$release" -n "$namespace" -o json >"$dir/helm-$release.json" || return 1 + status="$(jq -r '.info.status' "$dir/helm-$release.json")" + if [ "$status" != deployed ]; then + # Never capture a pending/failed revision as the recovery target. + helm history "$release" -n "$namespace" -o json >"$dir/helm-$release.history.json" || return 1 + revision="$(jq '[.[] | select(.status == "deployed" or .status == "superseded") | .revision] | max // 0' \ + "$dir/helm-$release.history.json")" + jq --argjson revision "$revision" '.version = $revision' "$dir/helm-$release.json" >"$dir/helm-$release.tmp" + mv "$dir/helm-$release.tmp" "$dir/helm-$release.json" fi done # The verify stage compares this against the commit it is deploying, to refuse @@ -178,294 +210,24 @@ workload_generations() { # moved since the snapshot, i.e. the ones this apply actually touched. changed_workloads() { local before="$1" - local ns name kind gen old + local ns name kind gen old current + current="$(workload_generations)" || return 1 while read -r ns name kind gen; do [ -n "${gen:-}" ] || continue - old="$(awk -v want_ns="$ns" -v want_name="$name" \ - '$1 == want_ns && $2 == want_name { print $4; exit }' "$before" 2>/dev/null || true)" + old="$(awk -v want_ns="$ns" -v want_name="$name" -v want_kind="$kind" \ + '$1 == want_ns && $2 == want_name && $3 == want_kind { print $4; exit }' "$before" 2>/dev/null || true)" + if [ -n "${RUN_DIR:-}" ] && ! grep -qxF "$kind $ns $name" "$RUN_DIR/workload-refs"; then + continue + fi if [ "$old" != "$gen" ]; then printf '%s %s %s\n' "$kind" "$ns" "$name" fi - done < <(workload_generations) + done <<<"$current" } -# Prints " / " for every workload this repository owns that -# runs an image from our own registry. -# -# The repository is the scope, deliberately. The cluster also holds workloads on -# our registry that no manifest here declares (they are applied out of band), and -# those are somebody else's to deploy. Walking the manifests rather than the -# cluster means those can never be restarted by this pipeline, now or later. -owned_registry_workloads() { - local kd_rel f - while IFS= read -r kd_rel; do - [ -f "$REPO/$kd_rel/active" ] || continue - while IFS= read -r f; do - [ -n "$f" ] || continue - # A file that does not mention the registry cannot declare a workload on it, - # and parsing costs ~2.5s per file against a millisecond for the grep. The - # filter keeps this at a handful of parses instead of one per manifest. - grep -q 'gcr\.forust\.xyz/forust/' "$REPO/$f" 2>/dev/null || continue - # kubectl prints a bare object for a single-document file and a List for a - # multi-document one, so normalise both shapes before filtering. - kubectl apply --dry-run=client -f "$REPO/$f" -o json 2>/dev/null \ - | jq -r ' - (if .items then .items[] else . end) - | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) - | select(any((.spec.template.spec.containers // [])[]?; - (.image // "") | test("^gcr\\.forust\\.xyz/forust/"))) - | (.metadata.namespace // "default") as $ns - | ([.spec.template.spec.containers[].image - | select(test("^gcr\\.forust\\.xyz/forust/"))][0]) as $img - | "\($ns) \(.kind | ascii_downcase)/\(.metadata.name) \($img)" - ' 2>/dev/null || true - done < <(collect_k8s "$kd_rel" || true) - done < <( - git -C "$REPO" ls-files '*.yaml' '*.yml' \ - | grep -E '(^|/)k8s/' \ - | sed -E 's#((^|.*/)k8s)/.*#\1#' \ - | sort -u - ) -} - -# Prints the digest an image tag resolves to for this cluster's architecture, or -# nothing when it cannot be resolved. -# -# Only the manifest entry matching the node architecture counts. A multi-arch tag -# also carries `unknown/unknown` entries for the build attestation, and a pod's -# imageID is always the per-platform digest, so comparing the wrong entry would -# mark every workload stale forever and restart the whole cluster on every deploy. -registry_digest() { - local arch - arch="$(kubectl get nodes -o jsonpath='{.items[0].status.nodeInfo.architecture}' 2>/dev/null || true)" - [ -n "$arch" ] || arch=amd64 - # The || true is load-bearing. Every caller runs under set -euo pipefail, and - # pipefail reports the rightmost non-zero stage, so a ref the registry does not - # have would abort the caller at the assignment instead of yielding an empty - # string. The callers check for empty themselves and report it by name. - # - # Retried with a hard timeout because the registry has a known hang mode (and - # a known blink mode: a single failed lookup aborts the whole apply file in - # render_pinned). A short sleep between attempts lets a restarting registry - # come back instead of failing the deploy on one bad second. - local attempt=0 digest="" - while [ "$attempt" -lt 3 ]; do - digest="$(timeout 25s docker manifest inspect "$1" 2>/dev/null \ - | jq -r --arg arch "$arch" ' - .manifests[]? - | select(.platform.os == "linux" and .platform.architecture == $arch) - | .digest - ' 2>/dev/null \ - | head -1 || true)" - [ -n "$digest" ] && break - attempt=$((attempt + 1)) - if [ "$attempt" -lt 3 ]; then - echo "WARNING: registry lookup of $1 failed (attempt $attempt/3), retrying in 5s" >&2 - sleep 5 - fi - done - printf '%s' "$digest" -} - -# The commit this deploy is for: what CI validated, or - on a manual dispatch, -# whatever stage_preflight just checked out. -deploy_commit() { - local c="${DEPLOY_SHA:-}" - [ -n "$c" ] || c="$(git -C "$REPO" rev-parse HEAD 2>/dev/null || true)" - printf '%.12s' "${c:-}" -} - -# Resolves one of our image refs to the digest THIS commit's build produced. -# -# A manifest naming `:prod` names a pointer, not a version, and the deploy -# resolves it when the apply runs - which is not when CI ran it. Deploy runs are -# queued rather than cancelled (see deploy.yaml), so two pushes in a row leave -# the first deploy resolving the second push's build: the right manifests with -# the wrong code, and nothing anywhere reports it. ci therefore publishes every -# image it ships under `sha-`, a name that cannot move, and that is -# the name resolved here. -# -# The fallback to the plain tag is for an image this pipeline never built. It -# reports itself, because a fallback nobody sees is the failure this removes. -pinned_digest() { - local ref="$1" commit pinned - commit="$(deploy_commit)" - if [ -n "$commit" ]; then - pinned="$(registry_digest "${ref%:*}:sha-$commit")" - if [ -n "$pinned" ]; then - printf '%s' "$pinned" - return 0 - fi - fi - pinned="$(registry_digest "$ref")" - if [ -n "$pinned" ]; then - echo "WARNING: ${ref} carries no sha-${commit:-} tag; resolved the moving tag instead" >&2 - fi - printf '%s' "$pinned" -} - -# Rewrites our own images to immutable digests on the way into the cluster. -# Reads a manifest stream on stdin, writes the pinned stream to stdout. -# -# A digest is not knowable when a manifest is written, so it is never committed: -# git keeps a readable `:prod` tag and the exact bytes are chosen here, at apply -# time, from the tag ci published for the commit being deployed. That is what -# makes rollback mean something. `kubectl rollout undo` restores the previous -# ReplicaSet's pod template verbatim, and a template naming a digest restores the -# exact bytes that were serving before. A template naming a moving tag does not — -# the tag has already moved by the time the rollback runs, so the "rollback" -# re-pulls the very image that just failed and the cluster stays broken. -# -# imagePullPolicy is deliberately left alone. The manifests no longer set it, and a -# reference that is not `:latest` defaults to IfNotPresent, which is what the -# Kubernetes docs ask for alongside a digest: the bytes under a digest cannot -# change, so pulling again buys nothing. -# -# An image that cannot be resolved is fatal. Carrying on would quietly apply a -# mutable tag again, which is the exact failure this function exists to remove. +# Resolve owned image references exclusively from the checked CI artifact. render_pinned() { - local src refs map ref digest missing=0 - src="$(mktemp)" - refs="$(mktemp)" - map="$(mktemp)" - - cat >"$src" - grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+' "$src" | sort -u >"$refs" || true - - while read -r ref; do - [ -n "$ref" ] || continue - digest="$(pinned_digest "$ref")" - if [ -z "$digest" ]; then - echo "ERROR: cannot resolve ${ref} in the registry; applying nothing." >&2 - echo " The build job has to push that tag before the deploy resolves it." >&2 - missing=$((missing + 1)) - continue - fi - printf '%s\t%s\n' "$ref" "$digest" >>"$map" - done <"$refs" - if [ "$missing" -gt 0 ]; then - rm -f "$src" "$refs" "$map" - return 1 - fi - - awk -v mapfile="$map" ' - BEGIN { - while ((getline line < mapfile) > 0) { - i = index(line, "\t") - d[substr(line, 1, i - 1)] = substr(line, i + 1) - } - } - { - if (match($0, /^[[:space:]]*image:[[:space:]]*gcr\.forust\.xyz\/forust\/[A-Za-z0-9._-]+:[A-Za-z0-9._-]+[[:space:]]*$/)) { - name = $0 - sub(/^[[:space:]]*image:[[:space:]]*/, "", name) - sub(/[[:space:]]*$/, "", name) - if (name in d) { - pad = $0 - sub(/image:.*/, "", pad) - # Drop the tag: the canonical form used in the docs is repo@sha256:..., - # and leaving :prod next to the digest reads like it still matters. - repo = name - sub(/:[A-Za-z0-9._-]+$/, "", repo) - print pad "image: " repo "@" d[name] - next - } - } - print - } - ' "$src" - rm -f "$src" "$refs" "$map" -} - -# Restarts every owned workload whose running image is not the one its tag -# resolves to now. -# -# This used to be how a rebuild reached the cluster at all: the manifests pinned -# `:latest`, so a rebuild left the pod template byte-identical, `kubectl apply` -# decided there was nothing to do, and the cluster served the previous build -# indefinitely. The apply now pins digests via render_pinned, so a rebuild moves -# the pod template and rolls out on its own. -# -# What is left is the drift check: a hand-run `kubectl set image`, or anything -# else that edits a live workload behind the deploy's back, is the only way to end -# up serving a digest the tag has moved past. It stays idempotent, so a redeploy -# that changed no image still does not bounce healthy services. -# -# The container is matched on its repository rather than on the exact reference: -# once render_pinned has run, a pod's status reports `repo@sha256:...` while this -# still reads the repository's `:prod` tag out of the manifest. -restart_stale_images() { - local ns target image want selector running entry one - local unchecked=0 - local -A digests=() - local -a stale=() - while read -r ns target image; do - [ -n "${target:-}" ] || continue - if [ -z "${digests[$image]:-}" ]; then - digests[$image]="$(pinned_digest "$image")" - fi - want="${digests[$image]}" - if [ -z "$want" ]; then - warn "cannot resolve ${image##*/} in the registry, leaving $target alone" - unchecked=$((unchecked + 1)) - continue - fi - selector="$(kubectl get "$target" -n "$ns" -o jsonpath='{.spec.selector.matchLabels}' 2>/dev/null \ - | jq -r 'to_entries | map("\(.key)=\(.value)") | join(",")' 2>/dev/null)" - if [ -z "$selector" ]; then - warn "cannot read the pod selector of $target, skipping" - unchecked=$((unchecked + 1)) - continue - fi - running="$(kubectl get pods -n "$ns" -l "$selector" -o json 2>/dev/null \ - | jq -r --arg repo "${image%%:*}" ' - .items[] | .status.containerStatuses[]? - | select(.image == $repo - or (.image | startswith($repo + ":")) - or (.image | startswith($repo + "@"))) - | .imageID - ' 2>/dev/null)" - if [ -z "$running" ]; then - # Scaled to zero. Nothing is serving stale code, and imagePullPolicy - # resolves the tag when it is scaled back up. - continue - fi - entry="" - while IFS= read -r one; do - [ -n "$one" ] || continue - entry="${one##*@}" - if [ "$entry" != "$want" ]; then - stale+=("$ns $target") - break - fi - done <<<"$running" - done < <(owned_registry_workloads) - if [ "${#stale[@]}" -eq 0 ]; then - if [ "$unchecked" -gt 0 ]; then - # Say so plainly. Reporting "everything is current" after checking nothing - # would tell the operator the deploy is fine when it may not be. - warn "No workload needed a restart, but $unchecked could not be checked" - else - log "All owned workloads already run the image their tag points at" - fi - return 0 - fi - log "Restarting ${#stale[@]} workload(s) running an image their tag has moved past" - for ref in "${stale[@]}"; do - log " $ref" - done - local failed=() - for ref in "${stale[@]}"; do - ns="${ref%% *}" - target="${ref#* }" - if ! kubectl rollout restart "$target" -n "$ns" >/dev/null 2>&1; then - failed+=("$ref") - fi - done - if [ "${#failed[@]}" -gt 0 ]; then - warn "could not restart: ${failed[*]}" - return 1 - fi + python3 "$REPO/.gitea/workflows/release.py" render } # verify_workloads ... @@ -509,31 +271,44 @@ verify_workloads() { # settle. Prints a report and returns non-zero if any workload is still unhealthy, # so the operator knows manual recovery is required. rollback_workloads() { - local failed_file="$1" - local kind ns name unrecovered=() - local -a recovered=() + local failed_file="$1" snapshot kind ns name index=0 running=0 pid revision uid + local -a pids=() + snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")" while read -r kind ns name; do - [ -n "${kind:-}" ] || continue - # Helm-owned workloads are already rolled back by the release's --rollback-on-failure - # upgrade. `rollout undo` here would step back to the revision Helm just - # escaped (the failed one), so leave them for the operator instead. - if kubectl get "${kind}/${name}" -n "$ns" -o jsonpath='{.metadata.annotations}' 2>/dev/null | grep -q 'meta.helm.sh/release-name'; then - echo " skip (helm-managed, needs manual check): ${kind}/${ns}/${name}" - unrecovered+=("${kind}/${ns}/${name} (helm-managed)") - continue - fi - if kubectl rollout undo "${kind}/${name}" -n "$ns" >/dev/null 2>&1 \ - && kubectl rollout status "${kind}/${name}" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" >/dev/null 2>&1; then - echo " rolled back: ${kind}/${ns}/${name}" - recovered+=("${kind}/${ns}/${name}") - else - echo " NOT RECOVERED: ${kind}/${ns}/${name}" - unrecovered+=("${kind}/${ns}/${name}") + [[ "$kind" =~ ^(deployment|statefulset|daemonset)$ ]] || continue + index=$((index + 1)) + ( + if kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.annotations}' | grep -q 'meta.helm.sh/release-name'; then + echo " skip (Helm recovery owns this workload): $kind/$ns/$name" + exit 1 + fi + revision="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \ + '.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .revision' "$snapshot/revisions.json")" + uid="$(jq -r --arg ns "$ns" --arg name "$name" --arg kind "$kind" \ + '.[] | select(.namespace == $ns and .name == $name and .kind == $kind) | .uid' "$snapshot/revisions.json")" + if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]] || [ "$uid" != "$(kubectl get "$kind/$name" -n "$ns" -o jsonpath='{.metadata.uid}')" ]; then + echo " no safe previous revision: $kind/$ns/$name (new or replaced workload)" + exit 1 + fi + kubectl rollout undo "$kind/$name" -n "$ns" --to-revision="$revision" \ + && kubectl rollout status "$kind/$name" -n "$ns" --timeout="${ROLLOUT_TIMEOUT}s" + ) >"$snapshot/rollback-$index.log" 2>&1 & + pids+=($!) + running=$((running + 1)) + if [ "$running" -ge "$ROLLOUT_PARALLELISM" ]; then + wait -n 2>/dev/null || true + running=$((running - 1)) fi done <"$failed_file" - echo "ROLLED_BACK=${#recovered[@]}" >>"$failed_file" - echo "UNRECOVERED=${#unrecovered[@]}" >>"$failed_file" - [ "${#unrecovered[@]}" -eq 0 ] + local recovered=0 unrecovered=0 i=0 + for pid in "${pids[@]}"; do + i=$((i + 1)) + if wait "$pid"; then recovered=$((recovered + 1)); else unrecovered=$((unrecovered + 1)); fi + cat "$snapshot/rollback-$i.log" + done + echo "ROLLED_BACK=$recovered" >>"$failed_file" + echo "UNRECOVERED=$unrecovered" >>"$failed_file" + [ "$unrecovered" -eq 0 ] } # Helm releases owned by this stage, one line each: @@ -568,8 +343,12 @@ helm_repo_for() { helm_release_status() { local out if ! out="$(helm status "$1" -n "$2" 2>&1)"; then - echo "not-found" - return 0 + if [[ "$out" == *"release: not found"* ]]; then + echo "not-found" + return 0 + fi + printf 'ERROR: cannot read Helm status: %s\n' "$out" >&2 + return 1 fi awk '/^STATUS:/{print $2}' <<<"$out" | tr '[:upper:]' '[:lower:]' } @@ -581,16 +360,27 @@ helm_release_status() { # pending (deployed, failed, not-found). Returns non-zero when the release is # still not recoverable, so the pipeline fails loud instead of wedging. recover_pending_release() { - local release="$1" namespace="$2" status - status="$(helm_release_status "$release" "$namespace")" + local release="$1" namespace="$2" status revision snapshot + status="$(helm_release_status "$release" "$namespace")" || return 1 case "$status" in pending-upgrade|pending-rollback|pending-install) log "Release $release is $status, rolling back to the last deployed revision" - if ! helm rollback "$release" -n "$namespace" --wait --timeout 10m >/dev/null 2>&1; then + revision="" + if [ -s "$DEPLOY_SNAPSHOT_DIR/current" ]; then + snapshot="$(cat "$DEPLOY_SNAPSHOT_DIR/current")" + if [ -s "$snapshot/helm-$release.json" ]; then + revision="$(jq -r '.version' "$snapshot/helm-$release.json")" + fi + fi + if [[ ! "$revision" =~ ^[1-9][0-9]*$ ]]; then + echo "ERROR: no captured Helm revision for $release; manual recovery required" + return 1 + fi + if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then echo "WARN: helm rollback of $release did not complete" return 1 fi - status="$(helm_release_status "$release" "$namespace")" + status="$(helm_release_status "$release" "$namespace")" || return 1 if [ "$status" != "deployed" ]; then echo "WARN: $release is $status after rollback" return 1 @@ -626,11 +416,16 @@ upgrade_helm_releases() { local entry release chart namespace version values marker repo for entry in ${HELM_RELEASES[@]+"${HELM_RELEASES[@]}"}; do IFS='|' read -r release chart namespace version values marker <<<"$entry" + if [ -n "${DEPLOY_PLAN:-}" ] && ! jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null; then + echo "skip (unchanged Helm release): $release" + continue + fi + if [ ! -f "$REPO/$values" ] && [ -f "$CONFIG_REPO/$values" ]; then values="$CONFIG_REPO/$values"; else values="$REPO/$values"; fi if [ ! -f "$REPO/$marker" ]; then echo "skip (no $marker): $release" continue fi - if [ ! -f "$REPO/$values" ]; then + if [ ! -f "$values" ]; then echo "ERROR: $values is gitignored but missing on the workstation, restore it first." return 1 fi @@ -639,8 +434,8 @@ upgrade_helm_releases() { echo "ERROR: no Helm repository configured for chart $chart" return 1 fi - helm repo add "${repo%% *}" "${repo#* }" >/dev/null 2>&1 || true - helm repo update "${repo%% *}" >/dev/null 2>&1 || true + helm repo add "${repo%% *}" "${repo#* }" >/dev/null + helm repo update "${repo%% *}" >/dev/null log "Upgrading $release ($chart $version)" wait_for_calm "helm $release" # A previous run with --rollback-on-failure whose own rollback never finished leaves the @@ -656,7 +451,7 @@ upgrade_helm_releases() { if ! helm upgrade --install "$release" "$chart" \ --namespace "$namespace" \ --version "$version" \ - --values "$REPO/$values" \ + --values "$values" \ --wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then echo "WARN: upgrade of $release failed, checking release state" # --rollback-on-failure already attempted its own rollback; finish the job when that @@ -672,30 +467,42 @@ upgrade_helm_releases() { done } -stage_preflight() { - if [ ! -d "$REPO/.git" ]; then - echo "Repository not found at $REPO" - exit 1 - fi - if [ -n "$DEPLOY_SHA" ]; then - log "Checking out the commit CI validated ($DEPLOY_SHA)" - git -C "$REPO" fetch origin --quiet "$DEPLOY_SHA" 2>/dev/null \ - || git -C "$REPO" fetch origin main - else - git -C "$REPO" fetch origin main - fi - target="${DEPLOY_SHA:-origin/main}" - log "Workstation state" - echo " local: $(git -C "$REPO" rev-parse --short HEAD)" - echo " target: $(git -C "$REPO" rev-parse --short "$target")" - if [ -n "$(git -C "$REPO" status --porcelain --untracked-files=no)" ]; then - echo "ERROR: workstation has local tracked modifications, refusing reset:" - git -C "$REPO" status --porcelain --untracked-files=no - git -C "$REPO" diff --stat - echo "Fix it on the workstation (commit, or 'git restore .'), then re-run the deploy." - exit 1 - fi - git -C "$REPO" reset --hard "$target" +stage_doctor() { + local tool entry release chart namespace version values marker + for tool in git docker kubectl helm jq curl timeout flock python3; do + command -v "$tool" >/dev/null || { echo "Missing workstation tool: $tool"; return 1; } + done + docker compose version >/dev/null + docker buildx version >/dev/null + [ "$(kubectl config current-context)" = "${KUBE_CONTEXT:?configure KUBE_CONTEXT}" ] || { echo "Unexpected Kubernetes context"; return 1; } + [ "$(kubectl get namespace kube-system -o jsonpath='{.metadata.uid}')" = "${EXPECTED_CLUSTER_UID:?configure EXPECTED_CLUSTER_UID}" ] || { echo "Unexpected Kubernetes cluster"; return 1; } + kubectl get --raw=/readyz --request-timeout=10s >/dev/null + [ "$(git -C "$REPO" rev-parse HEAD)" = "$DEPLOY_SHA" ] || return 1 + select_manifests + for entry in "${HELM_RELEASES[@]}"; do + IFS='|' read -r release chart namespace version values marker <<<"$entry" + [ -f "$REPO/$marker" ] || continue + [ -f "$REPO/$values" ] || [ -f "$CONFIG_REPO/$values" ] || { echo "Missing values: $values"; return 1; } + done + jq '{sha, selected, helm, removed}' "$DEPLOY_PLAN" + local cf + for cf in "${COMPOSE_STACKS[@]}"; do + compose "$cf" config --quiet + while IFS= read -r network; do + docker network inspect "$network" >/dev/null || return 1 + done < <(compose "$cf" config --format json | jq -r '.networks // {} | to_entries[] | select(.value.external == true) | .value.name') + python3 "$REPO/.gitea/workflows/compose-release.py" "$cf" + done + local image refs m k + refs="$( + for m in "${K8S_MANIFESTS[@]}"; do render_pinned <"$m" || return 1; done + for k in "${KUSTOMIZE_APPS[@]}"; do kubectl kustomize "$k" | render_pinned || return 1; done + )" || return 1 + refs="$(grep -oE 'gcr\.forust\.xyz/forust/[A-Za-z0-9._-]+@sha256:[0-9a-f]{64}' <<<"$refs" | sort -u || true)" + while IFS= read -r image; do + [ -n "$image" ] || continue + timeout 60s docker buildx imagetools inspect "$image" >/dev/null + done <<<"$refs" } # Required pod Secrets, scoped to the resource namespace. TLS route Secrets are @@ -759,7 +566,7 @@ stage_validate() { log "Validate compose stacks" for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do echo " config: $cf" - validate_compose_file "$cf" + compose "$cf" config --quiet done log "Validate k8s manifests (kubectl dry-run=client)" for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do @@ -786,6 +593,21 @@ stage_validate() { check_referenced_secrets } +selected_workload_refs() { + local m k + for m in "${K8S_MANIFESTS[@]}"; do + if skip_uninstalled_vmagent_crd "$m" >/dev/null; then continue; fi + kubectl create --dry-run=client --validate=false -f "$m" -o json | jq -r ' + (if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) + | "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"' + done + for k in "${KUSTOMIZE_APPS[@]}"; do + kubectl kustomize "$k" | kubectl create --dry-run=client --validate=false -f - -o json | jq -r ' + (if .kind == "List" then .items[] else . end) | select(.kind | test("^(Deployment|StatefulSet|DaemonSet)$")) + | "\(.kind | ascii_downcase) \(.metadata.namespace // "default") \(.metadata.name)"' + done +} + stage_apply_k8s() { check_prune_mode || return 1 cd "$REPO" @@ -793,16 +615,18 @@ stage_apply_k8s() { local ns_files=() other_files=() m k for m in ${K8S_MANIFESTS[@]+"${K8S_MANIFESTS[@]}"}; do case "$m" in - */namespace.y?ml) ns_files+=("$m") ;; + */namespace.yaml|*/namespace.yml) ns_files+=("$m") ;; *) other_files+=("$m") ;; esac done # Record what is about to change, and publish it for the verify job, before # the first apply. Both are fatal on failure: see snapshot_dir. + selected_workload_refs >"$RUN_DIR/workload-refs" local snapshot snapshot="$(snapshot_dir)" || return 1 save_snapshot "$snapshot" || return 1 + touch "$snapshot/ready" if [ "${#ns_files[@]}" -gt 0 ]; then log "Applying namespaces (${#ns_files[@]} files)" @@ -810,8 +634,8 @@ stage_apply_k8s() { kubectl apply -f "$m" done fi - if [ -f "$REPO/prometheus-stack/k8s/active" ]; then - if [ ! -f "$REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then + if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then + if [ ! -f "$CONFIG_REPO/prometheus-stack/k8s/grafana-values.yaml" ]; then echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first." exit 1 fi @@ -821,6 +645,7 @@ stage_apply_k8s() { if [ "${#other_files[@]}" -gt 0 ]; then log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do + log "Applying ${m#"$REPO"/}" if ! render_pinned <"$m" | kubectl apply -f -; then echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 exit 1 @@ -834,7 +659,6 @@ stage_apply_k8s() { exit 1 fi done - restart_stale_images # No verification here on purpose. This stage may be killed at any point by # timeout-minutes, by the runner cancelling the job, or by a dropped SSH @@ -849,7 +673,7 @@ stage_apply_k8s() { # and rolls back the ones that never became healthy. stage_verify_k8s() { local pointer="$DEPLOY_SNAPSHOT_DIR/current" - local snapshot want have generations + local snapshot want have generations changed local -a touched=() if [ ! -s "$pointer" ]; then @@ -860,7 +684,7 @@ stage_verify_k8s() { return 1 fi snapshot="$(head -1 "$pointer")" - if [ ! -d "$snapshot" ]; then + if [ ! -d "$snapshot" ] || [ ! -f "$snapshot/ready" ]; then echo "ERROR: snapshot pointer refers to a missing directory: $snapshot" return 1 fi @@ -883,6 +707,13 @@ stage_verify_k8s() { fi echo " snapshot: $snapshot (commit ${have:0:12})" + local entry release chart namespace version values marker + for entry in "${HELM_RELEASES[@]}"; do + IFS='|' read -r release chart namespace version values marker <<<"$entry" + jq -e --arg name "$release" '.helm | index($name) != null' "$DEPLOY_PLAN" >/dev/null || continue + recover_pending_release "$release" "$namespace" || return 1 + done + generations="$snapshot/generations.before" if [ ! -s "$generations" ]; then # Without a baseline we cannot tell which workloads the apply touched, so @@ -891,9 +722,10 @@ stage_verify_k8s() { : >"$generations" fi + changed="$(changed_workloads "$generations")" || return 1 while read -r kind ns name; do [ -n "${kind:-}" ] && touched+=("$kind $ns $name") - done < <(changed_workloads "$generations") + done <<<"$changed" log "Verifying ${#touched[@]} changed workload(s) (timeout ${ROLLOUT_TIMEOUT}s each)" if [ "${#touched[@]}" -eq 0 ]; then @@ -932,21 +764,19 @@ stage_verify_k8s() { verify_compose_stack() { local cf="$1" local expected running missing=() - expected="$(docker compose -f "$cf" config --services 2>/dev/null | sort || true)" - running="$(docker compose -f "$cf" ps --status running --services 2>/dev/null | sort || true)" + expected="$(compose "$cf" config --format json | jq -r ' .services | to_entries[] | select(.value.restart != "no") | .key' | sort)" || return 1 + running="$(compose "$cf" ps --status running --services | sort)" || return 1 [ -n "$expected" ] || return 0 while IFS= read -r svc; do [ -n "$svc" ] || continue # restart:"no" services are allowed to have exited. - if ! printf '%s\n' "$running" | grep -qx "$svc" \ - && ! docker compose -f "$cf" config 2>/dev/null \ - | grep -A5 "^ ${svc}:" | grep -qE 'restart:\s*"?no"?'; then + if ! printf '%s\n' "$running" | grep -qx "$svc"; then missing+=("$svc") fi done <<<"$expected" if [ "${#missing[@]}" -gt 0 ]; then echo " NOT RUNNING: ${missing[*]}" - docker compose -f "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true + compose "$cf" ps --all 2>/dev/null | sed 's/^/ /' || true return 1 fi echo " all ${#expected} service(s) running" @@ -1030,10 +860,11 @@ traefik_routed_hosts() { # cases, so ask Traefik which routes it built and fail on the difference. stage_smoke() { cd "$REPO" + if [ -n "${DEPLOY_PLAN:-}" ] && jq -e '.full_smoke' "$DEPLOY_PLAN" >/dev/null; then + DEPLOY_SMOKE_ALL=true + fi select_manifests >/dev/null local -a hosts=() - # Not named failed: an array of that name already exists in restart_stale_images - # above, and a scalar shadowing an array is a trap rather than a shadow. local h code rc bad=0 while IFS= read -r h; do [ -n "$h" ] && hosts+=("$h") @@ -1041,8 +872,8 @@ stage_smoke() { if [ "${#hosts[@]}" -eq 0 ]; then # Nothing to probe means the extraction broke, not that the cluster is empty. - echo "ERROR: no public hostnames found in active manifests, refusing to report success" - return 1 + echo "No public routes in the selected components" + return 0 fi log "Probing ${#hosts[@]} public route(s)" @@ -1126,37 +957,17 @@ stage_apply_compose() { cd "$REPO" select_manifests >/dev/null local cf - log "Redeploying docker compose stacks (${#COMPOSE_STACKS[@]} stacks)" - for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do - echo " compose: $cf" - if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then - docker compose -f "$cf" build - docker compose -f "$cf" push - fi - docker compose -f "$cf" up -d --pull always --remove-orphans + for cf in "${COMPOSE_STACKS[@]}"; do + log "Applying Compose ${cf#"$REPO"/}" + compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans + verify_compose_stack "$cf" done - - local -a broken=() - for cf in ${COMPOSE_STACKS[@]+"${COMPOSE_STACKS[@]}"}; do - echo " verifying: $cf" - if ! verify_compose_stack "$cf"; then - broken+=("$cf") - fi - done - if [ "${#broken[@]}" -gt 0 ]; then - echo - echo "ERROR: ${#broken[@]} compose stack(s) did not come up:" - printf ' - %s\n' "${broken[@]}" - echo "Compose stacks are not rolled back automatically: their images use mutable" - echo "':latest' tags, so there is no previous version to return to. Check the logs" - echo "above, then re-run the deploy once the cause is fixed." - return 1 - fi + echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)" } run_stage() { case "${1:?stage required}" in - preflight) stage_preflight ;; + doctor) stage_doctor ;; validate) stage_validate ;; apply-k8s) stage_apply_k8s ;; verify-k8s) stage_verify_k8s ;; diff --git a/.gitea/workflows/deploy-plan.py b/.gitea/workflows/deploy-plan.py new file mode 100644 index 0000000..f794ad8 --- /dev/null +++ b/.gitea/workflows/deploy-plan.py @@ -0,0 +1,124 @@ +#!/usr/bin/env python3 +"""Calculate selected components against the last fully successful deploy.""" + +import hashlib +import json +import re +import subprocess +from pathlib import Path + + +def output(*args, **kwargs): + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603 + + +def tracked(repo): + return output('git', '-C', str(repo), 'ls-files').splitlines() + + +def helm_releases(repo): + text = (repo / '.gitea/workflows/deploy-lib.sh').read_text() + return [line.split('|') for line in re.findall(r'^ "([^"\n]+\|[^"\n]+)"$', text, re.MULTILINE)] + + +def inventory(repo): + files = tracked(repo) + k8s = sorted( + {f.split('/k8s/')[0] for f in files if '/k8s/' in f and (repo / f.split('/k8s/')[0] / 'k8s/active').is_file()} + ) + compose = sorted( + { + str(Path(f).parent) + for f in files + if Path(f).name in ('compose.yaml', 'compose.yml') and (repo / Path(f).parent / 'active').is_file() + } + ) + return {'k8s': k8s, 'compose': compose} + + +def file_hash(path): + return hashlib.sha256(path.read_bytes()).hexdigest() if path.is_file() else 'missing' + + +def make_plan(repo, config_repo, release, previous, mode, live_helm): + active = inventory(repo) + all_services = set(active['k8s'] + active['compose']) + helm_inputs = {} + helm_selected = [] + for name, chart, namespace, version, values, marker in helm_releases(repo): + if not (repo / marker).is_file(): + continue + value_path = repo / values if (repo / values).is_file() else config_repo / values + if not value_path.is_file(): + raise ValueError(f'Missing Helm values: {values}') + stamp = hashlib.sha256(f'{chart}|{version}|{file_hash(value_path)}'.encode()).hexdigest() + helm_inputs[name] = stamp + live = next((h for h in live_helm if h['name'] == name and h['namespace'] == namespace), None) + if ( + mode == 'full' + or previous is None + or previous.get('helm_inputs', {}).get(name) != stamp + or live is None + or live.get('status') != 'deployed' + or live.get('chart') != f'{chart.split("/")[-1]}-{version}' + ): + helm_selected.append(name) + local_inputs = {} + for service in all_services: + candidates = [config_repo / service / '.env'] + if service in active['compose']: + candidates.append(config_repo / '.env') + cfg = config_repo / service / 'config' + if cfg.is_dir(): + candidates.extend( + p for p in cfg.rglob('*') if p.is_file() and p.suffix in ('.yaml', '.yml', '.json', '.conf') + ) + local_inputs[service] = hashlib.sha256( + '\n'.join(f'{p.relative_to(config_repo)}:{file_hash(p)}' for p in sorted(candidates)).encode() + ).hexdigest() + if previous is None: + if mode == 'changed': + raise ValueError('No successful baseline; run deploy in full mode first') + changed = set(all_services) + removed = [] + else: + paths = output('git', '-C', str(repo), 'diff', '--name-only', previous['sha'], release['sha']).splitlines() + changed = {path.split('/')[0] for path in paths} + if any(path.startswith('.gitea/') for path in paths): + changed |= all_services + changed |= {s for s in all_services if previous.get('local_inputs', {}).get(s) != local_inputs[s]} + for file in tracked(repo): + service = file.split('/')[0] + if service not in all_services or not file.endswith(('.yaml', '.yml')): + continue + text = (repo / file).read_text() + if any( + image in text and previous.get('images', {}).get(image) != digest + for image, digest in release['images'].items() + ): + changed.add(service) + removed = sorted( + set(previous.get('active', {}).get('k8s', []) + previous.get('active', {}).get('compose', [])) + - all_services + ) + removed += [path for path in paths if '/k8s/' in path and not (repo / path).exists()] + if mode == 'full': + changed = set(all_services) + dependencies = json.loads((repo / '.gitea/deploy-dependencies.json').read_text()) + while True: + expanded = changed | {dependent for service in changed for dependent in dependencies.get(service, [])} + if expanded == changed: + break + changed = expanded + return { + 'version': 1, + 'sha': release['sha'], + 'images': release['images'], + 'active': active, + 'selected': {kind: sorted(set(services) & changed) for kind, services in active.items()}, + 'helm': helm_selected, + 'helm_inputs': helm_inputs, + 'local_inputs': local_inputs, + 'removed': sorted(set(removed)), + 'full_smoke': mode == 'full' or 'traefik' in changed, + } diff --git a/.gitea/workflows/deploy-stage.sh b/.gitea/workflows/deploy-stage.sh new file mode 100755 index 0000000..6d630f3 --- /dev/null +++ b/.gitea/workflows/deploy-stage.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +set -euo pipefail +source "${REPO:?}/.gitea/workflows/deploy-lib.sh" +case "${1:?stage required}" in + workload-count) + select_manifests >/dev/null + selected_workload_refs | sort -u | wc -l + ;; + *) run_stage "$1" ;; +esac diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 6c455b8..2ac1f60 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -1,202 +1,104 @@ name: deploy on: - # Deploy only what CI already validated. workflow_run is used instead of - # workflow_dispatch so a red lint/validate run can never reach the cluster. workflow_run: workflows: [ci] branches: [main] types: [completed] workflow_dispatch: + inputs: + deploy_ref: + description: "Commit already checked by successful main CI (main or SHA)" + default: main + required: true + deploy_mode: + description: "First deploy requires full; plan changes no production resources" + type: choice + options: [changed, full, plan] + default: changed + refresh_images: + description: "Explicitly refresh mutable third-party Compose tags" + type: boolean + default: false -# The deploy jobs read the tree, then reach the cluster over SSH with the -# deploy key. The Actions token itself is not part of that path, so it gets -# read-only contents and no more. permissions: contents: read + actions: read concurrency: group: deploy-main - # Queue instead of cancelling. Cancelling a run kills the apply job mid-loop and - # takes the verify job down with it, so a superseded deploy would leave the - # cluster half-applied and unchecked — the exact failure the verify job exists - # to catch. kubectl apply and docker compose up are both idempotent, so letting - # the older run finish and then deploying the newer commit costs little. cancel-in-progress: false env: - DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} - DEPLOY_PORT: ${{ secrets.DEPLOY_PORT }} - DEPLOY_USER: ${{ secrets.DEPLOY_USER }} - DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }} + DEPLOY_HOST: ${{ vars.DEPLOY_HOST || secrets.DEPLOY_HOST }} + DEPLOY_PORT: ${{ vars.DEPLOY_PORT || secrets.DEPLOY_PORT }} + DEPLOY_USER: ${{ vars.DEPLOY_USER || secrets.DEPLOY_USER }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} - APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} - # workflow_run's own GITHUB_SHA points at the branch head, not at the commit the - # finished ci run checked. Pin the exact validated commit instead, so a push - # landing mid-deploy cannot make the workstation deploy something else. Also - # what the verify job checks the snapshot against. Empty for workflow_dispatch, - # which falls back to the current origin/main. - DEPLOY_SHA: ${{ github.event.workflow_run.head_sha }} + DEPLOY_KNOWN_HOSTS: ${{ vars.DEPLOY_KNOWN_HOSTS }} + DEPLOY_RUN_ID: ${{ github.run_id }}-${{ github.run_attempt || 1 }} + DEPLOY_MODE: ${{ inputs.deploy_mode || 'changed' }} + REFRESH_IMAGES: ${{ inputs.refresh_images && 'true' || 'false' }} jobs: - preflight: - # Autodeploy defaults to OFF: pushes deploy only when the AUTODEPLOY repo - # variable is set to 'true' (Settings -> Actions -> Variables). A manual - # Run workflow always bypasses the switch: dispatching it is the explicit - # intent to deploy. + gate: if: >- + github.ref == 'refs/heads/main' && (vars.AUTODEPLOY == 'true' || github.event_name == 'workflow_dispatch') && (github.event_name != 'workflow_run' || - (github.event.workflow_run.conclusion == 'success' && - github.event.workflow_run.head_branch == 'main')) - runs-on: [self-hosted, linux, arch, homelab, prod] + (github.event.workflow_run.conclusion == 'success' && github.event.workflow_run.head_branch == 'main')) + runs-on: homelab timeout-minutes: 10 + outputs: + sha: ${{ steps.release.outputs.sha }} steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + fetch-depth: 0 + - name: Check successful CI and download the exact commit release + id: release + env: + GITEA_TOKEN: ${{ github.token }} + DEPLOY_REF: ${{ inputs.deploy_ref || 'main' }} + EVENT_SHA: ${{ github.event.workflow_run.head_sha }} + run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA" + - name: Submit durable deploy to workstation + run: bash .gitea/workflows/ssh-run.sh start - - name: Fetch and reset workstation - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh preflight - - validate: - needs: [preflight] - runs-on: [self-hosted, linux, arch, homelab, prod] - timeout-minutes: 20 + apply: + needs: [gate] + runs-on: homelab + timeout-minutes: 100 steps: - - name: Checkout repository + - name: Checkout checked commit uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + ref: ${{ needs.gate.outputs.sha }} + - name: Follow validation and sequential Kubernetes / Compose apply + run: bash .gitea/workflows/ssh-run.sh apply - - name: Dry-run manifests and check Secrets - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh validate - - apply-k8s: - needs: [validate] - runs-on: [self-hosted, linux, arch, homelab, prod] - # Apply only, no verification, so this is just the work itself: snapshot, - # then sequential `helm upgrade --install --wait --rollback-on-failure --timeout 10m`, then the apply loop. - # Verification has its own job and its own budget. - # - # 45 is roughly four times the measured cost of the stage, which is - # deliberately not raised on a theory: - # - # helm, healthy 3 no-op upgrades ~3-5 min - # helm, one release bad rollback-on-failure spends its 10m, ~10-15 min - # then rolls that one back - # apply loop ~40 manifests, 4 of which ~1 min - # resolve an image digest - # restart_stale_images 7.6s to find 8 workloads, ~0.5 min - # 9.8s to resolve their digests - # - # The helm figure is one release, not three: `set -e` aborts - # upgrade_helm_releases on the first failure, so a broken release costs - # 10m and the other two are never attempted. Multiplying 10m by three - # overstates the worst case by 20 minutes. - # - # The 45 minutes this was last raised to 45 were still not enough, and the - # job logs for those runs no longer exist, so what actually consumed the - # budget is not known - the two measurable candidates above account for - # ~15 of it. The unbounded `docker manifest inspect` against the registry's - # known hang mode is now bounded inside registry_digest (25s timeout, 3 - # attempts): a dead registry fails each owned image after ~85s instead of - # hanging the stage, and a blinking one is retried instead of failing the - # whole apply file. Still open: make the stage announce which manifest it - # is working on, so a killed run leaves a diagnosable last line. - timeout-minutes: 45 + verify: + needs: [gate, apply] + if: always() && needs.gate.result == 'success' + runs-on: homelab + timeout-minutes: 130 steps: - - name: Checkout repository + - name: Checkout checked commit uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + ref: ${{ needs.gate.outputs.sha }} + - name: Follow workload verification and recovery + run: bash .gitea/workflows/ssh-run.sh verify - - name: Apply Kubernetes manifests - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh apply-k8s - - apply-compose: - needs: [validate] - runs-on: [self-hosted, linux, arch, homelab, prod] - timeout-minutes: 30 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Redeploy docker compose stacks - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh apply-compose - - # Watches the workloads this deploy changed and rolls back the ones that never - # became healthy. Runs even when the apply jobs failed, timed out or were - # cancelled — that is the whole point of splitting it out. `always()` is what - # lets it start after a failed dependency; the needs on apply-compose are a - # barrier, so verification begins only once both applies are done. - verify-k8s: - needs: [preflight, apply-k8s, apply-compose] - if: >- - always() && - needs.preflight.result == 'success' && - needs.apply-k8s.result != 'skipped' && - needs.apply-compose.result != 'skipped' - runs-on: [self-hosted, linux, arch, homelab, prod] - # Not raised, because the arithmetic does not close. - # - # 32 workloads are under management and the wave width is 8, so the verify - # itself is 4 waves of ROLLOUT_TIMEOUT (300s) = 20 minutes worst case, when - # every rollout times out rather than converging. That is already 20 of 30. - # - # The other 10 would have to absorb rollback, and rollback_workloads is a - # serial `while read` loop at 300s per failed workload. 10 minutes buys two. - # Any larger number is buying a bigger multiple of an unbounded term rather - # than covering a known cost: 60 minutes buys eight, and 60 minutes is - # therefore not a bound, it is a guess with two digits. - # - # The number becomes derivable the moment rollback uses the same wave width - # as the verify: 32 failures then cost 4 waves = 20 minutes instead of 160, - # and 45 covers verify plus rollback at full width. That change is to the - # recovery path and is not folded into a timeout edit. - timeout-minutes: 30 - steps: - - name: Checkout repository - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Verify workloads and roll back on failure - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh verify-k8s - - # Asks the public route of every active service whether it is actually - # serving, which the rollout check above structurally cannot: a pod can - # converge and still be crash-looping, or be listening on a port no Service - # points at, or answer 500. - # - # `always()` for the same reason verify-k8s has it, and it runs after that job - # specifically because a rollback is when a route most needs re-checking. The - # needs is a barrier, not a filter: whether verify-k8s passed, failed or was - # cancelled, the probes are what say whether the cluster is serving, and - # suppressing them on a rollback would hide the one run where the answer - # matters most. smoke: - needs: [preflight, verify-k8s] - if: >- - always() && - needs.preflight.result == 'success' && - needs.verify-k8s.result != 'skipped' - runs-on: [self-hosted, linux, arch, homelab, prod] - timeout-minutes: 10 + needs: [gate, verify] + if: always() && needs.gate.result == 'success' + runs-on: homelab + timeout-minutes: 15 steps: - - name: Checkout repository + - name: Checkout checked commit uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 - - - name: Probe the public route of every active service - shell: bash - run: | - set -euo pipefail - ./.gitea/workflows/ssh-run.sh smoke + with: + ref: ${{ needs.gate.outputs.sha }} + - name: Follow public route checks + run: bash .gitea/workflows/ssh-run.sh smoke diff --git a/.gitea/workflows/install-ci-tools.sh b/.gitea/workflows/install-ci-tools.sh index 2fcb3b3..60c2d87 100755 --- a/.gitea/workflows/install-ci-tools.sh +++ b/.gitea/workflows/install-ci-tools.sh @@ -13,9 +13,14 @@ here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=tool-versions.env . "$here/tool-versions.env" -TOOLS_DIR="${TOOLS_DIR:-${RUNNER_TEMP:-/tmp}/homelab-tools}" +TOOLS_DIR="${TOOLS_DIR:-${XDG_CACHE_HOME:-$HOME/.cache}/homelab-ci}" BIN_DIR="$TOOLS_DIR/bin" mkdir -p "$BIN_DIR" +# A runner may accept overlapping workflows even though each workflow is sequential. +exec 9>"$TOOLS_DIR/install.lock" +flock -w 300 9 +export UV_TOOL_DIR="$TOOLS_DIR/uv-tools" +export UV_CACHE_DIR="$TOOLS_DIR/uv-cache" # The just-installed tools must resolve inside this script too: callers only # prepend BIN_DIR to PATH after the script exits, so a bare `uv` below would # miss the binary install_uv just placed (exit 127 on a clean runner). @@ -48,7 +53,7 @@ esac fetch() { # fetch if command -v curl >/dev/null 2>&1; then - curl -sSLf --retry 3 -o "$2" "$1" + curl -sSLf --connect-timeout 15 --max-time 120 --retry 3 -o "$2" "$1" elif command -v wget >/dev/null 2>&1; then wget -q -O "$2" "$1" else @@ -88,10 +93,13 @@ installed_version() { # at_version at_version() { - case "$(installed_version "$1")" in - *"$2"*) return 0 ;; - *) return 1 ;; - esac + local version expected="${2#v}" + version="$(installed_version "$1")" + if [[ "$version" =~ (^|[^0-9.])v?([0-9]+(\.[0-9]+)+) ]]; then + [ "${BASH_REMATCH[2]}" = "$expected" ] + else + return 1 + fi } install_kubeconform() { @@ -176,6 +184,7 @@ install_pip_audit() { } install_prettier() { + install_node if at_version prettier "${PRETTIER_VERSION}"; then return 0 fi @@ -236,29 +245,41 @@ install_actionlint() { rm -rf "$tmp" } -wanted=("$@") -if [ "${#wanted[@]}" -eq 0 ]; then - wanted=(kubeconform shellcheck actionlint prettier ruff yamllint hadolint) -fi +main() { + wanted=("$@") + if [ "${#wanted[@]}" -eq 0 ]; then + wanted=(node jq kubeconform shellcheck actionlint prettier ruff yamllint hadolint) + fi -for tool in "${wanted[@]}"; do - case "$tool" in - kubeconform) install_kubeconform ;; - shellcheck) install_shellcheck ;; - jq) install_jq ;; - actionlint) install_actionlint ;; - prettier) install_prettier ;; - ruff) install_ruff ;; - yamllint) install_yamllint ;; - pip-audit) install_pip_audit ;; - hadolint) install_hadolint ;; - node) install_node ;; - uv) install_uv ;; - *) - echo "install-ci-tools: unknown tool: $tool" >&2 - exit 1 - ;; - esac -done + for tool in "${wanted[@]}"; do + case "$tool" in + kubeconform) install_kubeconform ;; + shellcheck) install_shellcheck ;; + jq) install_jq ;; + actionlint) install_actionlint ;; + prettier) install_prettier ;; + ruff) install_ruff ;; + yamllint) install_yamllint ;; + pip-audit) install_pip_audit ;; + hadolint) install_hadolint ;; + node) install_node ;; + uv) install_uv ;; + *) + echo "install-ci-tools: unknown tool: $tool" >&2 + exit 1 + ;; + esac + done -printf '%s\n' "$BIN_DIR" + for old in "$BIN_DIR"/node-* "$BIN_DIR"/prettier-*; do + [ -d "$old" ] || continue + case "$(basename "$old")" in + "node-$NODE_VERSION"|"prettier-$PRETTIER_VERSION") ;; + *) rm -rf "$old" ;; + esac + done + if [ -x "$BIN_DIR/uv" ]; then "$BIN_DIR/uv" cache prune >/dev/null; fi + printf '%s\n' "$BIN_DIR" +} + +if [ "${BASH_SOURCE[0]}" = "$0" ]; then main "$@"; fi diff --git a/.gitea/workflows/release.py b/.gitea/workflows/release.py new file mode 100644 index 0000000..daff5aa --- /dev/null +++ b/.gitea/workflows/release.py @@ -0,0 +1,350 @@ +#!/usr/bin/env python3 +"""CI release artifacts and the SHA-specific Gitea deployment gate (stdlib only).""" + +import argparse +import hashlib +import io +import itertools +import json +import os +import re +import shutil +import subprocess +import sys +import tempfile +import urllib.error +import urllib.parse +import urllib.request +import zipfile +from pathlib import Path + +SHA = re.compile(r'[0-9a-f]{40}') +DIGEST = re.compile(r'sha256:[0-9a-f]{64}') +IMAGES = { + 'error-pages': ('errorpages', 'errorpages/Dockerfile'), + 'forust-homepage': ('homepages', 'homepages/Dockerfile.forust'), + 'xdfnx-homepage': ('homepages', 'homepages/Dockerfile.xdfnx'), +} + +# These images are released by the EDU application repository. +EXTERNAL_IMAGES = {'gcr.forust.xyz/forust/session-keeper', 'gcr.forust.xyz/forust/webinar-checker'} + + +def command(*args, **kwargs): + """Arguments are passed directly to the executable, never to a shell.""" + return subprocess.check_output(args, text=True, **kwargs).strip() # noqa: S603, S607 + + +def validate_release(data, sha=None): + if data.get('version') != 1 or not SHA.fullmatch(data.get('sha', '')): + raise ValueError('Invalid release version or SHA') + if sha is not None and data['sha'] != sha: + raise ValueError('Release SHA does not match the checked CI commit') + expected = {f'gcr.forust.xyz/forust/{name}' for name in IMAGES} + if set(data.get('images', {})) != expected: + raise ValueError('Release must contain all owned images') + if not all(DIGEST.fullmatch(value) for value in data['images'].values()): + raise ValueError('Release has an invalid image digest') + if set(data.get('inputs', {})) != expected or not all( + re.fullmatch(r'[0-9a-f]{64}', value) for value in data['inputs'].values() + ): + raise ValueError('Release has invalid build input fingerprints') + return data + + +class NoRedirect(urllib.request.HTTPRedirectHandler): + def redirect_request(self, _req, _fp, _code, _msg, _headers, _newurl): + return None + + +class Gitea: + def __init__(self): + self.origin = os.environ['GITHUB_SERVER_URL'].rstrip('/') + if urllib.parse.urlsplit(self.origin).scheme != 'https': + raise ValueError('Gitea API must use HTTPS') + self.repository = os.environ['GITHUB_REPOSITORY'] + if not re.fullmatch(r'[\w.-]+/[\w.-]+', self.repository): + raise ValueError('Invalid Gitea repository') + self.token = os.environ['GITEA_TOKEN'] + self.base = f'{self.origin}/api/v1/repos/{self.repository}' + + def request(self, url, *, archive=False): + if not url.startswith(self.base + '/'): + raise ValueError('Refusing to send the Actions token to another origin') + req = urllib.request.Request(url, headers={'Authorization': f'token {self.token}'}) # noqa: S310 -- HTTPS origin validated above + opener = urllib.request.build_opener(NoRedirect()) + try: + response = opener.open(req, timeout=30) # noqa: S310 + except urllib.error.HTTPError as error: + if not archive or error.code not in (301, 302, 303, 307, 308): + raise RuntimeError(f'Gitea API returned HTTP {error.code}') from None + target = urllib.parse.urljoin(url, error.headers['Location']) + if urllib.parse.urlsplit(target).scheme != 'https': + raise ValueError('Artifact redirect must use HTTPS') from None + # Signed storage redirects must never receive the Gitea token. + response = urllib.request.urlopen(target, timeout=30) # noqa: S310 + with response: + payload = response.read(8 * 1024 * 1024 + 1) + if len(payload) > 8 * 1024 * 1024: + raise ValueError('Gitea response exceeds 8 MiB') + return payload if archive else json.loads(payload) + + def pages(self, path, key, **params): + for page in range(1, 101): + query = urllib.parse.urlencode({**params, 'page': page, 'limit': 50}) + data = self.request(f'{self.base}/{path}?{query}') + entries = data[key] + yield from entries + if len(entries) < 50: + return + raise RuntimeError('Gitea pagination limit exceeded') + + def successful_runs(self, sha=None): + params = {'branch': 'main', 'status': 'success', 'exclude_pull_requests': 'true'} + if sha: + params['head_sha'] = sha + for run in self.pages('actions/workflows/ci.yaml/runs', 'workflow_runs', **params): + if ( + run.get('status') == 'completed' + and run.get('conclusion') == 'success' + and run.get('head_branch') == 'main' + and run.get('event') in ('push', 'workflow_dispatch') + and (run.get('repository') or {}).get('full_name') == self.repository + and (run.get('head_repository') or run.get('repository') or {}).get('full_name') == self.repository + and (sha is None or run.get('head_sha') == sha) + ): + yield run + + def release(self, run): + sha = run['head_sha'] + jobs = list(self.pages(f'actions/runs/{run["id"]}/jobs', 'jobs')) + # A green workflow with a skipped build must not authorize a deploy. + if not any(job.get('name') == 'build' and job.get('conclusion') == 'success' for job in jobs): + raise ValueError('CI build job did not succeed') + artifacts = self.request(f'{self.base}/actions/runs/{run["id"]}/artifacts')['artifacts'] + matching = [a for a in artifacts if a['name'] == f'release-{sha}' and not a.get('expired')] + if len(matching) != 1: + raise ValueError('CI release artifact is missing, expired or ambiguous; rerun CI') + blob = self.request(f'{self.base}/actions/artifacts/{matching[0]["id"]}/zip', archive=True) + with zipfile.ZipFile(io.BytesIO(blob)) as archive: + files = [entry for entry in archive.infolist() if not entry.is_dir()] + if len(files) != 1 or files[0].filename != 'release.json' or files[0].file_size > 256 * 1024: + raise ValueError('Unexpected release archive contents') + return validate_release(json.loads(archive.read(files[0])), sha) + + +def fingerprint(context, dockerfile): + tree = command('git', 'ls-tree', '-r', 'HEAD', '--', context, dockerfile, '.gitea/workflows/release.py') + return hashlib.sha256(tree.encode()).hexdigest() + + +def gate(output, requested_ref, event_sha): + command('git', 'fetch', '--quiet', 'origin', 'main') + if event_sha: + if not SHA.fullmatch(event_sha): + raise ValueError('Invalid workflow_run SHA') + sha = event_sha + else: + if requested_ref == 'main': + requested_ref = 'origin/main' + sha = command('git', 'rev-parse', '--verify', '--end-of-options', f'{requested_ref}^{{commit}}') + if not SHA.fullmatch(sha): + raise ValueError('Invalid deploy SHA') + command('git', 'merge-base', '--is-ancestor', sha, 'origin/main') + api = Gitea() + runs = list(api.successful_runs(sha)) + if not runs: + raise ValueError(f'No successful main CI for {sha}; run CI before deploying') + release = api.release(max(runs, key=lambda run: run['id'])) + output.write_text(json.dumps(release, indent=2) + '\n') + if os.environ.get('GITHUB_OUTPUT'): + with Path(os.environ['GITHUB_OUTPUT']).open('a') as stream: + stream.write(f'sha={sha}\n') + print(f'CI gate accepted {sha}') + + +def build(output): + sha = command('git', 'rev-parse', 'HEAD') + if sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha): + raise ValueError('Build checkout does not match GITHUB_SHA') + api = Gitea() + previous = None + for run in sorted(itertools.islice(api.successful_runs(), 50), key=lambda item: item['id'], reverse=True): + if str(run['id']) == os.environ.get('GITHUB_RUN_ID'): + continue + try: + previous = api.release(run) + break + except ValueError: + # Expired artifacts only cost a rebuild; mutable tags are never a fallback. + continue + docker_config = tempfile.mkdtemp(prefix='homelab-registry-') + builder_config = Path.home() / '.cache/homelab-ci/buildx' + builder_config.mkdir(parents=True, exist_ok=True) + env = {**os.environ, 'DOCKER_CONFIG': docker_config, 'BUILDX_CONFIG': str(builder_config)} + try: + subprocess.run( # noqa: S603, S607 + [ + shutil.which('docker') or '/usr/bin/docker', + 'login', + 'gcr.forust.xyz', + '-u', + os.environ['REGISTRY_USERNAME'], + '--password-stdin', + ], + input=os.environ['REGISTRY_PASSWORD'], + text=True, + check=True, + env=env, + ) + builder = 'homelab-ci' + versions = dict( + re.findall(r'^([A-Z_]+)="([^"\n]+)"$', Path('.gitea/workflows/tool-versions.env').read_text(), re.MULTILINE) + ) + image = versions['BUILDKIT_IMAGE'] + signature = builder_config / 'homelab-ci-image' + exists = ( + subprocess.run( # noqa: S603 + [shutil.which('docker') or '/usr/bin/docker', 'buildx', 'inspect', builder], + capture_output=True, + env=env, + ).returncode + == 0 + ) + if exists and (not signature.exists() or signature.read_text().strip() != image): + command('docker', 'buildx', 'rm', '--keep-state', builder, env=env) + exists = False + if not exists: + command( + 'docker', + 'buildx', + 'create', + '--name', + builder, + '--driver', + 'docker-container', + '--driver-opt', + f'image={image}', + '--buildkitd-config', + '.gitea/runner/buildkitd.toml', + env=env, + ) + signature.write_text(image + '\n') + release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}} + for name, (context, dockerfile) in IMAGES.items(): + image = f'gcr.forust.xyz/forust/{name}' + inputs = fingerprint(context, dockerfile) + old_digest = (previous or {}).get('images', {}).get(image) + exists = False + if old_digest and previous['inputs'].get(image) == inputs: + exists = ( + subprocess.run( # noqa: S603, S607 + [ + shutil.which('docker') or '/usr/bin/docker', + 'buildx', + 'imagetools', + 'inspect', + f'{image}@{old_digest}', + ], + capture_output=True, + env=env, + timeout=60, + ).returncode + == 0 + ) + if exists: + print(f'Reuse {name}: inputs unchanged') + digest = old_digest + else: + print(f'Build {name}', flush=True) + metadata = Path(docker_config) / 'metadata.json' + command( + 'docker', + 'buildx', + 'build', + '--builder', + builder, + '--push', + '--platform', + 'linux/amd64', + '--provenance=false', + '--cache-from', + f'type=registry,ref={image}:buildcache', + '--cache-to', + f'type=registry,ref={image}:buildcache,mode=max', + '--tag', + f'{image}:sha-{sha}', + '--metadata-file', + str(metadata), + '--file', + dockerfile, + context, + env=env, + ) + digest = json.loads(metadata.read_text())['containerimage.digest'] + release['images'][image] = digest + release['inputs'][image] = inputs + validate_release(release, sha) + output.write_text(json.dumps(release, indent=2) + '\n') + finally: + # Cleanup errors must neither leak credentials nor mask the original build error. + try: + subprocess.run( # noqa: S603 + [ + shutil.which('docker') or '/usr/bin/docker', + 'buildx', + 'prune', + '--builder', + 'homelab-ci', + '--force', + '--max-used-space', + '1gb', + ], + env=env, + timeout=60, + ) + except (OSError, subprocess.TimeoutExpired): + print('CI builder cache cleanup deferred', flush=True) + finally: + shutil.rmtree(docker_config) + + +def render(stream, destination): + release = validate_release(json.loads(Path(os.environ['RELEASE_FILE']).read_text()), os.environ['DEPLOY_SHA']) + image_line = re.compile( + r"^(\s*(?:-\s*)?image:\s*)(['\"]?)(gcr\.forust\.xyz/forust/[\w.-]+)(?::[\w.-]+|@sha256:[0-9a-f]{64})\2(\s*(?:#.*)?)$" + ) + rendered = [] + for line in stream: + match = image_line.fullmatch(line.rstrip('\n')) + if match: + prefix, quote, image, tail = match.groups() + if image in EXTERNAL_IMAGES and f'{image}@sha256:' in line: + rendered.append(line) + continue + if image not in release['images']: + raise ValueError(f'Owned image missing from checked release: {image}') + line = f'{prefix}{quote}{image}@{release["images"][image]}{quote}{tail}\n' + elif re.match(r'\s*(?:-\s*)?image:', line) and 'gcr.forust.xyz/forust/' in line: + raise ValueError('Unsupported owned image syntax; refusing to apply a mutable tag') + rendered.append(line) + destination.writelines(rendered) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('action', choices=('build', 'gate', 'render')) + parser.add_argument('--output', type=Path, default=Path('release.json')) + parser.add_argument('--ref', default='main') + parser.add_argument('--event-sha', default='') + args = parser.parse_args() + if args.action == 'render': + render(sys.stdin, sys.stdout) + elif args.action == 'gate': + gate(args.output, args.ref, args.event_sha) + else: + build(args.output) + + +if __name__ == '__main__': + main() diff --git a/.gitea/workflows/renovate-ci.yaml b/.gitea/workflows/renovate-ci.yaml index 0886d50..5592dc7 100644 --- a/.gitea/workflows/renovate-ci.yaml +++ b/.gitea/workflows/renovate-ci.yaml @@ -26,7 +26,7 @@ permissions: jobs: validate-renovate: - runs-on: [self-hosted, linux, arch, homelab] + runs-on: homelab timeout-minutes: 20 steps: - name: Checkout repository diff --git a/.gitea/workflows/renovate-run.yaml b/.gitea/workflows/renovate-run.yaml index 658a603..05d221a 100644 --- a/.gitea/workflows/renovate-run.yaml +++ b/.gitea/workflows/renovate-run.yaml @@ -32,7 +32,7 @@ concurrency: jobs: run-renovate: - runs-on: [self-hosted, linux, arch, homelab] + runs-on: homelab timeout-minutes: 60 steps: - name: Checkout repository diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index b2d8441..cfc94dc 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -1,71 +1,56 @@ #!/usr/bin/env bash -# usage: ssh-run.sh -# Runs one deploy-lib.sh stage on the workstation over SSH. +# The SSH client submits once and follows durable stages on workstation. set -euo pipefail - : "${DEPLOY_HOST:?missing DEPLOY_HOST}" : "${DEPLOY_USER:?missing DEPLOY_USER}" : "${DEPLOY_KEY:?missing DEPLOY_SSH_KEY}" - -deploy_port="${DEPLOY_PORT:-22}" -deploy_path="${DEPLOY_PATH:-/srv/homelab}" -deploy_path="$(printf '%s' "$deploy_path" | tr -d '\"' | tr -d '\r' | xargs)" - -# The private key is written to a per-run directory that is removed on exit, so a -# failed or cancelled job cannot leave deploy credentials in the runner's temp -# directory. Do not use a fixed path: apply-k8s and apply-compose run in parallel. +: "${DEPLOY_KNOWN_HOSTS:?configure pinned DEPLOY_KNOWN_HOSTS}" +: "${DEPLOY_RUN_ID:?missing DEPLOY_RUN_ID}" +[[ "$DEPLOY_USER" =~ ^[A-Za-z_][A-Za-z0-9_.-]*$ ]] || exit 1 +[[ "$DEPLOY_HOST" =~ ^[A-Za-z0-9_.:-]+$ ]] || exit 1 +[[ "$DEPLOY_RUN_ID" =~ ^[0-9]+-[0-9]+$ ]] || exit 1 +[[ "${DEPLOY_PORT:-22}" =~ ^[0-9]+$ ]] || exit 1 key_dir="$(mktemp -d "${RUNNER_TEMP:-/tmp}/homelab-deploy-key.XXXXXXXX")" -trap 'rm -rf "$key_dir"' EXIT INT TERM - -ssh_key="$key_dir/deploy_key" -printf '%s\n' "$DEPLOY_KEY" > "$ssh_key" -chmod 600 "$ssh_key" - -# A connection that died silently used to hang until the job timeout, and the -# stage was never re-run: one flaky TCP session cost a whole 45-minute apply. -# ServerAlive* bounds how long a dead peer goes unnoticed, ConnectTimeout bounds -# setup. Only exit 255 - ssh's own transport failures - is retried. A stage that -# fails on its own merits exits with the remote's status, so a real failure -# still surfaces its own log instead of burning three attempts. The stages are -# declarative applies, so re-running one that had already committed is harmless. -ssh_opts=( - -i "$ssh_key" -p "$deploy_port" - -o BatchMode=yes -o StrictHostKeyChecking=accept-new - -o ConnectTimeout=15 - -o ServerAliveInterval=15 -o ServerAliveCountMax=4 -) - -rc=0 -# apply-k8s and apply-compose are separate workflow jobs so the graph stays -# intact for the verify job, but on a single node they must not run at once: -# host docker churn on top of cluster churn is what melts the node (load 40+, -# netbird/ssh die, helm is left pending-*). Serialize them on the workstation -# with a shared lock; whoever arrives second waits. -remote_cmd=(bash -se) -case "$1" in - apply-k8s | apply-compose) - remote_cmd=(flock -w 5400 /tmp/homelab-apply.lock bash -se) +trap 'rm -rf "$key_dir"' EXIT +chmod 700 "$key_dir" +printf '%s\n' "$DEPLOY_KEY" >"$key_dir/key" +printf '%s\n' "$DEPLOY_KNOWN_HOSTS" >"$key_dir/known_hosts" +chmod 600 "$key_dir/key" "$key_dir/known_hosts" +ssh_opts=(-i "$key_dir/key" -p "${DEPLOY_PORT:-22}" -o BatchMode=yes -o StrictHostKeyChecking=yes + -o "UserKnownHostsFile=$key_dir/known_hosts" -o ConnectTimeout=15 + -o ServerAliveInterval=15 -o ServerAliveCountMax=4) +controller=.local/lib/homelab-deploy/controller.py +case "${1:?start, apply, verify or smoke required}" in + start) + python3 - <<'PY' >"$key_dir/request.json" +import json +import os +from pathlib import Path +release = json.loads(Path('release.json').read_text()) +print(json.dumps({'release': release, 'mode': os.environ.get('DEPLOY_MODE', 'changed'), + 'refresh_images': os.environ.get('REFRESH_IMAGES', 'false') == 'true'})) +PY + for attempt in 1 2 3; do + rc=0 + # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" start "$DEPLOY_RUN_ID" <"$key_dir/request.json" || rc=$? + [ "$rc" -eq 0 ] && exit 0 + [ "$rc" -eq 255 ] || exit "$rc" + sleep 5 + done + exit "$rc" ;; + apply|verify|smoke) + for attempt in 1 2 3; do + rc=0 + # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$? + [ "$rc" -eq 0 ] && exit 0 + [ "$rc" -eq 255 ] || exit "$rc" + echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" + sleep 5 + done + exit "$rc" + ;; + *) echo "Unknown SSH operation: $1" >&2; exit 1 ;; esac -for attempt in 1 2 3; do - if [ "$attempt" -gt 1 ]; then - echo ":: warning::ssh transport failed, retrying (${attempt}/3)" - sleep $((attempt * 5)) - fi - rc=0 - # shellcheck disable=SC2029 # remote_cmd/ssh_opts expand on the client on purpose: they select the local ssh invocation, only the heredoc runs remotely. - ssh "${ssh_opts[@]}" "${DEPLOY_USER}@${DEPLOY_HOST}" \ - env "REPO=$deploy_path" "APPLY_PRUNE=${APPLY_PRUNE:-false}" \ - "DEPLOY_SHA=${DEPLOY_SHA:-}" "DEPLOY_SNAPSHOT_DIR=${DEPLOY_SNAPSHOT_DIR:-}" \ - "STAGE=$1" "${remote_cmd[@]}" <<'EOF' || rc=$? -source "$REPO/.gitea/workflows/deploy-lib.sh" -run_stage "$STAGE" -EOF - [ "$rc" -eq 0 ] && break - [ "$rc" -ne 255 ] && break -done - -if [ "$rc" -ne 0 ]; then - echo ":: error::stage $1 failed over ssh (exit $rc)" -fi -exit "$rc" diff --git a/.gitea/workflows/tool-versions.env b/.gitea/workflows/tool-versions.env index cf8edb9..6d47ad5 100644 --- a/.gitea/workflows/tool-versions.env +++ b/.gitea/workflows/tool-versions.env @@ -34,3 +34,6 @@ NODE_VERSION="22.23.3" # Secret-reference regression tests parse rendered Kubernetes objects. JQ_VERSION="1.8.1" + +# BuildKit is the only auxiliary CI container; jobs themselves stay on the host. +BUILDKIT_IMAGE="moby/buildkit:v0.33.1" diff --git a/renovate/k8s/configmap.yaml b/renovate/k8s/configmap.yaml index 76baa8b..0b75b93 100644 --- a/renovate/k8s/configmap.yaml +++ b/renovate/k8s/configmap.yaml @@ -178,6 +178,13 @@ data: "datasourceTemplate": "helm", "depNameTemplate": "reloader", "registryUrlTemplate": "https://stakater.github.io/stakater-charts" + }, + { + "customType": "regex", + "description": "Pinned CI BuildKit helper image", + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], + "matchStrings": ["BUILDKIT_IMAGE=\"(?moby/buildkit):(?[^\"\\n]+)\""], + "datasourceTemplate": "docker" } ], "packageRules": [ diff --git a/renovate/renovate.json b/renovate/renovate.json index e9eae92..bd29619 100644 --- a/renovate/renovate.json +++ b/renovate/renovate.json @@ -167,6 +167,13 @@ "datasourceTemplate": "helm", "depNameTemplate": "reloader", "registryUrlTemplate": "https://stakater.github.io/stakater-charts" + }, + { + "customType": "regex", + "description": "Pinned CI BuildKit helper image", + "managerFilePatterns": [".gitea/workflows/tool-versions.env"], + "matchStrings": ["BUILDKIT_IMAGE=\"(?moby/buildkit):(?[^\"\\n]+)\""], + "datasourceTemplate": "docker" } ], "packageRules": [ diff --git a/tests/test_cicd.py b/tests/test_cicd.py new file mode 100644 index 0000000..367c82f --- /dev/null +++ b/tests/test_cicd.py @@ -0,0 +1,267 @@ +"""CI gate, selection, persistent configuration and recovery regression tests.""" + +import importlib.util +import json +import os +import subprocess +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[1] + + +def module(name, filename): + spec = importlib.util.spec_from_file_location(name, ROOT / '.gitea/workflows' / filename) + loaded = importlib.util.module_from_spec(spec) + spec.loader.exec_module(loaded) + return loaded + + +release_module = module('release_test', 'release.py') +planner = module('plan_test', 'deploy-plan.py') +compose_module = module('compose_test', 'compose-release.py') +controller = module('controller_test', 'deploy-controller.py') + + +def release(sha='a' * 40): + return { + 'version': 1, + 'sha': sha, + 'images': {f'gcr.forust.xyz/forust/{name}': 'sha256:' + 'b' * 64 for name in release_module.IMAGES}, + 'inputs': {f'gcr.forust.xyz/forust/{name}': 'c' * 64 for name in release_module.IMAGES}, + } + + +class ReleaseGateTests(unittest.TestCase): + def test_release_rejects_wrong_sha_missing_images_and_mutable_tags(self): + for mutation in ('sha', 'missing', 'tag'): + data = release() + if mutation == 'sha': + data['sha'] = 'd' * 40 + elif mutation == 'missing': + data['images'].pop(next(iter(data['images']))) + else: + data['images'][next(iter(data['images']))] = 'prod' + with self.assertRaises(ValueError): + release_module.validate_release(data, 'a' * 40) + + def test_gate_excludes_pr_wrong_branch_and_failed_runs(self): + api = object.__new__(release_module.Gitea) + api.repository = 'forust/homelab' + good = { + 'id': 1, + 'status': 'completed', + 'conclusion': 'success', + 'head_branch': 'main', + 'event': 'push', + 'head_sha': 'a' * 40, + 'repository': {'full_name': api.repository}, + } + entries = [ + good, + {**good, 'event': 'pull_request'}, + {**good, 'head_branch': 'dev'}, + {**good, 'conclusion': 'failure'}, + {**good, 'head_sha': 'b' * 40}, + {**good, 'head_repository': {'full_name': 'attacker/fork'}}, + ] + with patch.object(api, 'pages', return_value=iter(entries)): + self.assertEqual(list(api.successful_runs('a' * 40)), [good]) + + def test_green_ci_with_skipped_build_is_rejected(self): + api = object.__new__(release_module.Gitea) + with ( + patch.object(api, 'pages', return_value=iter([{'name': 'build', 'conclusion': 'skipped'}])), + self.assertRaisesRegex(ValueError, 'build job'), + ): + api.release({'id': 1, 'head_sha': 'a' * 40}) + + def test_expired_or_ambiguous_artifacts_are_rejected(self): + api = object.__new__(release_module.Gitea) + api.base = 'https://example.test/api/v1/repos/a/b' + artifact = {'id': 1, 'name': 'release-' + 'a' * 40} + for artifacts in ([{**artifact, 'expired': True}], [artifact, artifact], []): + with ( + patch.object(api, 'pages', return_value=iter([{'name': 'build', 'conclusion': 'success'}])), + patch.object(api, 'request', return_value={'artifacts': artifacts}), + self.assertRaisesRegex(ValueError, 'artifact'), + ): + api.release({'id': 1, 'head_sha': 'a' * 40}) + + +class SelectionTests(unittest.TestCase): + def setUp(self): + self.scratch = tempfile.TemporaryDirectory() + self.addCleanup(self.scratch.cleanup) + self.repo = Path(self.scratch.name) + self.git('init', '-q') + self.git('config', 'user.email', 'test@example.test') + self.git('config', 'user.name', 'CI Test') + for service in ('one', 'two', 'postgres'): + directory = self.repo / service / 'k8s' + directory.mkdir(parents=True) + (directory / 'active').touch() + (directory / 'app.yaml').write_text('kind: Deployment\n') + (self.repo / '.gitea/workflows').mkdir(parents=True) + (self.repo / '.gitea/workflows/deploy-lib.sh').write_text('HELM_RELEASES=(\n)\n') + (self.repo / '.gitea/deploy-dependencies.json').write_text('{"postgres": ["one", "two"]}') + self.sha = self.commit() + self.initial = planner.make_plan(self.repo, self.repo, release(self.sha), None, 'full', []) + + def git(self, *args): + return planner.output('git', '-C', str(self.repo), *args) + + def commit(self): + self.git('add', '.') + self.git('commit', '-qm', 'Test state') + return self.git('rev-parse', 'HEAD') + + def test_first_changed_deploy_requires_explicit_full(self): + with self.assertRaisesRegex(ValueError, 'full'): + planner.make_plan(self.repo, self.repo, release(self.sha), None, 'changed', []) + + def test_only_changed_service_is_selected(self): + (self.repo / 'one/k8s/app.yaml').write_text('kind: StatefulSet\n') + result = planner.make_plan(self.repo, self.repo, release(self.commit()), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one']) + self.assertEqual(result['helm'], []) + + def test_failed_intermediate_deploy_does_not_lose_changes(self): + (self.repo / 'one/k8s/app.yaml').write_text('kind: StatefulSet\n') + self.commit() # This commit failed deploy: baseline must remain initial. + (self.repo / 'two/k8s/app.yaml').write_text('kind: StatefulSet\n') + result = planner.make_plan(self.repo, self.repo, release(self.commit()), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one', 'two']) + + def test_dependencies_and_removals_are_reported(self): + (self.repo / 'postgres/k8s/app.yaml').write_text('kind: StatefulSet\n') + (self.repo / 'two/k8s/active').unlink() + result = planner.make_plan(self.repo, self.repo, release(self.commit()), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one', 'postgres']) + self.assertIn('two', result['removed']) + + def test_local_configuration_change_selects_service(self): + (self.repo / 'one/.env').write_text('TEST_VALUE=changed\n') + result = planner.make_plan(self.repo, self.repo, release(self.sha), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], ['one']) + + def test_redeploy_is_noop_and_full_includes_all(self): + result = planner.make_plan(self.repo, self.repo, release(self.sha), self.initial, 'changed', []) + self.assertEqual(result['selected']['k8s'], []) + result = planner.make_plan(self.repo, self.repo, release(self.sha), self.initial, 'full', []) + self.assertEqual(result['selected']['k8s'], ['one', 'postgres', 'two']) + + +class ComposeConfigurationTests(unittest.TestCase): + def test_pin_preserves_project_volumes_paths_and_previous_image(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + run = root / 'run' + source = run / 'source' + config_repo = root / 'persistent' + (source / 'headscale').mkdir(parents=True) + config_repo.mkdir() + (run / 'release.json').write_text(json.dumps(release())) + old = 'busybox@sha256:' + 'd' * 64 + new = 'busybox@sha256:' + 'e' * 64 + config = { + 'name': 'headscale', + 'services': { + 'app': { + 'image': 'busybox:latest', + 'volumes': [ + {'type': 'bind', 'source': str(config_repo / 'headscale/config.yaml'), 'target': '/config'}, + {'type': 'volume', 'source': 'data', 'target': '/data'}, + ], + } + }, + 'volumes': {'data': {'name': 'headscale_data'}}, + } + + def fake_output(*args, **kwargs): + if args[:2] == ('docker', 'compose'): + self.assertEqual(kwargs['cwd'], config_repo) + self.assertIn(str(config_repo / 'headscale'), args) + return json.dumps(config) + if args[:2] == ('docker', 'ps'): + return 'container' + if args[:2] == ('docker', 'inspect'): + return 'sha256:' + 'f' * 64 + return json.dumps([old]) + + with ( + patch.dict(os.environ, {'CONFIG_REPO': str(config_repo), 'REPO': str(source), 'RUN_DIR': str(run)}), + patch.object(compose_module, 'output', side_effect=fake_output), + patch.object(compose_module, 'resolve', return_value=new), + ): + compose_module.prepare(source / 'headscale/compose.yaml') + pinned = json.loads((run / 'compose/headscale.json').read_text()) + before = json.loads((run / 'compose-before/headscale.json').read_text()) + self.assertEqual(pinned['name'], 'headscale') + self.assertEqual(pinned['volumes'], config['volumes']) + self.assertEqual(pinned['services']['app']['volumes'], config['services']['app']['volumes']) + self.assertEqual(pinned['services']['app']['image'], new) + self.assertEqual(before['services']['app']['image'], old) + self.assertEqual((run / 'compose/headscale.json').stat().st_mode & 0o777, 0o600) + + def test_registry_index_and_single_image_descriptors(self): + for digest in ('a' * 64, 'b' * 64): + with patch.object(compose_module, 'output', return_value=json.dumps({'digest': 'sha256:' + digest})): + self.assertEqual( + compose_module.resolve('registry.test:5000/repo:latest'), f'registry.test:5000/repo@sha256:{digest}' + ) + + +class ControllerTests(unittest.TestCase): + def test_completed_stage_cannot_apply_again(self): + with tempfile.TemporaryDirectory() as scratch: + directory = Path(scratch) + controller.atomic_json( + directory / 'status.json', {'state': 'success', 'stages': {'apply-k8s': {'result': 'success'}}} + ) + with patch.object(subprocess, 'run') as execute: + self.assertTrue(controller.stage(directory, 'apply-k8s', 60)) + execute.assert_not_called() + + def test_run_id_is_not_shell_or_path_input(self): + for invalid in ('../123', '-1', '1;touch bad', 'abc', '1/2'): + with self.assertRaises(ValueError): + controller.run_directory(invalid) + + def test_exact_previous_revision_is_used_for_rollback(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + (root / 'current').write_text(str(root)) + (root / 'revisions.json').write_text( + json.dumps([{'kind': 'deployment', 'namespace': 'app', 'name': 'web', 'uid': 'same', 'revision': 7}]) + ) + (root / 'failed').write_text('deployment app web\n') + script = """set -euo pipefail +source "$LIB" +kubectl() { + case "$*" in + *metadata.annotations*) printf '{}' ;; + *metadata.uid*) printf same ;; + 'rollout undo'*) printf '%s\\n' "$*" >>"$CALLS" ;; + 'rollout status'*) return 0 ;; + *) return 1 ;; + esac +} +rollback_workloads "$FAILED" +""" + env = { + **os.environ, + 'REPO': str(ROOT), + 'LIB': str(ROOT / '.gitea/workflows/deploy-lib.sh'), + 'DEPLOY_SNAPSHOT_DIR': str(root), + 'CALLS': str(root / 'calls'), + 'FAILED': str(root / 'failed'), + } + subprocess.run(['/usr/bin/bash', '-c', script], env=env, check=True) # noqa: S603 + self.assertIn('--to-revision=7', (root / 'calls').read_text()) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_cicd_lifecycle.py b/tests/test_cicd_lifecycle.py new file mode 100644 index 0000000..f14caab --- /dev/null +++ b/tests/test_cicd_lifecycle.py @@ -0,0 +1,237 @@ +"""Release publication and controller recovery tests without a live server.""" + +import io +import json +import os +import subprocess +import tempfile +import unittest +import zipfile +from pathlib import Path +from unittest.mock import Mock, patch + +from test_cicd import ROOT, controller, release, release_module + + +class ArtifactTests(unittest.TestCase): + def test_archive_rejects_nested_or_extra_files(self): + api = object.__new__(release_module.Gitea) + api.base = 'https://example.test/api/v1/repos/a/b' + for names in (['../release.json'], ['release.json', 'credentials']): + blob = io.BytesIO() + with zipfile.ZipFile(blob, 'w') as archive: + for name in names: + archive.writestr(name, json.dumps(release())) + replies = [{'artifacts': [{'id': 1, 'name': 'release-' + 'a' * 40}]}, blob.getvalue()] + with ( + patch.object(api, 'pages', return_value=iter([{'name': 'build', 'conclusion': 'success'}])), + patch.object(api, 'request', side_effect=replies), + self.assertRaisesRegex(ValueError, 'archive'), + ): + api.release({'id': 1, 'head_sha': 'a' * 40}) + + def test_quoted_and_single_platform_images_render_from_checked_release(self): + with tempfile.TemporaryDirectory() as scratch: + path = Path(scratch) / 'release.json' + path.write_text(json.dumps(release())) + result = io.StringIO() + image = 'gcr.forust.xyz/forust/error-pages' + with patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}): + release_module.render(io.StringIO(f' image: "{image}:prod" # note\n'), result) + self.assertEqual(result.getvalue(), f' image: "{image}@sha256:{"b" * 64}" # note\n') + + def test_edu_release_digest_is_preserved(self): + with tempfile.TemporaryDirectory() as scratch: + path = Path(scratch) / 'release.json' + path.write_text(json.dumps(release())) + image = 'gcr.forust.xyz/forust/session-keeper' + line = f'image: {image}@sha256:{"e" * 64}\n' + result = io.StringIO() + with patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}): + release_module.render(io.StringIO(line), result) + self.assertEqual(result.getvalue(), line) + with self.assertRaises(ValueError): + release_module.render(io.StringIO(f'image: {image}:prod\n'), io.StringIO()) + + def test_unknown_image_cannot_emit_partial_manifest(self): + with tempfile.TemporaryDirectory() as scratch: + path = Path(scratch) / 'release.json' + path.write_text(json.dumps(release())) + result = io.StringIO() + with ( + patch.dict(os.environ, {'RELEASE_FILE': str(path), 'DEPLOY_SHA': 'a' * 40}), + self.assertRaisesRegex(ValueError, 'missing'), + ): + release_module.render( + io.StringIO('kind: Deployment\nimage: gcr.forust.xyz/forust/unknown:prod\n'), result + ) + self.assertEqual(result.getvalue(), '') + + def test_only_changed_image_is_built_and_credentials_are_removed(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + built = [] + auth_directories = [] + old = release() + + def fake_command(*args, **kwargs): + if args[:2] == ('git', 'rev-parse'): + return 'e' * 40 + if args[:3] == ('docker', 'buildx', 'build'): + built.append(args[args.index('--file') + 1]) + metadata = Path(args[args.index('--metadata-file') + 1]) + metadata.write_text(json.dumps({'containerimage.digest': 'sha256:' + 'f' * 64})) + auth_directories.append(Path(kwargs['env']['DOCKER_CONFIG'])) + return '' + + api = Mock() + api.successful_runs.return_value = iter([{'id': 1}]) + api.release.return_value = old + with ( + patch.dict( + os.environ, + { + 'GITHUB_SHA': 'e' * 40, + 'GITHUB_RUN_ID': '2', + 'REGISTRY_USERNAME': 'test', + 'REGISTRY_PASSWORD': 'placeholder', + }, + ), + patch.object(release_module.Path, 'home', return_value=root), + patch.object(release_module, 'Gitea', return_value=api), + patch.object( + release_module, + 'fingerprint', + side_effect=lambda context, _file: ('d' if context == 'errorpages' else 'c') * 64, + ), + patch.object(release_module, 'command', side_effect=fake_command), + patch.object(subprocess, 'run', return_value=subprocess.CompletedProcess([], 0)), + ): + release_module.build(root / 'release.json') + self.assertEqual(built, ['errorpages/Dockerfile']) + self.assertTrue(all(not directory.exists() for directory in auth_directories)) + self.assertEqual(json.loads((root / 'release.json').read_text())['sha'], 'e' * 40) + + +class DurableRunTests(unittest.TestCase): + def test_duplicate_start_only_reattaches(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + directory.mkdir(parents=True) + request = {'release': release(), 'mode': 'full', 'refresh_images': False} + controller.atomic_json(directory / 'request.json', request) + controller.atomic_json(directory / 'status.json', {'state': 'running', 'stages': {}}) + with ( + patch.object(controller, 'STATE', state), + patch.object(controller.sys, 'stdin', io.TextIOWrapper(io.BytesIO(json.dumps(request).encode()))), + patch.object(controller, 'command') as execute, + ): + controller.start('123-1') + execute.assert_not_called() + + def test_failed_apply_still_verifies_and_does_not_advance_baseline(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + directory.mkdir(parents=True) + controller.atomic_json( + directory / 'request.json', {'release': release(), 'mode': 'full', 'refresh_images': False} + ) + controller.atomic_json(directory / 'status.json', {'state': 'queued', 'stages': {}}) + called = [] + + def fake_stage(folder, name, _budget): + called.append(name) + status = json.loads((folder / 'status.json').read_text()) + status['stages'][name] = {'result': 'failure' if name == 'apply-k8s' else 'success'} + controller.atomic_json(folder / 'status.json', status) + return name != 'apply-k8s' + + with ( + patch.object(controller, 'STATE', state), + patch.object(controller, 'make_plan', return_value={'selected': {}, 'helm': [], 'removed': []}), + patch.object(controller, 'stage', side_effect=fake_stage), + patch.object(controller, 'command', return_value='1'), + self.assertRaises(RuntimeError), + ): + controller.execute('123-1') + self.assertIn('verify-k8s', called) + self.assertIn('smoke', called) + self.assertNotIn('apply-compose', called) + self.assertFalse((state / 'last-success.json').exists()) + self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure') + + def test_recovery_finishes_baseline_after_all_stages_completed(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + directory.mkdir(parents=True) + names = ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke') + controller.atomic_json( + directory / 'status.json', + {'state': 'running', 'stages': {name: {'result': 'success'} for name in names}}, + ) + controller.atomic_json(directory / 'plan.json', {'sha': 'a' * 40}) + with patch.object(controller, 'STATE', state), patch.object(controller, 'stage') as execute: + controller.recover(directory) + execute.assert_not_called() + self.assertEqual(json.loads((state / 'last-success.json').read_text())['run_id'], '123-1') + self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'success') + + def test_manual_recovery_retries_checks_without_repeating_apply(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + (directory / 'snapshot').mkdir(parents=True) + (directory / 'snapshot/current').write_text('snapshot') + controller.atomic_json( + directory / 'status.json', + { + 'state': 'failure', + 'stages': { + 'apply-k8s': {'result': 'failure'}, + 'verify-k8s': {'result': 'failure'}, + 'smoke': {'result': 'failure'}, + }, + }, + ) + called = [] + + def checks(folder, name, _budget): + status = json.loads((folder / 'status.json').read_text()) + self.assertNotIn(name, status['stages']) + called.append(name) + status['stages'][name] = {'result': 'success'} + controller.atomic_json(folder / 'status.json', status) + return True + + with patch.object(controller, 'STATE', state), patch.object(controller, 'stage', side_effect=checks): + controller.recover(directory, retry=True) + self.assertEqual(called, ['verify-k8s', 'smoke']) + self.assertFalse((state / 'last-success.json').exists()) + self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure') + + +class InstallerTests(unittest.TestCase): + def test_version_comparison_is_exact_without_network_or_host_packages(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + binary = root / 'bin/fake' + binary.parent.mkdir() + binary.write_text('#!/bin/sh\necho fake-v1.7.70\n') + binary.chmod(0o755) + script = """set -euo pipefail +source "$LIB" +if at_version fake 1.7.7; then exit 1; fi +at_version fake 1.7.70 +""" + subprocess.run( # noqa: S603 + ['/usr/bin/bash', '-c', script], + check=True, + env={**os.environ, 'TOOLS_DIR': str(root), 'LIB': str(ROOT / '.gitea/workflows/install-ci-tools.sh')}, + ) + + +if __name__ == '__main__': + unittest.main() From b677d553b4ebe06458edcc9a4fca1a8611a20e73 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 23:22:05 +0200 Subject: [PATCH 29/53] refactor(ci): split checks into visible jobs --- .gitea/runner/README.md | 4 +- .gitea/workflows/ci.yaml | 111 +++++++++++++++++++++++++++++++++++---- 2 files changed, 105 insertions(+), 10 deletions(-) diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md index 803d303..f5b06b1 100644 --- a/.gitea/runner/README.md +++ b/.gitea/runner/README.md @@ -1,7 +1,9 @@ # Homelab CI/CD The native Gitea runner runs on **vps**; production runs on **workstation**. -Jobs run on `homelab:host`, one at a time. No job images or Kubernetes credentials +Compose, workflow, shell, Python, formatting, YAML, Dockerfile and Kubernetes +checks appear as separate jobs. Jobs run on `homelab:host`, one at a time; the +build waits for every check to pass. No job images or Kubernetes credentials are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows. ## Runner installation diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 6bb9910..101778e 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -12,18 +12,13 @@ concurrency: group: ci-${{ github.ref }} cancel-in-progress: ${{ github.ref != 'refs/heads/main' }} jobs: - checks: + compose: + name: Compose runs-on: homelab - timeout-minutes: 30 + timeout-minutes: 15 steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 - - name: Prepare pinned tools - shell: bash - run: | - set -euo pipefail - tools_dir="$(bash .gitea/workflows/install-ci-tools.sh)" - echo "$tools_dir" >> "$GITHUB_PATH" - name: Validate Compose files shell: bash run: | @@ -52,11 +47,37 @@ jobs: exit 1 fi echo "checked ${#files[@]} Compose file(s)" + workflows: + name: Workflows + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint shellcheck)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Lint Gitea Actions workflows with actionlint shell: bash run: | set -euo pipefail actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml + shell: + name: Shell + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Lint shell scripts with ShellCheck shell: bash run: | @@ -70,6 +91,19 @@ jobs: fi shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" bash .gitea/tests/deploy-validation.sh + formatting: + name: Formatting + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Check formatting with Prettier shell: bash run: | @@ -87,6 +121,19 @@ jobs: fi prettier --check --ignore-unknown "${prettier_files[@]}" + python: + name: Python and tests + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff jq)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Lint and format-check Python with Ruff shell: bash run: | @@ -94,6 +141,19 @@ jobs: ruff check . .gitea/workflows ruff format --check . .gitea/workflows python3 -m unittest discover -s tests -v + yaml: + name: YAML + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Lint YAML syntax shell: bash run: | @@ -111,6 +171,19 @@ jobs: fi yamllint -c .yamllint "${yaml_files[@]}" + dockerfiles: + name: Dockerfiles + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Lint Dockerfiles shell: bash run: | @@ -126,6 +199,19 @@ jobs: fi hadolint -c .hadolint.yaml "${dockerfiles[@]}" + kubernetes: + name: Kubernetes + runs-on: homelab + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - name: Prepare pinned tools + shell: bash + run: | + set -euo pipefail + tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)" + echo "$tools_dir" >> "$GITHUB_PATH" - name: Validate Kubernetes manifests against JSON schemas shell: bash run: | @@ -148,7 +234,14 @@ jobs: "${manifests[@]}" build: needs: - - checks + - compose + - workflows + - shell + - formatting + - python + - yaml + - dockerfiles + - kubernetes if: github.event_name != 'pull_request' && github.ref == 'refs/heads/main' runs-on: homelab timeout-minutes: 60 From d74822cd27f71f159836a7043ec34428797527b3 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 23:28:56 +0200 Subject: [PATCH 30/53] Add CI and deploy summaries --- .gitea/runner/README.md | 4 ++- .gitea/workflows/deploy-controller.py | 41 ++++++++++++++++++++++++++- .gitea/workflows/release.py | 21 ++++++++++++++ .gitea/workflows/ssh-run.sh | 26 ++++++++++++++++- 4 files changed, 89 insertions(+), 3 deletions(-) diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md index f5b06b1..91b9046 100644 --- a/.gitea/runner/README.md +++ b/.gitea/runner/README.md @@ -3,7 +3,9 @@ The native Gitea runner runs on **vps**; production runs on **workstation**. Compose, workflow, shell, Python, formatting, YAML, Dockerfile and Kubernetes checks appear as separate jobs. Jobs run on `homelab:host`, one at a time; the -build waits for every check to pass. No job images or Kubernetes credentials +build waits for every check to pass. CI and deploy runs also show a summary with +the release SHA, image build or reuse results, deploy mode, selected services, +and image digests. No job images or Kubernetes credentials are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows. ## Runner installation diff --git a/.gitea/workflows/deploy-controller.py b/.gitea/workflows/deploy-controller.py index 432f672..4bf80f5 100644 --- a/.gitea/workflows/deploy-controller.py +++ b/.gitea/workflows/deploy-controller.py @@ -294,10 +294,47 @@ def follow(run_id, phase): time.sleep(3) +def summary(run_id): + directory = run_directory(run_id) + request = json.loads((directory / 'request.json').read_text()) + release = request['release'] + plan_file = directory / 'plan.json' + lines = [ + f'## Deploy `{release["sha"]}`', + '', + f'- Mode: `{request["mode"]}`', + f'- Refresh third-party images: `{request["refresh_images"]}`', + ] + if not plan_file.exists(): + lines.extend(['', 'Plan was not created. Check the controller log.']) + print('\n'.join(lines)) + return + plan = json.loads(plan_file.read_text()) + lines.extend(['', '### Selected services']) + count = 0 + for kind, services in plan['selected'].items(): + for service in services: + lines.append(f'- `{kind}`: `{service}`') + count += 1 + if not count: + lines.append('- None') + lines.extend(['', '### Selected Helm releases']) + lines.extend(f'- `{release}`' for release in plan.get('helm', [])) + if not plan.get('helm'): + lines.append('- None') + lines.extend(['', '### Images pinned in the checked release']) + lines.extend(f'- `{image}@{digest}`' for image, digest in sorted(release['images'].items())) + lines.extend(['', '### Removed resources requiring manual review']) + lines.extend(f'- `{item}`' for item in plan.get('removed', [])) + if not plan.get('removed'): + lines.append('- None') + print('\n'.join(lines)) + + def main(): os.umask(0o077) parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow')) + parser.add_argument('action', choices=('start', 'execute', 'recover', 'status', 'follow', 'summary')) parser.add_argument('run_id') parser.add_argument('phase', nargs='?', choices=('apply', 'verify', 'smoke')) parser.add_argument('--retry', action='store_true', help='Retry failed recovery checks; never repeat apply') @@ -315,6 +352,8 @@ def main(): if (directory / 'plan.json').exists(): plan = json.loads((directory / 'plan.json').read_text()) print(json.dumps({k: plan[k] for k in ('sha', 'selected', 'helm', 'removed')}, indent=2)) + elif args.action == 'summary': + summary(args.run_id) elif not follow(args.run_id, args.phase): sys.exit(1) diff --git a/.gitea/workflows/release.py b/.gitea/workflows/release.py index daff5aa..cc9e9bd 100644 --- a/.gitea/workflows/release.py +++ b/.gitea/workflows/release.py @@ -231,6 +231,8 @@ def build(output): ) signature.write_text(image + '\n') release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}} + built = [] + reused = [] for name, (context, dockerfile) in IMAGES.items(): image = f'gcr.forust.xyz/forust/{name}' inputs = fingerprint(context, dockerfile) @@ -255,8 +257,10 @@ def build(output): if exists: print(f'Reuse {name}: inputs unchanged') digest = old_digest + reused.append((name, image, digest)) else: print(f'Build {name}', flush=True) + built.append((name, image)) metadata = Path(docker_config) / 'metadata.json' command( 'docker', @@ -286,6 +290,23 @@ def build(output): release['inputs'][image] = inputs validate_release(release, sha) output.write_text(json.dumps(release, indent=2) + '\n') + summary = os.environ.get('GITHUB_STEP_SUMMARY') + if summary: + lines = [f'## Image release for `{sha}`', '', '### Built'] + lines.extend(f'- `{name}` — `{image}`' for name, image in built) + if not built: + lines.append('- None') + lines.extend(['', '### Reused from successful CI']) + lines.extend(f'- `{name}` — `{image}@{digest}`' for name, image, digest in reused) + if not reused: + lines.append('- None') + lines.extend(['', '### Release digests']) + lines.extend( + f'- `{name}` — `{image}@{release["images"][image]}`' + for name in IMAGES + for image in [f'gcr.forust.xyz/forust/{name}'] + ) + Path(summary).write_text('\n'.join(lines) + '\n') finally: # Cleanup errors must neither leak credentials nor mask the original build error. try: diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index cfc94dc..8caecfa 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -40,7 +40,31 @@ PY done exit "$rc" ;; - apply|verify|smoke) + apply) + result=0 + for attempt in 1 2 3; do + rc=0 + # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" apply || rc=$? + [ "$rc" -eq 0 ] && break + [ "$rc" -eq 255 ] || { result="$rc"; break; } + echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" + if [ "$attempt" -eq 3 ]; then result=255; break; fi + sleep 5 + done + if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + rc=0 + # shellcheck disable=SC2029 # The run ID is validated above. + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" summary "$DEPLOY_RUN_ID" >"$key_dir/deploy-summary.md" || rc=$? + if [ "$rc" -eq 0 ]; then + cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" + else + echo 'Deploy summary is unavailable. Check the controller log.' >>"$GITHUB_STEP_SUMMARY" + fi + fi + exit "$result" + ;; + verify|smoke) for attempt in 1 2 3; do rc=0 # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. From 0c76426c17e1eabc1996a3c2e72968313bbc85e1 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 23:44:00 +0200 Subject: [PATCH 31/53] Show failure details in CI and deploy summaries --- .gitea/runner/README.md | 7 +- .gitea/workflows/ci.yaml | 161 ++++++++++++++++++++++++++ .gitea/workflows/deploy-controller.py | 60 ++++++++++ .gitea/workflows/deploy-lib.sh | 37 +++++- .gitea/workflows/deploy.yaml | 36 ++++++ .gitea/workflows/release.py | 92 +++++++++++---- .gitea/workflows/ssh-run.sh | 26 ++--- tests/test_cicd_lifecycle.py | 58 ++++++++++ 8 files changed, 432 insertions(+), 45 deletions(-) diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md index 91b9046..bd203a2 100644 --- a/.gitea/runner/README.md +++ b/.gitea/runner/README.md @@ -5,7 +5,12 @@ Compose, workflow, shell, Python, formatting, YAML, Dockerfile and Kubernetes checks appear as separate jobs. Jobs run on `homelab:host`, one at a time; the build waits for every check to pass. CI and deploy runs also show a summary with the release SHA, image build or reuse results, deploy mode, selected services, -and image digests. No job images or Kubernetes credentials +and image digests. Failed runs keep a summary of completed image builds, stage +results, apply results, and recorded Kubernetes recovery. The final deploy +summary is in the smoke job; earlier jobs show the state observed at that time. +Apply success is separate from health and recovery. Update the installed +workstation controller with `setup-workstation.sh` when no deploy is running. +No job images or Kubernetes credentials are needed on the VPS. Builds use one pinned BuildKit helper container. CI and deploy are separate workflows. ## Runner installation diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 101778e..84f63d3 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -19,6 +19,7 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Validate Compose files shell: bash run: | @@ -47,6 +48,22 @@ jobs: exit 1 fi echo "checked ${#files[@]} Compose file(s)" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Compose + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.source.conclusion == 'failure' + && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi workflows: name: Workflows runs-on: homelab @@ -54,17 +71,35 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh actionlint shellcheck)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint Gitea Actions workflows with actionlint shell: bash run: | set -euo pipefail actionlint -config-file .gitea/actionlint.yaml -color .gitea/workflows/*.yaml + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Workflows + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi shell: name: Shell runs-on: homelab @@ -72,12 +107,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh shellcheck jq)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint shell scripts with ShellCheck shell: bash run: | @@ -91,6 +128,22 @@ jobs: fi shellcheck --external-sources --source-path=SCRIPTDIR --severity=style "${scripts[@]}" bash .gitea/tests/deploy-validation.sh + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Shell + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi formatting: name: Formatting runs-on: homelab @@ -98,12 +151,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh prettier)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Check formatting with Prettier shell: bash run: | @@ -121,6 +176,22 @@ jobs: fi prettier --check --ignore-unknown "${prettier_files[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Formatting + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi python: name: Python and tests runs-on: homelab @@ -128,12 +199,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh ruff jq)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint and format-check Python with Ruff shell: bash run: | @@ -141,6 +214,22 @@ jobs: ruff check . .gitea/workflows ruff format --check . .gitea/workflows python3 -m unittest discover -s tests -v + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Python and tests + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi yaml: name: YAML runs-on: homelab @@ -148,12 +237,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh yamllint)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint YAML syntax shell: bash run: | @@ -171,6 +262,22 @@ jobs: fi yamllint -c .yamllint "${yaml_files[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: YAML + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi dockerfiles: name: Dockerfiles runs-on: homelab @@ -178,12 +285,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh hadolint)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Lint Dockerfiles shell: bash run: | @@ -199,6 +308,22 @@ jobs: fi hadolint -c .hadolint.yaml "${dockerfiles[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Dockerfiles + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi kubernetes: name: Kubernetes runs-on: homelab @@ -206,12 +331,14 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + id: source - name: Prepare pinned tools shell: bash run: | set -euo pipefail tools_dir="$(bash .gitea/workflows/install-ci-tools.sh kubeconform)" echo "$tools_dir" >> "$GITHUB_PATH" + id: tools - name: Validate Kubernetes manifests against JSON schemas shell: bash run: | @@ -232,6 +359,22 @@ jobs: -ignore-missing-schemas \ -summary \ "${manifests[@]}" + id: check + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Kubernetes + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.tools.conclusion == 'failure' + && 'Tool setup' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi build: needs: - compose @@ -250,12 +393,14 @@ jobs: uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 with: fetch-depth: 0 + id: source - name: Build changed images and write release env: GITEA_TOKEN: ${{ github.token }} REGISTRY_USERNAME: ${{ secrets.REGISTRY_USERNAME }} REGISTRY_PASSWORD: ${{ secrets.REGISTRY_PASSWORD }} run: python3 .gitea/workflows/release.py build + id: check - name: Store commit release uses: actions/upload-artifact@c6a366c94c3e0affe28c06c8df20a878f24da3cf with: @@ -263,3 +408,19 @@ jobs: path: release.json if-no-files-found: error retention-days: 30 + id: artifact + - name: Write the job result + if: always() + env: + SUMMARY_CHECK: Image build and release artifact + SUMMARY_RESULT: ${{ job.status }} + SUMMARY_FAILED_STEP: + ${{ steps.check.conclusion == 'failure' && 'Check or image build' || steps.artifact.conclusion == 'failure' + && 'Release artifact upload' || steps.source.conclusion == 'failure' && 'Source checkout' || '' }} + shell: bash + run: | + if [ -f .gitea/workflows/release.py ]; then + python3 .gitea/workflows/release.py check-summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## %s\n\n- Result: **%s**\n- Failed step: %s\n' "$SUMMARY_CHECK" "$SUMMARY_RESULT" "$SUMMARY_FAILED_STEP" >>"$GITHUB_STEP_SUMMARY" || true + fi diff --git a/.gitea/workflows/deploy-controller.py b/.gitea/workflows/deploy-controller.py index 4bf80f5..be51277 100644 --- a/.gitea/workflows/deploy-controller.py +++ b/.gitea/workflows/deploy-controller.py @@ -212,14 +212,17 @@ def execute(run_id): raise ValueError(f'Interrupted deploy {other.name}; run recover first') status['state'] = 'running' atomic_json(directory / 'status.json', status) + phase = 'plan' try: plan = make_plan(directory) print( json.dumps({'selected': plan['selected'], 'helm': plan['helm'], 'manual_removals': plan['removed']}), flush=True, ) + phase = 'doctor' if not stage(directory, 'doctor', 600): raise RuntimeError('Preflight failed') + phase = 'validate' if not stage(directory, 'validate', 1200): raise RuntimeError('Validation failed') if json.loads((directory / 'request.json').read_text())['mode'] == 'plan': @@ -228,6 +231,7 @@ def execute(run_id): atomic_json(directory / 'status.json', status) return # Budget includes both rollout checks and rollback waves, plus API overhead. + phase = 'Recovery budget' count = int( command( 'bash', @@ -239,14 +243,24 @@ def execute(run_id): verify_budget = max(600, 2 * math.ceil(count / 4) * 300 + 120) if verify_budget > 7200: raise ValueError('More than two hours of recovery required; split this deploy') + phase = 'apply-k8s' k8s_ok = stage(directory, 'apply-k8s', 2700) + phase = 'apply-compose' compose_ok = stage(directory, 'apply-compose', 1800) if k8s_ok else False + phase = 'verify-k8s' verify_ok = stage(directory, 'verify-k8s', verify_budget) + phase = 'smoke' smoke_ok = stage(directory, 'smoke', 600) if not all((k8s_ok, compose_ok, verify_ok, smoke_ok)): raise RuntimeError('Deploy failed; inspect stage logs and recovery report') + phase = 'Save the successful baseline' finish_success(directory, plan) except Exception as error: + status = json.loads((directory / 'status.json').read_text()) + status['failure_stage'] = next( + (name for name, result in status['stages'].items() if result.get('result') == 'failure'), phase + ) + atomic_json(directory / 'status.json', status) with (directory / 'controller.log').open('a') as stream: stream.write(f'{error}\n') recover(directory) @@ -305,6 +319,52 @@ def summary(run_id): f'- Mode: `{request["mode"]}`', f'- Refresh third-party images: `{request["refresh_images"]}`', ] + status = json.loads((directory / 'status.json').read_text()) + if status.get('failure_stage'): + lines.append(f'- Failed stage: **{status["failure_stage"]}**') + lines.extend( + [ + '', + f'- Observed run state: **{status["state"]}**', + '', + '### Stage results', + '| Stage | Result | Exit code |', + '| --- | --- | --- |', + ] + ) + for name in ('doctor', 'validate', 'apply-k8s', 'apply-compose', 'verify-k8s', 'smoke'): + stage_result = status['stages'].get(name, {}) + lines.append(f'| {name} | {stage_result.get("result", "not started")} | {stage_result.get("exit_code", "—")} |') + lines.extend(['', '### Apply and Helm recovery results']) + events_file = directory / 'apply-events.jsonl' + events = [] + if events_file.exists(): + for line in events_file.read_text().splitlines(): + try: + events.append(json.loads(line)) + except json.JSONDecodeError: + lines.append('- An operation record is incomplete. Check the stage log.') + latest = {(event['action'], event['target']): event['result'] for event in events} + lines.extend(f'- `{action}` `{target}`: **{result}**' for (action, target), result in latest.items()) + if not latest: + lines.append('- No apply results were recorded.') + lines.append('- A completed apply does not confirm health. See verification and smoke results.') + lines.extend(['', '### Kubernetes recovery']) + pointer = directory / 'snapshot/current' + failed = Path(pointer.read_text().strip()) / 'failed-workloads' if pointer.exists() else None + if failed and failed.exists(): + contents = failed.read_text() + counts = dict(re.findall(r'^(ROLLED_BACK|UNRECOVERED)=([0-9]+)$', contents, re.MULTILINE)) + if not contents.strip(): + lines.append('- No failed workloads were recorded. See the verification result above.') + elif counts: + lines.append(f'- Workloads restored: **{counts.get("ROLLED_BACK", "unknown")}**') + lines.append(f'- Workloads that need manual recovery: **{counts.get("UNRECOVERED", "unknown")}**') + else: + lines.append('- Rollback has no recorded result yet. Check the verification log.') + else: + lines.append('- No workload rollback was recorded. This does not confirm health.') + lines.append('- Compose requires manual recovery. Use the saved command in the apply log.') if not plan_file.exists(): lines.extend(['', 'Plan was not created. Check the controller log.']) print('\n'.join(lines)) diff --git a/.gitea/workflows/deploy-lib.sh b/.gitea/workflows/deploy-lib.sh index 8731b90..58f04e7 100644 --- a/.gitea/workflows/deploy-lib.sh +++ b/.gitea/workflows/deploy-lib.sh @@ -27,6 +27,15 @@ log() { echo "== $* ==" } +# Store operation results without command output or local configuration values. +record_apply() { + [ -n "${RUN_DIR:-}" ] || return 0 + jq -cn --arg action "$1" --arg target "$2" --arg result "$3" \ + '{action: $action, target: $target, result: $result}' >>"$RUN_DIR/apply-events.jsonl" \ + || echo 'WARNING: cannot record an apply result' >&2 + return 0 +} + warn() { echo "WARNING: $*" >&2 } @@ -376,15 +385,19 @@ recover_pending_release() { echo "ERROR: no captured Helm revision for $release; manual recovery required" return 1 fi + record_apply helm-rollback "$namespace/$release" started if ! helm rollback "$release" "$revision" -n "$namespace" --wait --timeout 10m; then + record_apply helm-rollback "$namespace/$release" failure echo "WARN: helm rollback of $release did not complete" return 1 fi status="$(helm_release_status "$release" "$namespace")" || return 1 if [ "$status" != "deployed" ]; then + record_apply helm-rollback "$namespace/$release" failure echo "WARN: $release is $status after rollback" return 1 fi + record_apply helm-rollback "$namespace/$release" success ;; esac return 0 @@ -448,11 +461,13 @@ upgrade_helm_releases() { # --rollback-on-failure (+ --wait) rolls the release back when the upgrade # times out or the workloads it touches never become ready, so a bad chart # bump is not left half applied. (--atomic was this combo; deprecated.) + record_apply helm-upgrade "$namespace/$release" started if ! helm upgrade --install "$release" "$chart" \ --namespace "$namespace" \ --version "$version" \ --values "$values" \ --wait --rollback-on-failure --cleanup-on-fail --timeout 10m; then + record_apply helm-upgrade "$namespace/$release" failure echo "WARN: upgrade of $release failed, checking release state" # --rollback-on-failure already attempted its own rollback; finish the job when that # rollback never completed, otherwise the release stays pending-* and @@ -462,8 +477,10 @@ upgrade_helm_releases() { else echo "ERROR: upgrade of $release failed (release is back on its previous revision)." fi + record_apply helm-recovery-state "$namespace/$release" "$(helm_release_status "$release" "$namespace" || echo unknown)" return 1 fi + record_apply helm-upgrade "$namespace/$release" success done } @@ -631,7 +648,12 @@ stage_apply_k8s() { if [ "${#ns_files[@]}" -gt 0 ]; then log "Applying namespaces (${#ns_files[@]} files)" for m in "${ns_files[@]}"; do - kubectl apply -f "$m" + record_apply kubectl "${m#"$REPO"/}" started + if ! kubectl apply -f "$m"; then + record_apply kubectl "${m#"$REPO"/}" failure + return 1 + fi + record_apply kubectl "${m#"$REPO"/}" success done fi if selected_service k8s prometheus-stack && [ -f "$REPO/prometheus-stack/k8s/active" ]; then @@ -646,18 +668,24 @@ stage_apply_k8s() { log "Applying resources (${#other_files[@]} files, our images pinned to digests)" for m in "${other_files[@]}"; do log "Applying ${m#"$REPO"/}" + record_apply kubectl "${m#"$REPO"/}" started if ! render_pinned <"$m" | kubectl apply -f -; then + record_apply kubectl "${m#"$REPO"/}" failure echo "ERROR: apply failed for ${m#"$REPO"/}" >&2 exit 1 fi + record_apply kubectl "${m#"$REPO"/}" success done fi for k in ${KUSTOMIZE_APPS[@]+"${KUSTOMIZE_APPS[@]}"}; do log "Applying kustomize app: ${k#"$REPO"/} (our images pinned to digests)" + record_apply kustomize "${k#"$REPO"/}" started if ! kubectl kustomize "$k" | render_pinned | kubectl apply -f -; then + record_apply kustomize "${k#"$REPO"/}" failure echo "ERROR: apply failed for kustomize app ${k#"$REPO"/}" >&2 exit 1 fi + record_apply kustomize "${k#"$REPO"/}" success done # No verification here on purpose. This stage may be killed at any point by @@ -959,7 +987,12 @@ stage_apply_compose() { local cf for cf in "${COMPOSE_STACKS[@]}"; do log "Applying Compose ${cf#"$REPO"/}" - compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans + record_apply compose "${cf#"$REPO"/}" started + if ! compose "$cf" up -d --wait --wait-timeout 180 --pull missing --remove-orphans; then + record_apply compose "${cf#"$REPO"/}" failure + return 1 + fi + record_apply compose "${cf#"$REPO"/}" success verify_compose_stack "$cf" done echo "Compose recovery files: $RUN_DIR/compose-before (manual recovery only)" diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 2ac1f60..900280c 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -64,6 +64,18 @@ jobs: run: python3 .gitea/workflows/release.py gate --ref "$DEPLOY_REF" --event-sha "$EVENT_SHA" - name: Submit durable deploy to workstation run: bash .gitea/workflows/ssh-run.sh start + - name: Write the request result + if: always() + env: + REQUEST_RESULT: ${{ job.status }} + CHECKED_SHA: ${{ steps.release.outputs.sha }} + run: | + if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + printf '## Deploy request\n\n- Result: **%s**\n- Checked commit: `%s`\n- Mode: `%s`\n' "$REQUEST_RESULT" "${CHECKED_SHA:-not checked}" "$DEPLOY_MODE" >>"$GITHUB_STEP_SUMMARY" + if [ "$REQUEST_RESULT" != success ]; then + echo 'Open the failed step log. If SSH submission failed, check the remote controller state.' >>"$GITHUB_STEP_SUMMARY" + fi + fi apply: needs: [gate] @@ -76,6 +88,14 @@ jobs: ref: ${{ needs.gate.outputs.sha }} - name: Follow validation and sequential Kubernetes / Compose apply run: bash .gitea/workflows/ssh-run.sh apply + - name: Write the deploy result + if: always() + run: | + if [ -f .gitea/workflows/ssh-run.sh ]; then + bash .gitea/workflows/ssh-run.sh summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY" + fi verify: needs: [gate, apply] @@ -89,6 +109,14 @@ jobs: ref: ${{ needs.gate.outputs.sha }} - name: Follow workload verification and recovery run: bash .gitea/workflows/ssh-run.sh verify + - name: Write the deploy result + if: always() + run: | + if [ -f .gitea/workflows/ssh-run.sh ]; then + bash .gitea/workflows/ssh-run.sh summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY" + fi smoke: needs: [gate, verify] @@ -102,3 +130,11 @@ jobs: ref: ${{ needs.gate.outputs.sha }} - name: Follow public route checks run: bash .gitea/workflows/ssh-run.sh smoke + - name: Write the deploy result + if: always() + run: | + if [ -f .gitea/workflows/ssh-run.sh ]; then + bash .gitea/workflows/ssh-run.sh summary + elif [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + echo 'Source checkout failed. The remote deploy state is unknown. Check the job log.' >>"$GITHUB_STEP_SUMMARY" + fi diff --git a/.gitea/workflows/release.py b/.gitea/workflows/release.py index cc9e9bd..00d2a79 100644 --- a/.gitea/workflows/release.py +++ b/.gitea/workflows/release.py @@ -163,10 +163,11 @@ def gate(output, requested_ref, event_sha): print(f'CI gate accepted {sha}') -def build(output): +def build_images(output, report): sha = command('git', 'rev-parse', 'HEAD') if sha != os.environ['GITHUB_SHA'] or not SHA.fullmatch(sha): raise ValueError('Build checkout does not match GITHUB_SHA') + report['phase'] = 'Find a successful CI release' api = Gitea() previous = None for run in sorted(itertools.islice(api.successful_runs(), 50), key=lambda item: item['id'], reverse=True): @@ -183,6 +184,7 @@ def build(output): builder_config.mkdir(parents=True, exist_ok=True) env = {**os.environ, 'DOCKER_CONFIG': docker_config, 'BUILDX_CONFIG': str(builder_config)} try: + report['phase'] = 'Registry login' subprocess.run( # noqa: S603, S607 [ shutil.which('docker') or '/usr/bin/docker', @@ -197,6 +199,7 @@ def build(output): check=True, env=env, ) + report['phase'] = 'Prepare the builder' builder = 'homelab-ci' versions = dict( re.findall(r'^([A-Z_]+)="([^"\n]+)"$', Path('.gitea/workflows/tool-versions.env').read_text(), re.MULTILINE) @@ -231,9 +234,10 @@ def build(output): ) signature.write_text(image + '\n') release = {'version': 1, 'sha': sha, 'images': {}, 'inputs': {}} - built = [] - reused = [] + report['images'] = release['images'] for name, (context, dockerfile) in IMAGES.items(): + report['phase'] = f'Build or reuse {name}' + report['current'] = name image = f'gcr.forust.xyz/forust/{name}' inputs = fingerprint(context, dockerfile) old_digest = (previous or {}).get('images', {}).get(image) @@ -257,10 +261,9 @@ def build(output): if exists: print(f'Reuse {name}: inputs unchanged') digest = old_digest - reused.append((name, image, digest)) + report['reused'].append(name) else: print(f'Build {name}', flush=True) - built.append((name, image)) metadata = Path(docker_config) / 'metadata.json' command( 'docker', @@ -286,27 +289,13 @@ def build(output): env=env, ) digest = json.loads(metadata.read_text())['containerimage.digest'] + report['built'].append(name) release['images'][image] = digest release['inputs'][image] = inputs validate_release(release, sha) output.write_text(json.dumps(release, indent=2) + '\n') - summary = os.environ.get('GITHUB_STEP_SUMMARY') - if summary: - lines = [f'## Image release for `{sha}`', '', '### Built'] - lines.extend(f'- `{name}` — `{image}`' for name, image in built) - if not built: - lines.append('- None') - lines.extend(['', '### Reused from successful CI']) - lines.extend(f'- `{name}` — `{image}@{digest}`' for name, image, digest in reused) - if not reused: - lines.append('- None') - lines.extend(['', '### Release digests']) - lines.extend( - f'- `{name}` — `{image}@{release["images"][image]}`' - for name in IMAGES - for image in [f'gcr.forust.xyz/forust/{name}'] - ) - Path(summary).write_text('\n'.join(lines) + '\n') + report['current'] = None + report['phase'] = 'Release file saved' finally: # Cleanup errors must neither leak credentials nor mask the original build error. try: @@ -330,6 +319,59 @@ def build(output): shutil.rmtree(docker_config) +def write_summary(lines): + path = os.environ.get('GITHUB_STEP_SUMMARY') + if path: + try: + with Path(path).open('a') as stream: + stream.write('\n'.join(lines) + '\n\n') + except OSError: + print('WARNING: cannot write the job summary') + + +def check_summary(): + lines = [ + f'## {os.environ["SUMMARY_CHECK"]}', + '', + f'- Commit: `{os.environ.get("GITHUB_SHA", "unknown")}`', + f'- Result: **{os.environ["SUMMARY_RESULT"]}**', + ] + if os.environ.get('SUMMARY_FAILED_STEP'): + lines.append(f'- Failed step: {os.environ["SUMMARY_FAILED_STEP"]}') + if os.environ['SUMMARY_RESULT'] != 'success': + lines.append('- Open the failed step log for the error details.') + write_summary(lines) + + +def build(output): + report = {'phase': 'Check the source commit', 'current': None, 'built': [], 'reused': [], 'images': {}} + result = 'failure' + try: + build_images(output, report) + result = 'success' + finally: + lines = [ + f'## Image release `{os.environ.get("GITHUB_SHA", "unknown")}`', + '', + f'- Result: **{result}**', + f'- Last stage: {report["phase"]}', + ] + if result == 'failure': + lines.append('- No release from this build can be deployed. Open the failed step log.') + if report['current']: + lines.append(f'- Image at the failure: `{report["current"]}`') + for title, key in (('Built', 'built'), ('Reused from successful CI', 'reused')): + lines.extend(['', f'### {title}']) + lines.extend(f'- `{name}`' for name in report[key]) + if not report[key]: + lines.append('- None') + lines.extend(['', '### Completed image digests']) + lines.extend(f'- `{image}@{digest}`' for image, digest in report['images'].items()) + if not report['images']: + lines.append('- None') + write_summary(lines) + + def render(stream, destination): release = validate_release(json.loads(Path(os.environ['RELEASE_FILE']).read_text()), os.environ['DEPLOY_SHA']) image_line = re.compile( @@ -354,12 +396,14 @@ def render(stream, destination): def main(): parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument('action', choices=('build', 'gate', 'render')) + parser.add_argument('action', choices=('build', 'gate', 'render', 'check-summary')) parser.add_argument('--output', type=Path, default=Path('release.json')) parser.add_argument('--ref', default='main') parser.add_argument('--event-sha', default='') args = parser.parse_args() - if args.action == 'render': + if args.action == 'check-summary': + check_summary() + elif args.action == 'render': render(sys.stdin, sys.stdout) elif args.action == 'gate': gate(args.output, args.ref, args.event_sha) diff --git a/.gitea/workflows/ssh-run.sh b/.gitea/workflows/ssh-run.sh index 8caecfa..7f29be4 100755 --- a/.gitea/workflows/ssh-run.sh +++ b/.gitea/workflows/ssh-run.sh @@ -20,7 +20,7 @@ ssh_opts=(-i "$key_dir/key" -p "${DEPLOY_PORT:-22}" -o BatchMode=yes -o StrictHo -o "UserKnownHostsFile=$key_dir/known_hosts" -o ConnectTimeout=15 -o ServerAliveInterval=15 -o ServerAliveCountMax=4) controller=.local/lib/homelab-deploy/controller.py -case "${1:?start, apply, verify or smoke required}" in +case "${1:?start, apply, verify, smoke or summary required}" in start) python3 - <<'PY' >"$key_dir/request.json" import json @@ -40,41 +40,31 @@ PY done exit "$rc" ;; - apply) + apply|verify|smoke) result=0 for attempt in 1 2 3; do rc=0 # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. - ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" apply || rc=$? + ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$? [ "$rc" -eq 0 ] && break [ "$rc" -eq 255 ] || { result="$rc"; break; } echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" if [ "$attempt" -eq 3 ]; then result=255; break; fi sleep 5 done + exit "$result" + ;; + summary) if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then rc=0 # shellcheck disable=SC2029 # The run ID is validated above. ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" summary "$DEPLOY_RUN_ID" >"$key_dir/deploy-summary.md" || rc=$? if [ "$rc" -eq 0 ]; then - cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" + cat "$key_dir/deploy-summary.md" >>"$GITHUB_STEP_SUMMARY" || echo "WARNING: cannot write the deploy summary" else - echo 'Deploy summary is unavailable. Check the controller log.' >>"$GITHUB_STEP_SUMMARY" + echo 'Deploy summary is unavailable. The SSH connection failed or the controller did not respond. Check the job log.' >>"$GITHUB_STEP_SUMMARY" || true fi fi - exit "$result" - ;; - verify|smoke) - for attempt in 1 2 3; do - rc=0 - # shellcheck disable=SC2029 # The run ID and operation are validated local arguments, not remote variables. - ssh "${ssh_opts[@]}" "$DEPLOY_USER@$DEPLOY_HOST" python3 "$controller" follow "$DEPLOY_RUN_ID" "$1" || rc=$? - [ "$rc" -eq 0 ] && exit 0 - [ "$rc" -eq 255 ] || exit "$rc" - echo "SSH disconnected; reconnecting to the existing deploy ($attempt/3)" - sleep 5 - done - exit "$rc" ;; *) echo "Unknown SSH operation: $1" >&2; exit 1 ;; esac diff --git a/tests/test_cicd_lifecycle.py b/tests/test_cicd_lifecycle.py index f14caab..8f332c2 100644 --- a/tests/test_cicd_lifecycle.py +++ b/tests/test_cicd_lifecycle.py @@ -213,6 +213,64 @@ class DurableRunTests(unittest.TestCase): self.assertEqual(json.loads((directory / 'status.json').read_text())['state'], 'failure') +class FailureSummaryTests(unittest.TestCase): + def test_build_failure_keeps_progress_and_does_not_expose_exception_text(self): + with tempfile.TemporaryDirectory() as scratch: + summary = Path(scratch) / 'summary.md' + + def failed_build(_output, report): + report.update(phase='Build or reuse xdfnx-homepage', current='xdfnx-homepage', built=['error-pages']) + report['images']['gcr.forust.xyz/forust/error-pages'] = 'sha256:' + 'b' * 64 + raise RuntimeError('private value must not appear in the summary') + + with ( + patch.dict(os.environ, {'GITHUB_STEP_SUMMARY': str(summary), 'GITHUB_SHA': 'a' * 40}), + patch.object(release_module, 'build_images', side_effect=failed_build), + self.assertRaises(RuntimeError), + ): + release_module.build(Path(scratch) / 'release.json') + content = summary.read_text() + self.assertIn('**failure**', content) + self.assertIn('error-pages', content) + self.assertIn('xdfnx-homepage', content) + self.assertNotIn('private value', content) + + def test_deploy_failure_reports_completed_apply_and_rollback_result(self): + with tempfile.TemporaryDirectory() as scratch: + state = Path(scratch) + directory = state / 'runs/123-1' + snapshot = directory / 'snapshot/before' + snapshot.mkdir(parents=True) + (directory / 'snapshot/current').write_text(str(snapshot)) + (snapshot / 'failed-workloads').write_text('deployment app api\nROLLED_BACK=1\nUNRECOVERED=0\n') + controller.atomic_json( + directory / 'request.json', {'release': release(), 'mode': 'changed', 'refresh_images': False} + ) + controller.atomic_json( + directory / 'status.json', + { + 'state': 'failure', + 'stages': { + 'apply-k8s': {'result': 'success', 'exit_code': 0}, + 'verify-k8s': {'result': 'failure', 'exit_code': 1}, + }, + }, + ) + controller.atomic_json(directory / 'plan.json', {'selected': {'k8s': ['app'], 'compose': []}}) + (directory / 'apply-events.jsonl').write_text( + json.dumps({'action': 'kubectl', 'target': 'app/k8s/api.yaml', 'result': 'success'}) + '\n' + ) + output = io.StringIO() + with patch.object(controller, 'STATE', state), patch('sys.stdout', output): + controller.summary('123-1') + content = output.getvalue() + self.assertIn('verify-k8s | failure | 1', content) + self.assertIn('app/k8s/api.yaml', content) + self.assertIn('Workloads restored: **1**', content) + self.assertIn('manual recovery: **0**', content) + self.assertIn('Compose requires manual recovery', content) + + class InstallerTests(unittest.TestCase): def test_version_comparison_is_exact_without_network_or_host_packages(self): with tempfile.TemporaryDirectory() as scratch: From fb400eea6ed2b9dfa4b440edb655eddbfdfcb5c7 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Tue, 6 Oct 2026 23:44:24 +0200 Subject: [PATCH 32/53] Use plain text for the deploy request details --- .gitea/workflows/deploy.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.gitea/workflows/deploy.yaml b/.gitea/workflows/deploy.yaml index 900280c..60dc0ee 100644 --- a/.gitea/workflows/deploy.yaml +++ b/.gitea/workflows/deploy.yaml @@ -71,7 +71,7 @@ jobs: CHECKED_SHA: ${{ steps.release.outputs.sha }} run: | if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then - printf '## Deploy request\n\n- Result: **%s**\n- Checked commit: `%s`\n- Mode: `%s`\n' "$REQUEST_RESULT" "${CHECKED_SHA:-not checked}" "$DEPLOY_MODE" >>"$GITHUB_STEP_SUMMARY" + printf '## Deploy request\n\n- Result: **%s**\n- Checked commit: %s\n- Mode: %s\n' "$REQUEST_RESULT" "${CHECKED_SHA:-not checked}" "$DEPLOY_MODE" >>"$GITHUB_STEP_SUMMARY" if [ "$REQUEST_RESULT" != success ]; then echo 'Open the failed step log. If SSH submission failed, check the remote controller state.' >>"$GITHUB_STEP_SUMMARY" fi From 1d93588e06d3e60e6a8c5f654b873e8d6f7cd470 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Wed, 7 Oct 2026 10:19:12 +0200 Subject: [PATCH 33/53] refactor: remove EDU ownership from homelab --- .gitea/EDU_HANDOFF.md | 55 + .gitea/runner/README.md | 3 +- .gitignore | 1 - edu_master/.env.example | 14 - edu_master/PLAYWRIGHT_VERSION | 1 - edu_master/README.md | 14 - edu_master/compose.yaml | 49 - edu_master/k8s/active | 0 edu_master/k8s/alerts.yaml | 115 -- edu_master/k8s/namespace.yaml | 4 - edu_master/k8s/playwright.yaml | 69 - edu_master/k8s/redis-networkpolicy.yaml | 22 - edu_master/k8s/redis.yaml | 96 - edu_master/k8s/restore-seed-job.yaml.example | 50 - edu_master/k8s/secrets.yaml.example | 29 - edu_master/k8s/service.yaml | 15 - edu_master/k8s/servicemonitor.yaml | 16 - edu_master/k8s/session-keeper.yaml | 69 - edu_master/k8s/webinar-checker.yaml | 87 - edu_master/phpsessid-bot/Dockerfile | 15 - edu_master/phpsessid-bot/bot.py | 132 -- edu_master/webinar-checker/Dockerfile | 13 - edu_master/webinar-checker/checker.py | 1821 ------------------ renovate/k8s/configmap.yaml | 36 - renovate/renovate.json | 36 - 25 files changed, 56 insertions(+), 2706 deletions(-) create mode 100644 .gitea/EDU_HANDOFF.md delete mode 100644 edu_master/.env.example delete mode 100644 edu_master/PLAYWRIGHT_VERSION delete mode 100644 edu_master/README.md delete mode 100644 edu_master/compose.yaml delete mode 100644 edu_master/k8s/active delete mode 100644 edu_master/k8s/alerts.yaml delete mode 100644 edu_master/k8s/namespace.yaml delete mode 100644 edu_master/k8s/playwright.yaml delete mode 100644 edu_master/k8s/redis-networkpolicy.yaml delete mode 100644 edu_master/k8s/redis.yaml delete mode 100644 edu_master/k8s/restore-seed-job.yaml.example delete mode 100644 edu_master/k8s/secrets.yaml.example delete mode 100644 edu_master/k8s/service.yaml delete mode 100644 edu_master/k8s/servicemonitor.yaml delete mode 100644 edu_master/k8s/session-keeper.yaml delete mode 100644 edu_master/k8s/webinar-checker.yaml delete mode 100644 edu_master/phpsessid-bot/Dockerfile delete mode 100644 edu_master/phpsessid-bot/bot.py delete mode 100644 edu_master/webinar-checker/Dockerfile delete mode 100644 edu_master/webinar-checker/checker.py diff --git a/.gitea/EDU_HANDOFF.md b/.gitea/EDU_HANDOFF.md new file mode 100644 index 0000000..1571ea1 --- /dev/null +++ b/.gitea/EDU_HANDOFF.md @@ -0,0 +1,55 @@ +# EDU ownership handoff + +## Review findings + +The current `main` branch can verify and roll back changed workloads across the cluster. This is unsafe when an external repository owns an application. Open PR #99 already changes this behavior to use selected workload references and immutable releases. This change is based on PR #99 branch `codex/ci-visible-checks` and keeps its protected check names: + +- `ci / Compose*` +- `ci / Workflows*` +- `ci / Shell*` +- `ci / Formatting*` +- `ci / Python and tests*` +- `ci / YAML*` +- `ci / Dockerfiles*` +- `ci / Kubernetes*` + +Open PR #100 adds service metrics. It is independent and is not required for this handoff. + +Live Gitea 28.0.0 supports dynamic job outputs, matrices, and `max-parallel`. The runner has 1 CPU, 1.7 GiB RAM, and 6 GiB free disk. Its capacity details could not be verified because the diagnostic required a sudo password. Runner concurrency capacity remains unverified. + +The workstation checkout `/srv/edu-master` exists, is clean at `7ed537f`, and uses the SSH remote `gitssh.forust.xyz:2221`. In Kubernetes context `Default`, namespace `edu-master`, the `session-keeper` and `webinar-checker` workloads are Ready. Their health and live endpoints return 200. Their running image digests match the digests in the old homelab configuration. The Redis PVC `redis-data-pvc` is bound to PV `pvc-a4f2a79a-363a-4c12-ae91-92cdfc2a0d2e`; its reclaim policy is `Delete`. Never delete, recreate, or apply this PVC. + +The EDU repository still needs its live `playwright-service` manifest and reconciled auto-reloader annotations. Those changes, plus backup and monitoring verification improvements, are in the separate EDU branch `fix/handoff-runtime`. + +## Implementation decisions + +The homelab change removes the complete `edu_master` subtree, EDU build and deploy selection, image pinning, rollback and verification cases, route probes, related tests, and Renovate references. It removes the external EDU image bypass from generic image scans. Application source and release ownership: [forust/edu-master](https://git.forust.xyz/forust/edu-master). + +The image plan compares fingerprints for all three homelab images with the last successful CI release. Each matrix job builds its image or reuses the matching immutable digest. The matrix runs one job at a time and continues after a job failure so each image has a visible result. The final build job waits for all image jobs, checks the commit, inputs, and digests, applies the full-SHA tags, and writes the existing release artifact format. This preserves the deploy gate. A manifest-only change still runs three reuse jobs and the final tag stage. Pull requests do not publish images. + +Retain the protected check names from PR #99. Keep CI and deploy separate. `AUTODEPLOY=false` is configured explicitly. The EDU main-push deploy policy is independent of this setting. + +## Dependencies and rollout order + +1. Merge PR #99 first, because this change uses its selected-workload and immutable-release behavior. PR #100 is not a dependency. +2. Complete the EDU runtime reconciliation on the EDU feature branch. Do not merge EDU into `main` yet. Configure and verify the EDU deployment secrets and trusted SSH host key outside Git. +3. Confirm the EDU release can pass its CI and immutable-SHA deploy gate. Keep the existing namespace, Secret, Redis data, Fernet key, and runtime credentials. +4. Stop or drain pending homelab runs that can deploy EDU. Remove the EDU active marker from the authoritative homelab source before any later homelab deploy. Confirm that the removal path does not prune resources. +5. Merge the homelab removal. Then merge EDU PR #1 into `main` to start a successful main-push release and deployment. +6. Verify that homelab no longer selects EDU and that EDU is the sole owner. Check rollout health, `/health`, `/live`, Redis AUTH, session TTL, delivery backlog, and monitoring. Record the release SHA, image digests, downtime, and any remaining limits in the relevant PR. + +## Rollback + +For an EDU release failure, use the EDU release snapshot and reapply the last recorded immutable digests. Inspect workload and application health after recovery. Do not restore old Redis data unless recovery requires it. + +To reverse the ownership handoff, stop EDU deployment triggers and runs first. Restore the reviewed homelab configuration and active marker only after EDU is inactive. Reapply the recorded image digests and verify workload health. Never enable both deployment paths at the same time. The PVC and its PV must remain intact. + +## Verification status + +Verified: Gitea version and matrix support; current runner CPU, memory, and free disk readings; clean EDU checkout and SSH remote; Kubernetes context and namespace; workload readiness and health endpoints; matching live image digests; and Redis PVC binding and reclaim policy. + +Not verified: runner capacity limits, merged PR state, post-merge image release, EDU deployment, or final handoff acceptance. The merge and live deployment steps remain pending. Do not report the handoff as complete until the live checks above pass. + +Live pre-handoff checks also passed: Redis AUTH (`NOAUTH` without credentials and `PONG` with them), positive session TTL, zero delivery backlog, and nine EDU vmalert rules with matching expressions and healthy evaluation. The private snapshot is `/home/forust/.local/state/edu-master-deploy/handoff-20261007T080838Z` on workstation. It includes the Redis RDB, Secret, manifests, source files, and checkout commits. Workloads have not been redeployed. + +The Redis RDB checksum passed with twelve keys. The unauthorized Redis pod test passed and the pod was removed. VictoriaMetrics reported the EDU scrape target UP. EDU Actions has the dedicated path and port secrets and `EDU_KUBE_CONTEXT=Default`. Existing deployment and registry credentials were retained. diff --git a/.gitea/runner/README.md b/.gitea/runner/README.md index bd203a2..e229986 100644 --- a/.gitea/runner/README.md +++ b/.gitea/runner/README.md @@ -67,8 +67,7 @@ The deploy user's existing Docker registry authentication remains necessary. CI publishes `release-` as a Gitea artifact with all three owned image digests and build input fingerprints. Unchanged images are reused only from a -successful main CI artifact, never from `:prod`. EDU images remain pinned to the -digests released by their application repository. Expired artifacts cause CI to +successful main CI artifact, never from `:prod`. Expired artifacts cause CI to rebuild images; they block deployment until CI is rerun. Run deploy from main with `deploy_ref=main` or a checked SHA: diff --git a/.gitignore b/.gitignore index d49fe11..9df6ca4 100644 --- a/.gitignore +++ b/.gitignore @@ -94,7 +94,6 @@ replacements.txt .idea # Temp files -edu_master/temp/ temp/* # Local-only tooling scratch space (pinned CI tools, verification scripts) tmp/ diff --git a/edu_master/.env.example b/edu_master/.env.example deleted file mode 100644 index ec9c514..0000000 --- a/edu_master/.env.example +++ /dev/null @@ -1,14 +0,0 @@ -EDU_LOGIN=your_edu_login_here -EDU_PASSWORD=your_edu_password_here -EDU_URL_LOGIN=https://edu.edu.vn.ua/user/login -EDU_URL_VERIFY=https://edu.edu.vn.ua/course/userlist -PHPSESSID_INTERVAL=10 -USER_AGENT="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36" -WEBINAR_URL=https://edu.edu.vn.ua/webinar/useractive -WEBINAR_CHECK_INTERVAL=60 -REDIS_HOST=redis -REDIS_PORT=6379 -PLAYWRIGHT_WS=ws://playwright-service:3000/ws -TZ=Europe/Kyiv -WEBINAR_TELEGRAM_TOKEN=your_telegram_bot_token_here -WEBINAR_ADMIN_ID=123456789 diff --git a/edu_master/PLAYWRIGHT_VERSION b/edu_master/PLAYWRIGHT_VERSION deleted file mode 100644 index 3ebf789..0000000 --- a/edu_master/PLAYWRIGHT_VERSION +++ /dev/null @@ -1 +0,0 @@ -1.56.0 diff --git a/edu_master/README.md b/edu_master/README.md deleted file mode 100644 index 093d47c..0000000 --- a/edu_master/README.md +++ /dev/null @@ -1,14 +0,0 @@ -# EDU deployment ownership - -Application source and release builds: `forust/edu-master`. -The homelab pipeline deploys `edu_master/k8s` and preserves explicit image digests. -The application copies in this directory are legacy and are not build inputs. -Do not publish EDU `prod` images from homelab or resolve releases from moving tags. - -For an EDU release, validate both images, select their digests in the keeper and -checker manifests, and run the existing homelab validation/apply/verification -helpers against this service. Keep the existing Secret and Redis PVC. -Coordinate Redis authentication changes with both clients and all init/probes; -keep a pre-rollout Redis backup and both previous compatible image references. -The current HTTP checker does not depend on Playwright; check other consumers -before removing the separate browser service. diff --git a/edu_master/compose.yaml b/edu_master/compose.yaml deleted file mode 100644 index a87d9ff..0000000 --- a/edu_master/compose.yaml +++ /dev/null @@ -1,49 +0,0 @@ -services: - redis: - image: redis:8.10.2-alpine - restart: unless-stopped - volumes: - - redis-data:/data - healthcheck: - test: ["CMD", "redis-cli", "ping"] - interval: 5s - timeout: 3s - retries: 5 - - playwright-service: - image: mcr.microsoft.com/playwright:v1.56.0-jammy - restart: unless-stopped - command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws - - session-keeper: - build: ./phpsessid-bot - image: gcr.forust.xyz/forust/session-keeper:prod - pull_policy: build - env_file: .env - restart: unless-stopped - depends_on: - redis: - condition: service_healthy - healthcheck: - test: ["CMD-SHELL", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"] - interval: 30s - timeout: 5s - retries: 10 - start_period: 60s - - webinar-checker: - build: ./webinar-checker - image: gcr.forust.xyz/forust/webinar-checker:prod - pull_policy: build - env_file: .env - restart: unless-stopped - depends_on: - redis: - condition: service_healthy - session-keeper: - condition: service_healthy - playwright-service: - condition: service_started - -volumes: - redis-data: diff --git a/edu_master/k8s/active b/edu_master/k8s/active deleted file mode 100644 index e69de29..0000000 diff --git a/edu_master/k8s/alerts.yaml b/edu_master/k8s/alerts.yaml deleted file mode 100644 index 82c62dc..0000000 --- a/edu_master/k8s/alerts.yaml +++ /dev/null @@ -1,115 +0,0 @@ -apiVersion: monitoring.coreos.com/v1 -kind: PrometheusRule -metadata: - name: edu-master-webinar - namespace: edu-master - labels: - release: prometheus-stack -spec: - groups: - - name: edu_master.webinar - rules: - # No successful webinar check for 5m (~2-3 missed 2-min checks). - # Catches: playwright hangs/timeouts, version skew, site changes, hung job. - # The last_success > 0 guard is mandatory: checker.py initialises - # last_success to 0, so without it `time() - 0` equals the current epoch - # and humanizeDuration renders ~20722d on every pod restart. Keep the - # duration expression on the left so $value stays the real gap. - - alert: WebinarCheckerNoSuccessfulCheck - expr: | - ((time() - webinar_check_last_success_timestamp_seconds) > 300) - and (webinar_check_last_success_timestamp_seconds > 0) - and (webinar_check_last_run_timestamp_seconds > 0) - for: 2m - labels: - severity: critical - annotations: - summary: "Webinar checker has no successful check for 5m" - description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent." - - # Checks are running but none has ever succeeded since pod start. - # Split out from the rule above so a zeroed gauge never feeds - # humanizeDuration. - - alert: WebinarCheckerNeverSucceeded - expr: | - (webinar_check_last_success_timestamp_seconds == 0) - and (webinar_check_last_run_timestamp_seconds > 0) - for: 10m - labels: - severity: critical - annotations: - summary: "Webinar checker has never completed a successful check" - description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).' - - # Fast path: 3 consecutive failures (~6+ min at 2-min interval). - - alert: WebinarCheckerConsecutiveFailures - expr: | - webinar_check_consecutive_failures >= 3 - for: 5m - labels: - severity: critical - annotations: - summary: "Webinar checker failing consecutively" - description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / http error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).' - - - alert: WebinarCheckerNeverStarted - expr: | - (time() - edu_process_start > 120) - and (webinar_check_last_run_timestamp_seconds == 0) - for: 2m - labels: - severity: critical - annotations: - summary: "Webinar checker job has not started" - description: "The process exposes metrics but its webinar job has never started." - - - alert: WebinarDeliveryPending - expr: edu_delivery_pending > 0 - for: 5m - labels: - severity: warning - annotations: - summary: "Webinar notifications await delivery" - description: "Telegram delivery has pending recipients. Check delivery failures and retry status." - - - alert: EduRedisUnavailable - expr: edu_redis_connected == 0 - for: 2m - labels: - severity: critical - annotations: - summary: "EDU checker cannot reach Redis" - description: "Redis health checks are failing; checker commands and delivery may be unavailable." - - # Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken. - - alert: WebinarCheckerScrapeDown - expr: | - absent(webinar_check_last_run_timestamp_seconds) == 1 - for: 10m - labels: - severity: critical - annotations: - summary: "Webinar checker metrics missing" - description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved." - - # EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped. - - alert: EduPhpsessidMissing - expr: | - edu_phpsessid_present == 0 - for: 10m - labels: - severity: critical - annotations: - summary: "EDU_PHPSESSID missing" - description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials." - - # Hard deps: checker deployment unavailable. - - alert: WebinarCheckerDeploymentDown - expr: | - kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0 - for: 10m - labels: - severity: critical - annotations: - summary: "Webinar checker deployment unavailable" - description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m." diff --git a/edu_master/k8s/namespace.yaml b/edu_master/k8s/namespace.yaml deleted file mode 100644 index e241a25..0000000 --- a/edu_master/k8s/namespace.yaml +++ /dev/null @@ -1,4 +0,0 @@ -apiVersion: v1 -kind: Namespace -metadata: - name: edu-master diff --git a/edu_master/k8s/playwright.yaml b/edu_master/k8s/playwright.yaml deleted file mode 100644 index c071a05..0000000 --- a/edu_master/k8s/playwright.yaml +++ /dev/null @@ -1,69 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - name: playwright-service - namespace: edu-master - labels: - app: edu-master-playwright -spec: - replicas: 1 - selector: - matchLabels: - app: edu-master-playwright - strategy: - type: Recreate - template: - metadata: - labels: - app: edu-master-playwright - spec: - containers: - - name: playwright - # renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker - image: mcr.microsoft.com/playwright:v1.56.0-jammy - imagePullPolicy: IfNotPresent - # p95 412M, max 478M over 7 days, no limit before. Request is set at p95 - # so the pod is not an eviction candidate; the limit stays above 2x the - # request because browser page lifetimes are unpredictable. - resources: - requests: - cpu: "200m" - memory: "416Mi" - limits: - memory: "1Gi" - command: - - npx - - -y - - playwright@1.56.0 - - run-server - - --port - - "3000" - - --path - - /ws - ports: - - containerPort: 3000 - readinessProbe: - tcpSocket: - port: 3000 - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 3 - livenessProbe: - tcpSocket: - port: 3000 - initialDelaySeconds: 15 - periodSeconds: 20 - timeoutSeconds: 3 ---- -apiVersion: v1 -kind: Service -metadata: - name: playwright-service - namespace: edu-master -spec: - selector: - app: edu-master-playwright - ports: - - name: ws - port: 3000 - targetPort: 3000 diff --git a/edu_master/k8s/redis-networkpolicy.yaml b/edu_master/k8s/redis-networkpolicy.yaml deleted file mode 100644 index e7e4ec1..0000000 --- a/edu_master/k8s/redis-networkpolicy.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: networking.k8s.io/v1 -kind: NetworkPolicy -metadata: - name: redis-clients-only - namespace: edu-master -spec: - podSelector: - matchLabels: - app: edu-master-redis - policyTypes: - - Ingress - ingress: - - from: - - podSelector: - matchLabels: - app: edu-master-session-keeper - - podSelector: - matchLabels: - app: edu-master-webinar-checker - ports: - - protocol: TCP - port: 6379 diff --git a/edu_master/k8s/redis.yaml b/edu_master/k8s/redis.yaml deleted file mode 100644 index e4f139c..0000000 --- a/edu_master/k8s/redis.yaml +++ /dev/null @@ -1,96 +0,0 @@ -apiVersion: apps/v1 -kind: StatefulSet -metadata: - name: redis - namespace: edu-master - labels: - app: edu-master-redis -spec: - serviceName: redis - replicas: 1 - selector: - matchLabels: - app: edu-master-redis - template: - metadata: - labels: - app: edu-master-redis - spec: - containers: - - name: redis - image: redis:8.10.2-alpine - imagePullPolicy: IfNotPresent - env: - - name: REDIS_PASSWORD - valueFrom: - secretKeyRef: - name: edu-master-secrets - key: REDIS_PASSWORD - - name: REDISCLI_AUTH - valueFrom: - secretKeyRef: - name: edu-master-secrets - key: REDIS_PASSWORD - command: - - /bin/sh - - -ec - - | - case "$REDIS_PASSWORD" in *[!0-9a-fA-F]*|'') echo 'REDIS_PASSWORD must be 64 hex characters' >&2; exit 1;; esac - [ "${#REDIS_PASSWORD}" -eq 64 ] || { echo 'REDIS_PASSWORD must be 64 hex characters' >&2; exit 1; } - umask 077 - printf 'requirepass "%s"\n' "$REDIS_PASSWORD" > /tmp/redis-auth.conf - chown redis:redis /tmp/redis-auth.conf - exec docker-entrypoint.sh redis-server /tmp/redis-auth.conf - ports: - - containerPort: 6379 - volumeMounts: - - name: redis-data - mountPath: /data - resources: - requests: - cpu: 25m - memory: 32Mi - limits: - cpu: 250m - memory: 128Mi - readinessProbe: - exec: - command: ["redis-cli", "ping"] - initialDelaySeconds: 5 - periodSeconds: 5 - timeoutSeconds: 3 - livenessProbe: - exec: - command: ["redis-cli", "ping"] - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 3 - volumes: - - name: redis-data - persistentVolumeClaim: - claimName: redis-data-pvc ---- -apiVersion: v1 -kind: PersistentVolumeClaim -metadata: - name: redis-data-pvc - namespace: edu-master -spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 1Gi ---- -apiVersion: v1 -kind: Service -metadata: - name: redis - namespace: edu-master -spec: - selector: - app: edu-master-redis - ports: - - name: redis - port: 6379 - targetPort: 6379 diff --git a/edu_master/k8s/restore-seed-job.yaml.example b/edu_master/k8s/restore-seed-job.yaml.example deleted file mode 100644 index c85b078..0000000 --- a/edu_master/k8s/restore-seed-job.yaml.example +++ /dev/null @@ -1,50 +0,0 @@ -# One-time Job to migrate redis state from docker compose to k8s (maintenance window). -# The .example file is not applied by the deploy pipeline (mask *.example.yaml). -# -# Runbook: -# 1. docker compose -f /edu_master/compose.yaml stop # SIGTERM -> redis will flush dump.rdb -# 2. docker run --rm -v edu_master_redis-data:/data \ -# -v /tmp/edu-master-backup:/backup \ -# redis:alpine sh -c "cp /data/dump.rdb /backup/ && ls -la /backup" -# 3. kubectl apply -f edu_master/k8s/namespace.yaml -# 4. kubectl apply -f # seed must come BEFORE redis pod starts -# 5. kubectl apply -f edu_master/k8s/restore-seed-job.yaml.example -# kubectl wait --for=condition=complete job/redis-restore-seed -n edu-master --timeout=120s -# 6. kubectl delete job redis-restore-seed -n edu-master -# 7. kubectl apply -f edu_master/k8s/ -R # apply remaining manifests -apiVersion: batch/v1 -kind: Job -metadata: - name: redis-restore-seed - namespace: edu-master -spec: - backoffLimit: 2 - ttlSecondsAfterFinished: 3600 - template: - spec: - restartPolicy: Never - containers: - - name: seed - image: redis:alpine - command: - - /bin/sh - - -ec - - | - ls -la /backup - cp /backup/dump.rdb /data/dump.rdb - chmod 644 /data/dump.rdb - ls -la /data - volumeMounts: - - name: redis-data - mountPath: /data - - name: backup - mountPath: /backup - readOnly: true - volumes: - - name: redis-data - persistentVolumeClaim: - claimName: redis-data-pvc - - name: backup - hostPath: - path: /tmp/edu-master-backup - type: DirectoryOrCreate diff --git a/edu_master/k8s/secrets.yaml.example b/edu_master/k8s/secrets.yaml.example deleted file mode 100644 index 7f1cf9b..0000000 --- a/edu_master/k8s/secrets.yaml.example +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: v1 -kind: Secret -metadata: - name: edu-master-secrets - namespace: edu-master -type: Opaque -stringData: - # Session keeper credentials - KEEPER_LOGIN: "" - KEEPER_PASSWORD: "" - KEEPER_INTERVAL: "10" - # EDU links - EDU_URL_BASE: "https://edu.edu.vn.ua" - EDU_URL_LOGIN: "/user/login" - EDU_URL_COURSES: "/course/userlist" - EDU_URL_WEBINAR: "/webinar/useractive" - # Playwright - USER_AGENT: "" - PLAYWRIGHT_WS: "ws://playwright-service:3000/ws" - # Webinar-checker - WEBINAR_TELEGRAM_TOKEN: "" - WEBINAR_ADMIN_ID: "" - WEBINAR_CHECK_INTERVAL: "60" - # Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml) - METRICS_PORT: "8000" - # Database - REDIS_HOST: "redis" - REDIS_PORT: "6379" - TZ: "Europe/Kyiv" diff --git a/edu_master/k8s/service.yaml b/edu_master/k8s/service.yaml deleted file mode 100644 index a20026e..0000000 --- a/edu_master/k8s/service.yaml +++ /dev/null @@ -1,15 +0,0 @@ -apiVersion: v1 -kind: Service -metadata: - name: webinar-checker - namespace: edu-master - labels: - app: edu-master-webinar-checker -spec: - selector: - app: edu-master-webinar-checker - ports: - - name: metrics - port: 8000 - targetPort: metrics - protocol: TCP diff --git a/edu_master/k8s/servicemonitor.yaml b/edu_master/k8s/servicemonitor.yaml deleted file mode 100644 index f57abd3..0000000 --- a/edu_master/k8s/servicemonitor.yaml +++ /dev/null @@ -1,16 +0,0 @@ -apiVersion: monitoring.coreos.com/v1 -kind: ServiceMonitor -metadata: - name: webinar-checker - namespace: edu-master - labels: - release: prometheus-stack -spec: - selector: - matchLabels: - app: edu-master-webinar-checker - endpoints: - - port: metrics - path: /metrics - interval: 30s - scrapeTimeout: 10s diff --git a/edu_master/k8s/session-keeper.yaml b/edu_master/k8s/session-keeper.yaml deleted file mode 100644 index 65bfd97..0000000 --- a/edu_master/k8s/session-keeper.yaml +++ /dev/null @@ -1,69 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: - reloader.stakater.com/auto: "true" - name: session-keeper - namespace: edu-master - labels: - app: edu-master-session-keeper -spec: - replicas: 1 - selector: - matchLabels: - app: edu-master-session-keeper - strategy: - type: Recreate - template: - metadata: - annotations: - edu.forust.xyz/source-commit: "90829d6c8080b9928f9da23587678e640939e10a" - labels: - app: edu-master-session-keeper - spec: - initContainers: - - name: wait-redis - image: redis:8.10.2-alpine - env: - - name: REDISCLI_AUTH - valueFrom: - secretKeyRef: - name: edu-master-secrets - key: REDIS_PASSWORD - command: - - /bin/sh - - -ec - - | - i=0 - until redis-cli -h redis ping | grep -q PONG; do - i=$((i+1)) - [ "$i" -ge 300 ] && echo "TIMEOUT: redis not ready" && exit 1 - sleep 2 - done - echo "redis is ready" - containers: - - name: session-keeper - image: gcr.forust.xyz/forust/session-keeper@sha256:49285e87cc5bc4cf4ffe190813d87927916c2df8a206daac0aeb7d227c636450 - envFrom: - - secretRef: - name: edu-master-secrets - env: - - name: REDISCLI_AUTH - valueFrom: - secretKeyRef: - name: edu-master-secrets - key: REDIS_PASSWORD - resources: - requests: - cpu: 25m - memory: 32Mi - limits: - cpu: 250m - memory: 128Mi - readinessProbe: - exec: - command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"] - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml deleted file mode 100644 index 6280de1..0000000 --- a/edu_master/k8s/webinar-checker.yaml +++ /dev/null @@ -1,87 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: - reloader.stakater.com/auto: "true" - name: webinar-checker - namespace: edu-master - labels: - app: edu-master-webinar-checker -spec: - replicas: 1 - selector: - matchLabels: - app: edu-master-webinar-checker - strategy: - type: Recreate - template: - metadata: - annotations: - edu.forust.xyz/source-commit: "90829d6c8080b9928f9da23587678e640939e10a" - labels: - app: edu-master-webinar-checker - spec: - # Enforces dependency order like compose depends_on: - # redis healthy -> session-keeper healthy (EXISTS EDU_PHPSESSID) - initContainers: - - name: wait-deps - image: redis:8.10.2-alpine - env: - - name: REDISCLI_AUTH - valueFrom: - secretKeyRef: - name: edu-master-secrets - key: REDIS_PASSWORD - command: - - /bin/sh - - -ec - - | - i=0 - until redis-cli -h redis ping | grep -q PONG; do - i=$((i+1)) - [ "$i" -ge 300 ] && echo "TIMEOUT: redis not ready" && exit 1 - sleep 2 - done - echo "redis ok" - until [ "$(redis-cli -h redis EXISTS EDU_PHPSESSID)" = "1" ]; do - i=$((i+1)) - [ "$i" -ge 300 ] && echo "TIMEOUT: no PHPSESSID (session-keeper down?)" && exit 1 - sleep 2 - done - echo "PHPSESSID ok" - containers: - - name: webinar-checker - image: gcr.forust.xyz/forust/webinar-checker@sha256:66c146f7b43cb9f0dc31ba9aa36d217e01df42ddafba5971b79c12ec215b2c01 - ports: - - name: metrics - containerPort: 8000 - protocol: TCP - readinessProbe: - httpGet: - path: /health - port: metrics - periodSeconds: 10 - timeoutSeconds: 3 - failureThreshold: 12 - initialDelaySeconds: 10 - livenessProbe: - httpGet: - path: /live - port: metrics - initialDelaySeconds: 60 - periodSeconds: 15 - timeoutSeconds: 3 - failureThreshold: 4 - envFrom: - - secretRef: - name: edu-master-secrets - env: - - name: TZ - value: "Europe/Kyiv" - resources: - requests: - cpu: "50m" - memory: "192Mi" - limits: - cpu: "600m" - memory: "384Mi" diff --git a/edu_master/phpsessid-bot/Dockerfile b/edu_master/phpsessid-bot/Dockerfile deleted file mode 100644 index 2c9f2c1..0000000 --- a/edu_master/phpsessid-bot/Dockerfile +++ /dev/null @@ -1,15 +0,0 @@ -FROM python:3.14-slim - -WORKDIR /app - -# Install system dependencies -RUN apt-get update && apt-get install -y --no-install-recommends redis-tools && rm -rf /var/lib/apt/lists/* - -# Install dependencies -RUN pip install --no-cache-dir requests==2.32.3 redis==5.2.1 - -# Copy application code -COPY . . - -# Run the bot -CMD ["python", "bot.py"] diff --git a/edu_master/phpsessid-bot/bot.py b/edu_master/phpsessid-bot/bot.py deleted file mode 100644 index 963a143..0000000 --- a/edu_master/phpsessid-bot/bot.py +++ /dev/null @@ -1,132 +0,0 @@ -import logging -import os -import time -from datetime import datetime - -import redis -import requests - -# Configure logging -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger(__name__) - - -# Load configuration (adapted to .env keys) -def _env(key, default=None): - v = os.getenv(key, default) - if isinstance(v, str) and len(v) >= 2 and ((v[0] == '"' and v[-1] == '"') or (v[0] == "'" and v[-1] == "'")): - return v[1:-1] - return v - - -LOGIN = _env('KEEPER_LOGIN') -PASSWORD = _env('KEEPER_PASSWORD') - -EDU_BASE = _env('EDU_URL_BASE', 'https://edu.edu.vn.ua') -EDU_LOGIN_PATH = _env('EDU_URL_LOGIN', '/user/login') -EDU_COURSES_PATH = _env('EDU_URL_COURSES', '/course/userlist') -URL_LOGIN = f'{EDU_BASE.rstrip("/")}/{EDU_LOGIN_PATH.lstrip("/")}' -URL_VERIFY = f'{EDU_BASE.rstrip("/")}/{EDU_COURSES_PATH.lstrip("/")}' - -INTERVAL = int(_env('KEEPER_INTERVAL', 10)) -USER_AGENT = _env( - 'USER_AGENT', - 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36', -) -REDIS_HOST = _env('REDIS_HOST', 'redis') -REDIS_PORT = int(_env('REDIS_PORT', 6379)) - -SUCCESS_FILE = '/tmp/last_success' # noqa: S108 - - -def touch_success_file(): - """Updates the timestamp of the success file for healthchecks.""" - try: - with open(SUCCESS_FILE, 'w') as f: - f.write(str(datetime.now().timestamp())) - except Exception as e: - logger.error(f'Failed to touch success file: {e}') - - -def main(): - logger.info('Starting Session Keeper Bot') - - # Connect to Redis - try: - redis_client = redis.Redis(host=REDIS_HOST, port=REDIS_PORT, decode_responses=True) - redis_client.ping() - logger.info(f'Connected to Redis at {REDIS_HOST}:{REDIS_PORT}') - except Exception as e: - logger.error(f'Failed to connect to Redis: {e}') - return - - session = requests.Session() - - # Set headers - headers = { - 'User-Agent': USER_AGENT, - 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7', - 'Accept-Language': 'en-US,en;q=0.9', - 'Cache-Control': 'max-age=0', - 'Upgrade-Insecure-Requests': '1', - 'Sec-Fetch-Site': 'same-origin', - 'Sec-Fetch-Mode': 'navigate', - 'Sec-Fetch-User': '?1', - 'Sec-Fetch-Dest': 'document', - 'Sec-Ch-Ua': '"Not_A Brand";v="99", "Chromium";v="142"', - 'Sec-Ch-Ua-Mobile': '?0', - 'Sec-Ch-Ua-Platform': '"Linux"', - 'Accept-Encoding': 'gzip, deflate, br', - 'Priority': 'u=0, i', - } - session.headers.update(headers) - - while True: - try: - logger.info('Attempting login...') - - # Login payload - payload = {'login': LOGIN, 'password': PASSWORD} - - # Perform Login - # Note: The user request shows a POST to /user/login with form data - # We need to make sure we handle the PHPSESSID correctly. - # If we already have a PHPSESSID, requests will send it. - - login_response = session.post(URL_LOGIN, data=payload, allow_redirects=True) - - logger.info(f'Login Response Status: {login_response.status_code}') - logger.info(f'Cookies after login: {session.cookies.get_dict()}') - - # Verify Session - logger.info('Verifying session...') - verify_response = session.get(URL_VERIFY, allow_redirects=False) - - logger.info(f'Verify Response Status: {verify_response.status_code}') - - if verify_response.status_code == 200: - logger.info('Session verification SUCCESS (200 OK).') - touch_success_file() - - # Save PHPSESSID to Redis - phpsessid = session.cookies.get('PHPSESSID') - if phpsessid: - try: - redis_client.set('EDU_PHPSESSID', phpsessid) - logger.info(f'Saved PHPSESSID to Redis: {phpsessid}') - except Exception as e: - logger.error(f'Failed to save PHPSESSID to Redis: {e}') - elif verify_response.status_code == 302: - logger.warning('Session verification FAILED (302 Redirect). Session might be invalid.') - else: - logger.warning(f'Session verification returned unexpected status: {verify_response.status_code}') - - except Exception as e: - logger.error(f'An error occurred: {e}') - - logger.info(f'Sleeping for {INTERVAL} minutes...') - time.sleep(INTERVAL * 60) - - -if __name__ == '__main__': - main() diff --git a/edu_master/webinar-checker/Dockerfile b/edu_master/webinar-checker/Dockerfile deleted file mode 100644 index 9edab9f..0000000 --- a/edu_master/webinar-checker/Dockerfile +++ /dev/null @@ -1,13 +0,0 @@ -FROM python:3.14-slim - -WORKDIR /app - -# renovate: datasource=pypi depName=playwright versioning=pep440 -ARG PLAYWRIGHT_VERSION=1.56.0 - -# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers -RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10" - -COPY checker.py . - -CMD ["python", "checker.py"] diff --git a/edu_master/webinar-checker/checker.py b/edu_master/webinar-checker/checker.py deleted file mode 100644 index ae3de7a..0000000 --- a/edu_master/webinar-checker/checker.py +++ /dev/null @@ -1,1821 +0,0 @@ -import asyncio -import contextlib -import json -import logging -import os -import re -import tempfile -import threading -import time -from datetime import datetime, timedelta -from html import escape -from http.server import BaseHTTPRequestHandler, HTTPServer - -import redis -from playwright.async_api import async_playwright -from telegram import ChatMember, InlineKeyboardButton, InlineKeyboardMarkup, Update -from telegram.constants import ChatType -from telegram.ext import Application, CallbackQueryHandler, CommandHandler, ContextTypes - -# Logger -logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s') -logger = logging.getLogger(__name__) - -# Suppress HTTP request logs -logging.getLogger('urllib3').setLevel(logging.WARNING) -logging.getLogger('httpx').setLevel(logging.WARNING) -logging.getLogger('telegram.ext._application').setLevel(logging.WARNING) - - -# Load environment variables -def _env(key, default=None): - v = os.getenv(key, default) - if isinstance(v, str) and len(v) >= 2 and ((v[0] == '"' and v[-1] == '"') or (v[0] == "'" and v[-1] == "'")): - return v[1:-1] - return v - - -EDU_BASE = _env('EDU_URL_BASE', 'https://edu.edu.vn.ua') -EDU_WEBINAR_PATH = _env('EDU_URL_WEBINAR', '/webinar/useractive') -WEBINAR_URL = f'{EDU_BASE.rstrip("/")}/{EDU_WEBINAR_PATH.lstrip("/")}' -DIARY_URL = f'{EDU_BASE.rstrip("/")}/user/diary' -SCHEDULE_URL = f'{EDU_BASE.rstrip("/")}/lessons/table' - -WEBINAR_CHECK_INTERVAL = int(_env('WEBINAR_CHECK_INTERVAL', 60)) -REDIS_HOST = _env('REDIS_HOST', 'redis') -REDIS_PORT = int(_env('REDIS_PORT', 6379)) -PLAYWRIGHT_WS = _env('PLAYWRIGHT_WS', 'ws://playwright-service:3000/ws') -USER_AGENT = _env( - 'USER_AGENT', - 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/142.0.0.0 Safari/537.36', -) -WEBINAR_TELEGRAM_TOKEN = _env('WEBINAR_TELEGRAM_TOKEN') -ADMIN_ID = int(_env('WEBINAR_ADMIN_ID', '0')) -METRICS_PORT = int(_env('METRICS_PORT', '8000')) - -# --- Prometheus metrics (stdlib only, no extra deps) --- -# Scraped by prometheus-stack via ServiceMonitor (edu_master/k8s/servicemonitor.yaml). -# Critical alerts in edu_master/k8s/alerts.yaml fire to Telegram via Alertmanager. -_METRICS_LOCK = threading.Lock() -_METRICS = { - 'last_run': 0.0, # Unix ts of last check start - 'last_success': 0.0, # Unix ts of last successful check - 'last_duration': 0.0, # Duration of last check in seconds - 'success_total': 0, - 'failure_total': 0, - 'consecutive_failures': 0, - 'phpsessid_present': 1, # 1 if EDU_PHPSESSID found in redis, else 0 -} - - -def _metric_check_start(): - with _METRICS_LOCK: - _METRICS['last_run'] = time.time() - - -def _metric_check_ok(duration: float): - now = time.time() - with _METRICS_LOCK: - _METRICS['last_success'] = now - _METRICS['last_duration'] = duration - _METRICS['success_total'] += 1 - _METRICS['consecutive_failures'] = 0 - _METRICS['phpsessid_present'] = 1 - - -def _metric_check_fail(duration: float, phpsessid_missing: bool = False): - with _METRICS_LOCK: - _METRICS['last_duration'] = duration - _METRICS['failure_total'] += 1 - _METRICS['consecutive_failures'] += 1 - _METRICS['phpsessid_present'] = 0 if phpsessid_missing else 1 - - -def _metrics_render() -> bytes: - with _METRICS_LOCK: - m = dict(_METRICS) - lines = [ - '# HELP webinar_check_last_run_timestamp_seconds Unix timestamp of last webinar check start.', - '# TYPE webinar_check_last_run_timestamp_seconds gauge', - f'webinar_check_last_run_timestamp_seconds {m["last_run"]}', - '# HELP webinar_check_last_success_timestamp_seconds Unix timestamp of last successful webinar check.', - '# TYPE webinar_check_last_success_timestamp_seconds gauge', - f'webinar_check_last_success_timestamp_seconds {m["last_success"]}', - '# HELP webinar_check_last_duration_seconds Duration of last webinar check in seconds.', - '# TYPE webinar_check_last_duration_seconds gauge', - f'webinar_check_last_duration_seconds {m["last_duration"]}', - '# HELP webinar_check_success_total Total successful webinar checks.', - '# TYPE webinar_check_success_total counter', - f'webinar_check_success_total {m["success_total"]}', - '# HELP webinar_check_failure_total Total failed webinar checks (timeout, playwright error, page error).', - '# TYPE webinar_check_failure_total counter', - f'webinar_check_failure_total {m["failure_total"]}', - '# HELP webinar_check_consecutive_failures Consecutive failed webinar checks (reset on success).', - '# TYPE webinar_check_consecutive_failures gauge', - f'webinar_check_consecutive_failures {m["consecutive_failures"]}', - '# HELP edu_phpsessid_present 1 if EDU_PHPSESSID exists in redis, 0 otherwise.', - '# TYPE edu_phpsessid_present gauge', - f'edu_phpsessid_present {m["phpsessid_present"]}', - ] - return ('\n'.join(lines) + '\n').encode() - - -class _MetricsHandler(BaseHTTPRequestHandler): - def do_GET(self): - if self.path == '/metrics': - body = _metrics_render() - self.send_response(200) - self.send_header('Content-Type', 'text/plain; version=0.0.4') - self.send_header('Content-Length', str(len(body))) - self.end_headers() - self.wfile.write(body) - elif self.path in ('/healthz', '/health'): - body = b'ok\n' - self.send_response(200) - self.send_header('Content-Type', 'text/plain') - self.send_header('Content-Length', str(len(body))) - self.end_headers() - self.wfile.write(body) - else: - self.send_response(404) - self.end_headers() - - def log_message(self, *args): - pass # keep bot logs clean - - -def start_metrics_server(port: int = METRICS_PORT): - server = HTTPServer(('0.0.0.0', port), _MetricsHandler) # noqa: S104 - k8s ServiceMonitor scrapes pod IP - thread = threading.Thread(target=server.serve_forever, name='metrics-server', daemon=True) - thread.start() - logger.info(f'Metrics server listening on :{port}/metrics') - return server - - -# Redis Keys -KEY_WHITELIST = 'bot:whitelist' -KEY_WHITELIST_ENABLED = 'bot:whitelist_enabled' -KEY_SUBSCRIBERS = 'bot:subscribers' -KEY_PHPSESSID = 'EDU_PHPSESSID' -KEY_WEBINAR_HISTORY = 'bot:webinar_history' # Stores last 3 webinars - -# Marker shown by the site when there are no active online lessons -NO_WEBINAR_MARKER = 'Жодного онлайн уроку зараз' - -# Initialize Redis -try: - redis_client = redis.Redis(host=REDIS_HOST, port=REDIS_PORT, decode_responses=True) - redis_client.ping() - logger.info(f'Connected to Redis at {REDIS_HOST}:{REDIS_PORT}') -except Exception as e: - logger.error(f'Failed to connect to Redis: {e}') - exit(1) - -# --- Translations --- - -TRANSLATIONS = { - 'ru': { - 'welcome': '👋 Привет, {name}!\n\nЯ бот-уведомитель о вебинарах. Я буду сообщать вам, когда появится новый вебинар.\nВы подписаны на уведомления.', - 'welcome_admin': '\n\n👑 Режим администратора активен', - 'access_denied': '⛔ Доступ запрещен. Вас нет в белом списке.', - 'help_title': '🤖 Помощь по боту\n\n', - 'help_commands': '/start - Подписаться на уведомления\n/stop - Отписаться от уведомлений\n/help - Показать это сообщение\n/language - Сменить язык', - 'help_admin': '\nКоманды администратора:\n/adduser [user_id] - Добавить пользователя в белый список\n/removeuser [user_id] - Удалить пользователя из белого списка\nИли используйте панель ниже для управления настройками.', - 'admin_only': '⛔ Только для администратора!', - 'user_added': '✅ Пользователь {user_id} добавлен в белый список', - 'user_removed': '✅ Пользователь {user_id} удален из белого списка', - 'user_not_in_whitelist': '⚠️ Пользователь {user_id} не был в белом списке', - 'cannot_remove_admin': '❌ Невозможно удалить администратора из белого списка', - 'invalid_user_id': '❌ Неверный ID пользователя. Должно быть число.', - 'usage_adduser': 'Использование: /adduser [user_id]', - 'usage_removeuser': 'Использование: /removeuser [user_id]', - 'whitelist_enabled': '✅ Белый список включен', - 'whitelist_disabled': '🚫 Белый список отключен', - 'whitelist_title': '📋 Белый список:\n', - 'subscribers_title': '👥 Подписчики:\n', - 'empty': 'Пусто', - 'force_check_running': '🔄 Запускаю проверку...', - 'check_failed': '❌ Проверка не удалась. Смотрите логи.', - 'check_completed_none': '✅ Проверка завершена. Вебинаров не найдено.', - 'check_completed': '✅ Проверка завершена. Найдено {count} вебинар(ов)!', - 'toggle_whitelist_disable': '🔒 Отключить белый список', - 'toggle_whitelist_enable': '🔓 Включить белый список', - 'view_whitelist': '📋 Посмотреть белый список', - 'view_subscribers': '👥 Посмотреть подписчиков', - 'force_check': '🔄 Принудительная проверка', - 'webinar_found': '🎓 Новый вебинар!\n\n', - 'webinar_item': '📌 {name}\n🔗 https://edu.edu.vn.ua{url}', - 'select_language': '🌐 Выберите язык / Оберіть мову / Select language:', - 'language_changed': '✅ Язык изменен на {lang}', - 'flag_ru': '🇷🇺 Русский', - 'flag_uk': '🇺🇦 Українська', - 'flag_en': '🇬🇧 English', - 'history_cleared': '✅ История вебинаров очищена', - 'history_clear_failed': '❌ Ошибка при очистке истории', - 'today': '📌 Сегодня', - 'tomorrow': '📌 Завтра', - 'week': '📅 Эта неделя', - 'month': '📅 Весь месяц', - 'no_events': 'Нет событий', - 'no_lessons': 'Нет уроков', - 'time_unknown': 'неизвестно', - 'free_period': 'Свободно', - 'unsubscribed': '🔕 Вы отписались от уведомлений.', - 'diary_title': '📅 Дневник — выберите период:', - 'diary_week_title': '📅 Неделя {start} – {end}', - 'schedule_title': '📅 {weekday} — {class_num} класс', - 'loading_diary': '🔄 Загружаю дневник...', - 'diary_load_failed': '❌ Не удалось загрузить дневник.', - 'diary_no_session': '❌ Не удалось загрузить дневник. Нет сессии или ошибка.', - 'diary_next_month': '❌ Данные за следующий месяц недоступны. Перейдите на сайт.', - 'schedule_load_failed': '❌ Не удалось загрузить расписание.', - 'class_not_set': '❌ Класс не настроен. Используйте /setclass.', - 'class_not_found': '❌ Классы не найдены в расписании.', - 'select_class': '🎒 Выберите класс — сохранится и больше не спросится:', - 'admin_class_not_set': '⚠️ Администратор еще не настроил класс для этой группы. Используйте /setclass 11', - 'loading_schedule': '🔄 Загружаю расписание...', - 'invalid_class_num': '❌ Неверный номер класса.', - 'class_saved': '✅ Класс {class_num} сохранен.', - 'admin_only_msg': '⚠️ Только администратор может настроить класс для группы.', - 'truncation': '\n\n✂️ ...(обрезано)', - 'day_mon': 'Пн', - 'day_tue': 'Вт', - 'day_wed': 'Ср', - 'day_thu': 'Чт', - 'day_fri': 'Пт', - 'day_sat': 'Сб', - 'day_sun': 'Вс', - }, - 'uk': { - 'welcome': "👋 Привіт, {name}!\n\nЯ бот-сповіщувач про вебінари. Я повідомлятиму вас, коли з'явиться новий вебінар.\nВи підписані на сповіщення.", - 'welcome_admin': '\n\n👑 Режим адміністратора активний', - 'access_denied': '⛔ Доступ заборонено. Вас немає в білому списку.', - 'help_title': '🤖 Довідка по боту\n\n', - 'help_commands': '/start - Підписатися на сповіщення\n/stop - Відписатися від сповіщень\n/help - Показати це повідомлення\n/language - Змінити мову', - 'help_admin': '\nКоманди адміністратора:\n/adduser [user_id] - Додати користувача до білого списку\n/removeuser [user_id] - Видалити користувача з білого списку\nАбо використовуйте панель нижче для керування налаштуваннями.', - 'admin_only': '⛔ Тільки для адміністратора!', - 'user_added': '✅ Користувач {user_id} доданий до білого списку', - 'user_removed': '✅ Користувач {user_id} видалений з білого списку', - 'user_not_in_whitelist': '⚠️ Користувач {user_id} не був у білому списку', - 'cannot_remove_admin': '❌ Неможливо видалити адміністратора з білого списку', - 'invalid_user_id': '❌ Невірний ID користувача. Має бути число.', - 'usage_adduser': 'Використання: /adduser [user_id]', - 'usage_removeuser': 'Використання: /removeuser [user_id]', - 'whitelist_enabled': '✅ Білий список увімкнено', - 'whitelist_disabled': '🚫 Білий список вимкнено', - 'whitelist_title': '📋 Білий список:\n', - 'subscribers_title': '👥 Підписники:\n', - 'empty': 'Порожньо', - 'force_check_running': '🔄 Запускаю перевірку...', - 'check_failed': '❌ Перевірка не вдалася. Дивіться логи.', - 'check_completed_none': '✅ Перевірка завершена. Вебінарів не знайдено.', - 'check_completed': '✅ Перевірка завершена. Знайдено {count} вебінар(ів)!', - 'toggle_whitelist_disable': '🔒 Вимкнути білий список', - 'toggle_whitelist_enable': '🔓 Увімкнути білий список', - 'view_whitelist': '📋 Переглянути білий список', - 'view_subscribers': '👥 Переглянути підписників', - 'force_check': '🔄 Примусова перевірка', - 'webinar_found': '🎓 Новий вебінар!\n\n', - 'webinar_item': '📌 {name}\n🔗 https://edu.edu.vn.ua{url}', - 'select_language': '🌐 Виберіть мову / Выберите язык / Select language:', - 'language_changed': '✅ Мову змінено на {lang}', - 'flag_ru': '🇷🇺 Русский', - 'flag_uk': '🇺🇦 Українська', - 'flag_en': '🇬🇧 English', - 'history_cleared': '✅ Історія вебінарів очищена', - 'history_clear_failed': '❌ Помилка при очищенні історії', - 'today': '📌 Сьогодні', - 'tomorrow': '📌 Завтра', - 'week': '📅 Цей тиждень', - 'month': '📅 Весь місяць', - 'no_events': 'Немає подій', - 'no_lessons': 'Немає уроків', - 'time_unknown': 'невідомо', - 'free_period': 'Вільно', - 'unsubscribed': '🔕 Ви відписалися від сповіщень.', - 'diary_title': '📅 Щоденник — виберіть період:', - 'diary_week_title': '📅 Тиждень {start} – {end}', - 'schedule_title': '📅 {weekday} — {class_num} клас', - 'loading_diary': '🔄 Завантажую щоденник...', - 'diary_load_failed': '❌ Не вдалося завантажити щоденник.', - 'diary_no_session': '❌ Не вдалося завантажити щоденник. Немає сесії або помилка.', - 'diary_next_month': '❌ Дані за наступний місяць недоступні. Перейдіть на сайт.', - 'schedule_load_failed': '❌ Не вдалося завантажити розклад.', - 'class_not_set': '❌ Клас не налаштовано. Використайте /setclass.', - 'class_not_found': '❌ Класи не знайдені в розкладі.', - 'select_class': '🎒 Оберіть клас — збережеться і більше не питатиметься:', - 'admin_class_not_set': '⚠️ Адміністратор ще не налаштував клас для цієї групи. Використайте /setclass 11', - 'loading_schedule': '🔄 Завантажую розклад...', - 'invalid_class_num': '❌ Невірний номер класу.', - 'class_saved': '✅ Клас {class_num} збережено.', - 'admin_only_msg': '⚠️ Тільки адміністратор може налаштувати клас для групи.', - 'truncation': '\n\n✂️ ...(обрізано)', - 'day_mon': 'Пн', - 'day_tue': 'Вт', - 'day_wed': 'Ср', - 'day_thu': 'Чт', - 'day_fri': 'Пт', - 'day_sat': 'Сб', - 'day_sun': 'Нд', - }, - 'en': { - 'welcome': '👋 Hello, {name}!\n\nI am the Webinar Checker Bot. I will notify you when a new webinar appears.\nYou have been subscribed to notifications.', - 'welcome_admin': '\n\n👑 Admin Mode Active', - 'access_denied': '⛔ Access denied. You are not on the whitelist.', - 'help_title': '🤖 Bot Help\n\n', - 'help_commands': '/start - Subscribe to notifications\n/stop - Unsubscribe from notifications\n/help - Show this message\n/language - Change language', - 'help_admin': '\nAdmin Commands:\n/adduser [user_id] - Add user to whitelist\n/removeuser [user_id] - Remove user from whitelist\nOr use the panel below to manage settings.', - 'admin_only': '⛔ Admin only!', - 'user_added': '✅ User {user_id} added to whitelist', - 'user_removed': '✅ User {user_id} removed from whitelist', - 'user_not_in_whitelist': '⚠️ User {user_id} was not in whitelist', - 'cannot_remove_admin': '❌ Cannot remove admin from whitelist', - 'invalid_user_id': '❌ Invalid user ID. Must be a number.', - 'usage_adduser': 'Usage: /adduser [user_id]', - 'usage_removeuser': 'Usage: /removeuser [user_id]', - 'whitelist_enabled': '✅ Whitelist Enabled', - 'whitelist_disabled': '🚫 Whitelist Disabled', - 'whitelist_title': '📋 Whitelist:\n', - 'subscribers_title': '👥 Subscribers:\n', - 'empty': 'Empty', - 'force_check_running': '🔄 Running immediate check...', - 'check_failed': '❌ Check failed. See logs for details.', - 'check_completed_none': '✅ Check completed. No webinars found.', - 'check_completed': '✅ Check completed. Found {count} webinar(s)!', - 'toggle_whitelist_disable': '🔒 Disable Whitelist', - 'toggle_whitelist_enable': '🔓 Enable Whitelist', - 'view_whitelist': '📋 View Whitelist', - 'view_subscribers': '👥 View Subscribers', - 'force_check': '🔄 Force Check', - 'webinar_found': '🎓 New webinar found!\n\n', - 'webinar_item': '📌 {name}\n🔗 https://edu.edu.vn.ua{url}', - 'select_language': '🌐 Select language / Виберіть мову / Выберите язык:', - 'language_changed': '✅ Language changed to {lang}', - 'flag_ru': '🇷🇺 Русский', - 'flag_uk': '🇺🇦 Українська', - 'flag_en': '🇬🇧 English', - 'history_cleared': '✅ Webinar history cleared', - 'history_clear_failed': '❌ Error clearing history', - 'today': '📌 Today', - 'tomorrow': '📌 Tomorrow', - 'week': '📅 This week', - 'month': '📅 Whole month', - 'no_events': 'No events', - 'no_lessons': 'No lessons', - 'time_unknown': 'unknown', - 'free_period': 'Free', - 'unsubscribed': '🔕 You have unsubscribed from notifications.', - 'diary_title': '📅 Diary — choose a period:', - 'diary_week_title': '📅 Week {start} – {end}', - 'schedule_title': '📅 {weekday} — {class_num} class', - 'loading_diary': '🔄 Loading diary...', - 'diary_load_failed': '❌ Failed to load diary.', - 'diary_no_session': '❌ Failed to load diary. No session or error.', - 'diary_next_month': '❌ Next month data is not available. Please visit the website.', - 'schedule_load_failed': '❌ Failed to load schedule.', - 'class_not_set': '❌ Class is not set. Use /setclass.', - 'class_not_found': '❌ No classes found in the schedule.', - 'select_class': '🎒 Choose a class — it will be saved and not asked again:', - 'admin_class_not_set': '⚠️ Admin has not set a class for this group yet. Use /setclass 11', - 'loading_schedule': '🔄 Loading schedule...', - 'invalid_class_num': '❌ Invalid class number.', - 'class_saved': '✅ Class {class_num} saved.', - 'admin_only_msg': '⚠️ Only an admin can set the class for this group.', - 'truncation': '\n\n✂️ ...(truncated)', - 'day_mon': 'Mon', - 'day_tue': 'Tue', - 'day_wed': 'Wed', - 'day_thu': 'Thu', - 'day_fri': 'Fri', - 'day_sat': 'Sat', - 'day_sun': 'Sun', - }, -} - -# --- Language Helper Functions --- - - -LANG_NAME_MAP = {'ru': 'Русский', 'uk': 'Українська', 'en': 'English'} - - -def get_user_language(user_id: int) -> str: - """Get user's preferred language from Redis. Default: English.""" - lang = redis_client.get(f'user:{user_id}:language') - return lang if lang in ['ru', 'uk', 'en'] else 'en' - - -def set_user_language(user_id: int, lang: str): - """Save user's language preference to Redis.""" - if lang in ['ru', 'uk', 'en']: - redis_client.set(f'user:{user_id}:language', lang) - logger.info(f'User {user_id} language set to {lang}') - - -def t(user_id: int, key: str, **kwargs) -> str: - """Translate message for user with optional formatting. - - Fallback chain: user lang → en → uk → ru, otherwise return the key. - """ - lang = get_user_language(user_id) - message = None - for fallback in (lang, 'en', 'uk', 'ru'): - message = TRANSLATIONS.get(fallback, {}).get(key) - if message is not None: - break - if message is None: - return key - if kwargs: - return message.format(**kwargs) - return message - - -def _tr(lang: str, key: str, **kwargs) -> str: - """Translate by explicit lang code (for format_* helpers). - - Fallback chain: lang → en → uk → ru, otherwise return the key. - """ - message = None - for fallback in (lang, 'en', 'uk', 'ru'): - message = TRANSLATIONS.get(fallback, {}).get(key) - if message is not None: - break - if message is None: - return key - if kwargs: - return message.format(**kwargs) - return message - - -def get_chat_language(chat_id: int) -> str: - """Get group chat language from Redis. Default: Ukrainian.""" - lang = redis_client.get(f'chat:{chat_id}:language') - return lang if lang in ['ru', 'uk', 'en'] else 'uk' - - -def set_chat_language(chat_id: int, lang: str): - """Save group chat language preference to Redis.""" - if lang in ['ru', 'uk', 'en']: - redis_client.set(f'chat:{chat_id}:language', lang) - logger.info(f'Chat {chat_id} language set to {lang}') - - -def resolve_lang(chat, user_id=None) -> str: - """Resolve effective language: private → user lang, groups → chat lang (default 'uk').""" - chat_id = getattr(chat, 'id', chat) - chat_type = getattr(chat, 'type', None) - if chat_type is None: - try: - is_private = int(chat_id) > 0 - except (TypeError, ValueError): - is_private = True - if is_private: - uid = user_id if user_id is not None else chat_id - return get_user_language(int(uid)) - return get_chat_language(int(chat_id)) - if chat_type in (ChatType.PRIVATE, 'private'): - uid = user_id if user_id is not None else chat_id - return get_user_language(int(uid)) - return get_chat_language(chat_id) - - -def t_chat(chat, user_id, key: str, **kwargs) -> str: - """Translate using chat-resolved language (private → user, groups → chat 'uk' default).""" - return _tr(resolve_lang(chat, user_id), key, **kwargs) - - -def get_language_keyboard(user_id: int | None = None): - """Generate language selection keyboard.""" - lang = get_user_language(user_id) if user_id is not None else 'en' - keyboard = [ - [ - InlineKeyboardButton(_tr(lang, 'flag_ru'), callback_data='lang_ru'), - InlineKeyboardButton(_tr(lang, 'flag_uk'), callback_data='lang_uk'), - ], - [ - InlineKeyboardButton(_tr(lang, 'flag_en'), callback_data='lang_en'), - ], - ] - return InlineKeyboardMarkup(keyboard) - - -# --- Helper Functions --- - - -def is_whitelisted(user_id: int) -> bool: - """Check if user is allowed to use the bot.""" - if user_id == ADMIN_ID: - return True - - enabled = redis_client.get(KEY_WHITELIST_ENABLED) - if enabled == '0': # Whitelist disabled - return True - - return redis_client.sismember(KEY_WHITELIST, str(user_id)) - - -async def is_group_admin(update: Update, context: ContextTypes.DEFAULT_TYPE) -> bool: - """Check if the user is an administrator in the group.""" - user = update.effective_user - chat = update.effective_chat - - if chat.type in [ChatType.PRIVATE, 'private']: - return True - - try: - member = await context.bot.get_chat_member(chat.id, user.id) - return member.status in [ChatMember.OWNER, ChatMember.ADMINISTRATOR] - except Exception as e: - logger.error(f'Failed to check admin status: {e}') - return False - - -def get_admin_keyboard(user_id: int): - """Generate admin panel keyboard.""" - whitelist_enabled = redis_client.get(KEY_WHITELIST_ENABLED) != '0' - toggle_text = t(user_id, 'toggle_whitelist_disable') if whitelist_enabled else t(user_id, 'toggle_whitelist_enable') - - keyboard = [ - [InlineKeyboardButton(toggle_text, callback_data='toggle_whitelist')], - [InlineKeyboardButton(t(user_id, 'view_whitelist'), callback_data='view_whitelist')], - [InlineKeyboardButton(t(user_id, 'view_subscribers'), callback_data='view_subscribers')], - [InlineKeyboardButton(t(user_id, 'force_check'), callback_data='force_check')], - ] - return InlineKeyboardMarkup(keyboard) - - -# --- Diary Functions --- - -SCHEDULE_WEEKDAYS_FULL = { - 'uk': ['Понеділок', 'Вівторок', 'Середа', 'Четвер', "П'ятниця", 'Субота', 'Неділя'], - 'ru': ['Понедельник', 'Вторник', 'Среда', 'Четверг', 'Пятница', 'Суббота', 'Воскресенье'], - 'en': ['Monday', 'Tuesday', 'Wednesday', 'Thursday', 'Friday', 'Saturday', 'Sunday'], -} -# Canonical weekday names used in callback_data and schedule lookup (site language). -SCHEDULE_WEEKDAYS_CANONICAL = SCHEDULE_WEEKDAYS_FULL['uk'] -SCHEDULE_CACHE_TTL = 18000 # 5 hours - - -def _norm_day(s): - """Normalize weekday name for tolerant matching (case, spaces, apostrophes).""" - if not isinstance(s, str): - return '' - s = s.strip().casefold() - for _q in ('\u2019', '\u2018', '\u02bc', '`'): - s = s.replace(_q, "'") - s = re.sub(r'\s+', ' ', s) - return s.strip() - - -_SCHEDULE_WEEKDAYS_CANONICAL_NORM = [_norm_day(d) for d in SCHEDULE_WEEKDAYS_CANONICAL] - -DIARY_WEEKDAYS_SHORT = { - 'uk': ['Пн', 'Вт', 'Ср', 'Чт', 'Пт', 'Сб', 'Нд'], - 'ru': ['Пн', 'Вт', 'Ср', 'Чт', 'Пт', 'Сб', 'Вс'], - 'en': ['Mon', 'Tue', 'Wed', 'Thu', 'Fri', 'Sat', 'Sun'], -} - - -def get_diary_keyboard(user_id: int | None = None): - today = datetime.now() - lang = get_user_language(user_id) if user_id is not None else 'en' - today_label = _tr(lang, 'today') - tomorrow_label = _tr(lang, 'tomorrow') - week_label = _tr(lang, 'week') - month_label = _tr(lang, 'month') - keyboard = [ - [ - InlineKeyboardButton(f'{today_label} ({today.day}.{today.month:02d})', callback_data='diary_today'), - InlineKeyboardButton(tomorrow_label, callback_data='diary_tomorrow'), - ], - [ - InlineKeyboardButton(week_label, callback_data='diary_week'), - InlineKeyboardButton(month_label, callback_data='diary_month'), - ], - ] - return InlineKeyboardMarkup(keyboard) - - -def _parse_calendar_html(table_html: str) -> tuple: - """Parse calendar HTML table into (month_text, {day_num: {weekday, events}}). - - Each event is a dict: {'title': str, 'id': site event id | None, 'time': 'HH:MM' | None}. - """ - days = {} - weekdays = [] - rows = re.findall(r']*>(.*?)', table_html, re.DOTALL) - - month_text = '' - for r_idx, row in enumerate(rows): - cells = re.findall(r']*>(.*?)', row, re.DOTALL) - - if r_idx == 0: - # Month navigation row: extract "Травень 2026" from nav text - raw = re.sub(r'<[^>]+>', ' ', row).strip() - raw = re.sub(r'\s+', ' ', raw) - m = re.search(r'([А-Яа-яіїєґ\']+\s*:?\s*\d{4})', raw) - month_text = m.group(1).replace(' : ', ' ').strip() if m else raw - elif r_idx == 1: - # Day names row - for cell in cells: - name = re.sub(r'<[^>]+>', '', cell).strip() - if name: - weekdays.append(name) - else: - # Data rows: each cell = a day - for col_idx, cell in enumerate(cells): - # Extract day number — first number in the cell text - text = re.sub(r'<[^>]+>', ' ', cell).strip() - text = re.sub(r'\s+', ' ', text) - dm = re.match(r'(\d+)', text) - if not dm: - continue - day_num = dm.group(1) - - # Extract events: title attribute (full name) of ALL tags inside the cell. - # Each event keeps its site data-event-id and a 'time' slot (filled later - # from the AJAX popup by fetch_diary_data). 'time' is 'HH:MM' or None. - events = [] - for a_match in re.finditer(r']*>(.*?)', cell, re.DOTALL): - a_tag = a_match.group(0) - # Prefer the title attribute (contains full name, not truncated) - title_m = re.search(r'title\s*=\s*"([^"]*)"', a_tag) - et = title_m.group(1).strip() if title_m else re.sub(r'<[^>]+>', '', a_match.group(1)).strip() - if not et: - continue - id_m = re.search(r'data-event-id\s*=\s*"?(\d+)"?', a_tag) - events.append( - { - 'title': et, - 'id': id_m.group(1) if id_m else None, - 'time': None, - } - ) - - weekday = weekdays[col_idx] if col_idx < len(weekdays) else '' - days[day_num] = {'weekday': weekday, 'weekday_idx': col_idx, 'events': events} - - return month_text, days - - -async def _collect_event_times(page) -> dict: - """Read event times straight from the rendered calendar DOM. - - The diary page embeds `div.event-full-info[data-event-full-info-id]` - containing `span.data` (e.g. "2026-09-02 16:30:00") for every event, so no - AJAX popup clicks are needed. Returns {event_id: 'HH:MM'} for events that - have a date; events without one are simply skipped. - """ - times_by_id: dict[str, str] = {} - try: - raw_times = await page.evaluate( - """() => { - const out = {}; - for (const div of document.querySelectorAll( - 'div.event-full-info[data-event-full-info-id]' - )) { - const id = div.getAttribute('data-event-full-info-id'); - const date_span = div.querySelector('p.date span.data'); - if (id && date_span) { - out[id] = date_span.textContent.trim(); - } - } - return out; - }""" - ) - except Exception as e: - logger.warning(f'Failed to read diary event times from DOM: {e}') - return times_by_id - - for event_id, time_text in (raw_times or {}).items(): - if not time_text: - continue - m = re.search(r'(\d{1,2}:\d{2})', time_text) - if m: - times_by_id[event_id] = m.group(1) - - logger.info(f'Diary event times collected for {len(times_by_id)} events') - return times_by_id - - -async def fetch_diary_data(phpsessid: str) -> dict | None: - logger.info('Fetching diary data via Playwright...') - try: - async with asyncio.timeout(60): - async with async_playwright() as p: - browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15) - try: - context_browser = await browser.new_context(user_agent=USER_AGENT) - await context_browser.add_cookies( - [{'name': 'PHPSESSID', 'value': phpsessid, 'domain': 'edu.edu.vn.ua', 'path': '/'}] - ) - page = await context_browser.new_page() - - try: - await asyncio.wait_for(page.goto(DIARY_URL, wait_until='domcontentloaded'), timeout=30) - await page.wait_for_selector('table.calendar', timeout=10000) - await page.wait_for_timeout(1500) - - table_html = await page.evaluate(""" - () => { - const t = document.querySelector('table.calendar'); - return t ? t.outerHTML : null; - } - """) - if not table_html: - logger.error('table.calendar not found in DOM') - return None - - # Debug: save HTML for troubleshooting - with contextlib.suppress(Exception), open('/tmp/diary_debug.html', 'w', encoding='utf-8') as f: # noqa: S108 - f.write(table_html) - - month_text, days = _parse_calendar_html(table_html) - - # Read event times by opening each event's AJAX popup. - times_by_id = await _collect_event_times(page) - if times_by_id: - for day_data in days.values(): - for ev in day_data.get('events', []): - eid = ev.get('id') - if eid and eid in times_by_id: - ev['time'] = times_by_id[eid] - - logger.info( - f'Diary parsed: month={month_text!r}, days_with_events={sum(1 for d in days.values() if d["events"])}/{len(days)}' - ) - - return {'monthFullText': month_text, 'days': days} - - except Exception as e: - logger.error(f'Error parsing diary: {e}') - return None - finally: - with contextlib.suppress(Exception): - await asyncio.wait_for(page.close(), timeout=5) - with contextlib.suppress(Exception): - await asyncio.wait_for(context_browser.close(), timeout=5) - finally: - with contextlib.suppress(Exception): - await asyncio.wait_for(browser.close(), timeout=5) - except TimeoutError: - logger.error('Diary fetch timed out (60s)') - return None - except Exception as e: - logger.error(f'Playwright error in diary fetch: {e}') - return None - - -def _parse_diary_month(text: str) -> str: - match = re.search(r'([А-Яа-яіїєґ\']+\s*:\s*\d{4})', text) - if match: - return match.group(1).replace(' : ', ' ').strip() - return text.strip() - - -def _format_event_line(event, lang: str) -> str: - """Render one diary event line: 📌 title, with time in parens when known. - - '08:00' is the site's placeholder for "no time specified" — show a localized - 'time_unknown' mark instead. Unknown/absent time renders as before, no parens. - """ - if isinstance(event, dict): - title = escape(str(event.get('title') or '')) - tme = event.get('time') - else: - title = escape(str(event)) - tme = None - if tme == '08:00': - return f'📌 {title} ({_tr(lang, "time_unknown")})' - if tme: - return f'📌 {title} ({escape(str(tme))})' - return f'📌 {title}' - - -def format_diary_day(data: dict, day_num: int, lang: str = 'en') -> str: - days = data.get('days', {}) - month_str = _parse_diary_month(data.get('monthFullText', '')) - day_data = days.get(str(day_num)) - lines = [f'📅 {day_num} {month_str}', '─' * 18] - if not day_data or not day_data.get('events'): - lines.append(_tr(lang, 'no_events')) - else: - for e in day_data['events']: - lines.append(_format_event_line(e, lang)) - lines.append(f'\n🔗 {DIARY_URL}') - return '\n'.join(lines) - - -def format_diary_week(data: dict, today: datetime, lang: str = 'en') -> str: - days = data.get('days', {}) - _parse_diary_month(data.get('monthFullText', '')) - monday = today - timedelta(days=today.weekday()) - friday = monday + timedelta(days=4) - start = f'{monday.day}.{monday.month}' - end = f'{friday.day}.{friday.month}' - short = DIARY_WEEKDAYS_SHORT.get(lang, DIARY_WEEKDAYS_SHORT['en']) - lines = [_tr(lang, 'diary_week_title', start=start, end=end) + '\n'] - for i in range(5): - d = monday + timedelta(days=i) - day_data = days.get(str(d.day)) - lines.append(f'─ {short[i]} {d.day}.{d.month} ─') - if not day_data or not day_data.get('events'): - lines.append(_tr(lang, 'no_events') + '\n') - else: - for e in day_data['events']: - lines.append(_format_event_line(e, lang)) - lines.append('') - lines.append(f'🔗 {DIARY_URL}') - return '\n'.join(lines) - - -def format_diary_month(data: dict, lang: str = 'en') -> str: - days = data.get('days', {}) - month_str = _parse_diary_month(data.get('monthFullText', '')) - lines = [f'📅 {month_str}\n'] - for day_num in sorted(days.keys(), key=int): - day_data = days[day_num] - if day_data.get('weekday_idx', 0) >= 5: - continue - if day_data.get('weekday', '').strip().lower() in ( - 'субота', - 'суббота', - 'saturday', - 'неділя', - 'воскресенье', - 'sunday', - ): - continue - events = day_data.get('events', []) - weekday = day_data.get('weekday', '') - lines.append(f'─ {weekday} {day_num} ─') - if not events: - lines.append(_tr(lang, 'no_events') + '\n') - else: - for e in events: - lines.append(_format_event_line(e, lang)) - lines.append('') - lines.append(f'🔗 {DIARY_URL}') - return '\n'.join(lines) - - -async def _get_diary_data(context: ContextTypes.DEFAULT_TYPE) -> dict | None: - cached = context.user_data.get('diary_cache') - now_ts = time.time() - if cached and (now_ts - cached.get('timestamp', 0)) < 300: - return cached['data'] - phpsessid = redis_client.get(KEY_PHPSESSID) - if not phpsessid: - return None - data = await fetch_diary_data(phpsessid) - if data: - context.user_data['diary_cache'] = {'data': data, 'timestamp': now_ts} - return data - - -# --- Schedule Functions --- - - -def get_schedule_day_keyboard(user_id: int | None = None): - lang = get_user_language(user_id) if user_id is not None else 'en' - full = SCHEDULE_WEEKDAYS_FULL.get(lang, SCHEDULE_WEEKDAYS_FULL['en']) - today = datetime.now() - tomorrow = today + timedelta(days=1) - today_label = _tr(lang, 'today') - tomorrow_label = _tr(lang, 'tomorrow') - short_keys = ['day_mon', 'day_tue', 'day_wed', 'day_thu', 'day_fri'] - keyboard = [ - [ - InlineKeyboardButton( - f'{today_label} ({full[today.weekday()]})', - callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[today.weekday()]}', - ), - InlineKeyboardButton( - f'{tomorrow_label} ({full[tomorrow.weekday()]})', - callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[tomorrow.weekday()]}', - ), - ], - [ - InlineKeyboardButton( - _tr(lang, short_keys[0]), callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[0]}' - ), - InlineKeyboardButton( - _tr(lang, short_keys[1]), callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[1]}' - ), - InlineKeyboardButton( - _tr(lang, short_keys[2]), callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[2]}' - ), - ], - [ - InlineKeyboardButton( - _tr(lang, short_keys[3]), callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[3]}' - ), - InlineKeyboardButton( - _tr(lang, short_keys[4]), callback_data=f'schedule_day_{SCHEDULE_WEEKDAYS_CANONICAL[4]}' - ), - ], - ] - return InlineKeyboardMarkup(keyboard) - - -def get_schedule_class_keyboard(classes: list): - rows = [] - for i in range(0, len(classes), 4): - row = [InlineKeyboardButton(str(c), callback_data=f'schedule_class_{c}') for c in classes[i : i + 4]] - rows.append(row) - return InlineKeyboardMarkup(rows) - - -def _get_chat_class(chat, user_id) -> str | None: - """Get stored schedule class for a private user or a group chat.""" - if chat.type == 'private': - return redis_client.get(f'user:{user_id}:schedule_class') - return redis_client.get(f'chat:{chat.id}:schedule_class') - - -def _set_chat_class(chat, user_id, class_num) -> None: - """Save schedule class for a private user or a group chat.""" - if chat.type == 'private': - redis_client.set(f'user:{user_id}:schedule_class', str(class_num)) - else: - redis_client.set(f'chat:{chat.id}:schedule_class', str(class_num)) - - -def _parse_schedule_cell(cell_html: str) -> list: - """Parse a single schedule cell, returning list of {subject, note, teacher}.""" - lessons = [] - parts = re.split(r'', cell_html, flags=re.IGNORECASE) - for part in parts: - part = re.sub(r'', '', part, flags=re.DOTALL) - part = re.sub(r'