chore(deploy): rework k8s pipeline, monitoring and postgres 17

Deploy workflow uses git-tracked manifests, DISABLED flag and kustomize overlays; add webinar-checker metrics with ServiceMonitor and alerts; upgrade shared postgres to 17 with statuspage DB and probes/resources.
This commit is contained in:
forust committed 2026-09-23 15:47:26 +02:00
1 parent 8b2cf29771
commit 46c7e99b1d
19 files changed
+500 -98

No files matched your search

-20
View File
@@ -347,23 +347,3 @@ jobs:
;; ;;
esac esac
done done
deploy-userbot-panel:
needs: build
if: github.ref_name == 'main' && contains(needs.build.outputs.services, 'userbot')
runs-on: [self-hosted, linux, arch, homelab, prod]
steps:
- name: Checkout repository
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Apply and roll out userbot panel
shell: bash
run: |
kubectl apply -f userbot/k8s/base/panel.yaml
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
# Keep legacy deployments (forust/anna) in sync with manifests; they have no replicas field, so apply leaves scaling to the user manager only.
kubectl apply -f userbot/k8s/base/userbots.yaml
kubectl rollout restart deployment/userbot-panel -n userbot
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
+169 -41
View File
@@ -19,9 +19,6 @@ jobs:
DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_USER: ${{ secrets.DEPLOY_USER }}
DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }} DEPLOY_PATH: ${{ secrets.DEPLOY_PATH }}
DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }} DEPLOY_KEY: ${{ secrets.DEPLOY_SSH_KEY }}
# Set APPLY_PRUNE=true to enable kubectl apply --prune. Requires every
# manifest to carry label app.kubernetes.io/managed-by=homelab-deploy,
# otherwise previously applied resources get deleted on the next run.
APPLY_PRUNE: ${{ vars.APPLY_PRUNE }} APPLY_PRUNE: ${{ vars.APPLY_PRUNE }}
run: | run: |
set -euo pipefail set -euo pipefail
@@ -57,59 +54,170 @@ jobs:
fi fi
git -C "$repo" fetch origin main git -C "$repo" fetch origin main
git -C "$repo" reset --hard origin/main
# Runtime selection: a service is k8s-managed when $SERVICE/k8s/active echo "== Workstation state =="
# exists. Otherwise it is compose-managed, and only k8s/routing/* echo " local: $(git -C "$repo" rev-parse --short HEAD)"
# manifests (external Services / EndpointSlices / ServersTransport / echo " remote: $(git -C "$repo" rev-parse --short origin/main)"
# Ingresses that route to docker backends) are applied.
# migrate: touch SERVICE/k8s/active (+ move routing files up) if [ -n "$(git -C "$repo" status --porcelain --untracked-files=no)" ]; then
# rollback: rm SERVICE/k8s/active echo "ERROR: workstation has local tracked modifications, refusing reset:"
git -C "$repo" status --porcelain --untracked-files=no
git -C "$repo" diff --stat
exit 1
fi
git -C "$repo" reset --hard origin/main
cd "$repo"
is_disabled() {
local target="$1"
if [ -f "$target" ]; then
target="$(dirname "$target")"
fi
while true; do
if [ -f "$target/DISABLED" ]; then
return 0
fi
if [ "$target" = "$repo" ]; then
break
fi
target="$(dirname "$target")"
case "$target" in
"$repo"/*) ;;
*) break ;;
esac
done
return 1
}
collect_k8s() { collect_k8s() {
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \ git ls-files -- "$1" \
! -path '*/routing/*' ! -path '*/overlays/*' \ | grep -E '\.ya?ml$' \
! -name 'kustomization.y*ml' ! -name '*.example.y*ml' \ | grep -Ev '/routing/|/overlays/' \
! -name '*values.y*ml' ! -name 'patch-*.y*ml' \ | grep -Ev '(^|/)(kustomization\.ya?ml|.*\.example\.ya?ml|.*values\.ya?ml|patch-.*\.ya?ml)$' \
| grep -Ev '(^|/)[^/]*secret[^/]*\.ya?ml$' \
| sort | sort
} }
collect_k8s_inactive() { collect_k8s_inactive() {
find "$1" -type f \( -name '*.yaml' -o -name '*.yml' \) \ collect_k8s "$1" \
\( -name 'namespace.y*ml' -o -path '*/routing/*' \) \ | grep -E '(^|/)namespace\.ya?ml$|/routing/'
! -path '*/overlays/*' ! -name '*.example.y*ml' \
| sort
} }
mapfile -t compose_stacks < <( kustomize_overlay() {
find "$repo" -type f \( -name 'compose.yaml' -o -name 'compose.yml' \) | sort if [ -f "$1/overlays/prod/kustomization.yaml" ]; then
echo "$1/overlays/prod"
elif [ -f "$1/base/kustomization.yaml" ]; then
echo "$1/base"
fi
}
mapfile -t k8s_dirs < <(
git ls-files '*.yaml' '*.yml' \
| grep -E '(^|/)k8s/' \
| sed -E 's#((^|.*/)k8s)/.*#\1#' \
| sort -u
) )
mapfile -t k8s_manifests < <( k8s_manifests=()
for kd in $(find "$repo" -type d -name k8s ! -path '*/.git/*' | sort); do kustomize_apps=()
for kd_rel in "${k8s_dirs[@]}"; do
kd="$repo/$kd_rel"
if is_disabled "$kd"; then
echo "skip (DISABLED): $kd_rel"
continue
fi
if [ -f "$kd/active" ]; then if [ -f "$kd/active" ]; then
collect_k8s "$kd" overlay="$(kustomize_overlay "$kd" || true)"
if [ -n "${overlay:-}" ]; then
echo "kustomize app: ${overlay#$repo/}"
kustomize_apps+=("$overlay")
else else
collect_k8s_inactive "$kd" while IFS= read -r f; do
[ -n "$f" ] && k8s_manifests+=("$repo/$f")
done < <(collect_k8s "$kd_rel" || true)
fi
else
while IFS= read -r f; do
[ -n "$f" ] && k8s_manifests+=("$repo/$f")
done < <(collect_k8s_inactive "$kd_rel" || true)
fi fi
done done
mapfile -t compose_rel < <(
git ls-files '*/compose.yaml' '*/compose.yml' compose.yaml compose.yml | sort
) )
compose_stacks=()
for cf_rel in "${compose_rel[@]}"; do
cf="$repo/$cf_rel"
if is_disabled "$cf"; then
echo "skip (DISABLED): $cf_rel"
continue
fi
if [ -f "$(dirname "$cf")/k8s/active" ]; then
echo "skip (k8s-managed): $cf_rel"
continue
fi
compose_stacks+=("$cf")
done
echo "== Validate compose stacks ==" echo "== Validate compose stacks =="
for cf in "${compose_stacks[@]}"; do for cf in "${compose_stacks[@]}"; do
dir=$(dirname "$cf")
if [ -f "$dir/k8s/active" ]; then
echo " skip (k8s-managed): $dir"
continue
fi
echo " config: $cf" echo " config: $cf"
docker compose -f "$cf" config --quiet docker compose -f "$cf" config --quiet
done done
echo "== Validate k8s manifests (kubectl dry-run) ==" echo "== Validate k8s manifests (kubectl dry-run=client) =="
for m in "${k8s_manifests[@]}"; do for m in "${k8s_manifests[@]}"; do
echo " apply --dry-run=client $m" echo " apply --dry-run=client $m"
kubectl apply --dry-run=client -f "$m" >/dev/null kubectl apply --dry-run=client -f "$m" >/dev/null
done done
for k in "${kustomize_apps[@]}"; do
echo " apply -k --dry-run=client $k"
kubectl apply -k "$k" --dry-run=client >/dev/null
done
echo "== Validate k8s manifests (kubectl dry-run=server) =="
for m in "${k8s_manifests[@]}"; do
echo " apply --dry-run=server $m"
kubectl apply --dry-run=server -f "$m" >/dev/null
done
for k in "${kustomize_apps[@]}"; do
echo " apply -k --dry-run=server $k"
kubectl apply -k "$k" --dry-run=server >/dev/null
done
echo "== Checking referenced Secrets exist =="
echo " (deploy never applies *secret*.yaml; create missing ones from the laptop)"
ref_secrets=()
if [ "${#k8s_manifests[@]}" -gt 0 ]; then
while IFS= read -r s; do
[ -n "$s" ] && ref_secrets+=("$s")
done < <(
{
grep -h -A1 -E 'secretRef:|secretKeyRef:' "${k8s_manifests[@]}" 2>/dev/null || true
grep -h -E 'secretName:' "${k8s_manifests[@]}" 2>/dev/null || true
} | grep -E 'name:' | sed -E 's/.*name:[[:space:]]*//' | tr -d '"'"'"' "'"'" | sed -E 's/[[:space:]]*#.*//' | awk 'NF' | sort -u || true
)
fi
missing_secrets=()
all_secrets="$(kubectl get secrets -A --no-headers -o custom-columns=:metadata.name 2>/dev/null || true)"
for s in "${ref_secrets[@]}"; do
if printf '%s\n' "$all_secrets" | grep -qx "$s"; then
echo " ok: $s"
else
echo " MISSING: $s"
missing_secrets+=("$s")
fi
done
if [ "${#missing_secrets[@]}" -gt 0 ]; then
echo "ERROR: ${#missing_secrets[@]} referenced Secret(s) not found in the cluster:"
printf ' - %s\n' "${missing_secrets[@]}"
echo "Create them manually from the laptop, e.g.:"
echo " kubectl apply -f SERVICE/k8s/secrets.yaml # see SERVICE/k8s/secrets.yaml.example"
exit 1
fi
echo "== Applying Kubernetes manifests ==" echo "== Applying Kubernetes manifests =="
ns_files=() ns_files=()
@@ -130,15 +238,19 @@ jobs:
echo " namespaces first: ${ns_files[*]}" echo " namespaces first: ${ns_files[*]}"
kubectl apply -f "${ns_files[@]}" kubectl apply -f "${ns_files[@]}"
fi fi
if [ -f "$repo/prometheus-stack/k8s/active" ]; then if [ -f "$repo/prometheus-stack/k8s/active" ] && ! is_disabled "$repo/prometheus-stack/k8s"; then
if [ ! -f "$repo/prometheus-stack/k8s/grafana-values.yaml" ]; then
echo "ERROR: prometheus-stack/k8s/grafana-values.yaml (gitignored) missing on workstation, restore it first."
exit 1
fi
echo "== Upgrading kube-prometheus-stack ==" echo "== Upgrading kube-prometheus-stack =="
helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \ helm upgrade --install prometheus-stack prometheus-community/kube-prometheus-stack \
--namespace prometheus \ --namespace prometheus \
--version 86.2.3 \ --version 86.2.3 \
--values "$repo/prometheus-stack/k8s/grafana-values.yaml" \ --values "$repo/prometheus-stack/k8s/grafana-values.yaml" \
--wait --wait --timeout 10m
fi fi
if [ -f "$repo/loki/k8s/active" ]; then if [ -f "$repo/loki/k8s/active" ] && ! is_disabled "$repo/loki/k8s"; then
echo "== Upgrading loki/alloy ==" echo "== Upgrading loki/alloy =="
helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true helm repo add grafana https://grafana.github.io/helm-charts >/dev/null 2>&1 || true
helm repo update grafana >/dev/null 2>&1 || true helm repo update grafana >/dev/null 2>&1 || true
@@ -146,12 +258,12 @@ jobs:
--version 7.3.0 \ --version 7.3.0 \
--namespace prometheus \ --namespace prometheus \
--values "$repo/loki/k8s/loki-values.yaml" \ --values "$repo/loki/k8s/loki-values.yaml" \
--wait --wait --timeout 10m
helm upgrade --install alloy grafana/alloy \ helm upgrade --install alloy grafana/alloy \
--version 1.12.1 \ --version 1.12.1 \
--namespace prometheus \ --namespace prometheus \
--values "$repo/loki/k8s/alloy-values.yaml" \ --values "$repo/loki/k8s/alloy-values.yaml" \
--wait --wait --timeout 10m
fi fi
if [ "${#other_files[@]}" -gt 0 ]; then if [ "${#other_files[@]}" -gt 0 ]; then
@@ -159,14 +271,30 @@ jobs:
kubectl apply "${prune_opts[@]}" -f "${other_files[@]}" kubectl apply "${prune_opts[@]}" -f "${other_files[@]}"
fi fi
for k in "${kustomize_apps[@]}"; do
echo "== Applying kustomize app: ${k#$repo/} =="
kubectl apply -k "$k"
done
if [ -f "$repo/userbot/k8s/active" ] && ! is_disabled "$repo/userbot"; then
echo "== userbot panel hook =="
if kubectl get secret userbot-common-secrets -n userbot >/dev/null 2>&1; then
echo " userbot-common-secrets already present in userbot ns, not touching"
elif kubectl get secret userbot-common-secrets -n default >/dev/null 2>&1; then
echo " bootstrapping userbot-common-secrets into userbot ns"
kubectl get secret userbot-common-secrets -n default -o json \
| jq 'del(.metadata.annotations,.metadata.creationTimestamp,.metadata.resourceVersion,.metadata.uid,.metadata.managedFields) | .metadata.namespace = "userbot"' \
| kubectl apply -f -
else
echo " WARNING: userbot-common-secrets missing in both default and userbot ns; create it manually from the laptop"
fi
kubectl rollout restart deployment/userbot-panel -n userbot
kubectl rollout status deployment/userbot-panel -n userbot --timeout=180s
fi
echo "== Redeploying docker compose stacks ==" echo "== Redeploying docker compose stacks =="
for cf in "${compose_stacks[@]}"; do for cf in "${compose_stacks[@]}"; do
dir=$(dirname "$cf") echo " compose: $cf"
if [ -f "$dir/k8s/active" ]; then
echo " skip (k8s-managed): $dir"
continue
fi
echo " compose: $dir"
if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then if grep -Eq '^\s+pull_policy:\s*build\b' "$cf"; then
docker compose -f "$cf" build docker compose -f "$cf" build
docker compose -f "$cf" push docker compose -f "$cf" push
+1
View File
@@ -0,0 +1 @@
1.56.0
+1 -1
View File
@@ -11,7 +11,7 @@ services:
retries: 5 retries: 5
playwright-service: playwright-service:
image: mcr.microsoft.com/playwright:v1.63.0-jammy image: mcr.microsoft.com/playwright:v1.56.0-jammy
restart: unless-stopped restart: unless-stopped
command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws command: npx -y playwright@1.56.0 run-server --port 3000 --path /ws
+77
View File
@@ -0,0 +1,77 @@
apiVersion: monitoring.coreos.com/v1
kind: PrometheusRule
metadata:
name: edu-master-webinar
namespace: edu-master
labels:
release: prometheus-stack
spec:
groups:
- name: edu_master.webinar
rules:
# No successful webinar check for 5m (~2-3 missed 2-min checks).
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
- alert: WebinarCheckerNoSuccessfulCheck
expr: |
(time() - webinar_check_last_success_timestamp_seconds > 300)
and (webinar_check_last_run_timestamp_seconds > 0)
for: 2m
labels:
severity: critical
annotations:
summary: "Webinar checker has no successful check for 5m"
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
- alert: WebinarCheckerConsecutiveFailures
expr: |
webinar_check_consecutive_failures >= 3
for: 5m
labels:
severity: critical
annotations:
summary: "Webinar checker failing consecutively"
description: "edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace=\"edu-master\", container=\"webinar-checker\"})."
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
- alert: WebinarCheckerScrapeDown
expr: |
absent(webinar_check_last_run_timestamp_seconds) == 1
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker metrics missing"
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
- alert: EduPhpsessidMissing
expr: |
edu_phpsessid_present == 0
for: 10m
labels:
severity: critical
annotations:
summary: "EDU_PHPSESSID missing"
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
# Hard deps: checker and playwright deployments unavailable.
- alert: WebinarCheckerDeploymentDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Webinar checker deployment unavailable"
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
- alert: PlaywrightServiceDown
expr: |
kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0
for: 10m
labels:
severity: critical
annotations:
summary: "Playwright service unavailable"
description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it."
+2 -1
View File
@@ -17,7 +17,8 @@ spec:
spec: spec:
containers: containers:
- name: playwright - name: playwright
image: mcr.microsoft.com/playwright:v1.63.0-jammy # renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker
image: mcr.microsoft.com/playwright:v1.56.0-jammy
imagePullPolicy: IfNotPresent imagePullPolicy: IfNotPresent
command: command:
- npx - npx
+2
View File
@@ -21,6 +21,8 @@ stringData:
WEBINAR_TELEGRAM_TOKEN: "" WEBINAR_TELEGRAM_TOKEN: ""
WEBINAR_ADMIN_ID: "" WEBINAR_ADMIN_ID: ""
WEBINAR_CHECK_INTERVAL: "60" WEBINAR_CHECK_INTERVAL: "60"
# Prometheus metrics endpoint (scraped via ServiceMonitor, alerts in k8s/alerts.yaml)
METRICS_PORT: "8000"
# Database # Database
REDIS_HOST: "redis" REDIS_HOST: "redis"
REDIS_PORT: "6379" REDIS_PORT: "6379"
+15
View File
@@ -0,0 +1,15 @@
apiVersion: v1
kind: Service
metadata:
name: webinar-checker
namespace: edu-master
labels:
app: edu-master-webinar-checker
spec:
selector:
app: edu-master-webinar-checker
ports:
- name: metrics
port: 8000
targetPort: metrics
protocol: TCP
+16
View File
@@ -0,0 +1,16 @@
apiVersion: monitoring.coreos.com/v1
kind: ServiceMonitor
metadata:
name: webinar-checker
namespace: edu-master
labels:
release: prometheus-stack
spec:
selector:
matchLabels:
app: edu-master-webinar-checker
endpoints:
- port: metrics
path: /metrics
interval: 30s
scrapeTimeout: 10s
+4
View File
@@ -47,6 +47,10 @@ spec:
- name: webinar-checker - name: webinar-checker
image: gcr.forust.xyz/forust/webinar-checker:latest image: gcr.forust.xyz/forust/webinar-checker:latest
imagePullPolicy: Always imagePullPolicy: Always
ports:
- name: metrics
containerPort: 8000
protocol: TCP
envFrom: envFrom:
- secretRef: - secretRef:
name: edu-master-secrets name: edu-master-secrets
+5 -2
View File
@@ -2,8 +2,11 @@ FROM python:3.11-slim
WORKDIR /app WORKDIR /app
# Install dependencies # renovate: datasource=pypi depName=playwright versioning=pep440
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==1.56.0 redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10" ARG PLAYWRIGHT_VERSION=1.56.0
# Install dependencies - PLAYWRIGHT_VERSION is single-source, renovate updates ARG above and all other places via regexManagers
RUN pip install --no-cache-dir pip==25.0.1 && pip install --no-cache-dir playwright==${PLAYWRIGHT_VERSION} redis==5.2.1 requests==2.32.3 "python-telegram-bot[job-queue]==21.10"
COPY checker.py . COPY checker.py .
+148 -15
View File
@@ -1,12 +1,15 @@
import asyncio
import contextlib import contextlib
import json import json
import logging import logging
import os import os
import re import re
import tempfile import tempfile
import threading
import time import time
from datetime import datetime, timedelta from datetime import datetime, timedelta
from html import escape from html import escape
from http.server import BaseHTTPRequestHandler, HTTPServer
import redis import redis
from playwright.async_api import async_playwright from playwright.async_api import async_playwright
@@ -48,6 +51,106 @@ USER_AGENT = _env(
) )
WEBINAR_TELEGRAM_TOKEN = _env('WEBINAR_TELEGRAM_TOKEN') WEBINAR_TELEGRAM_TOKEN = _env('WEBINAR_TELEGRAM_TOKEN')
ADMIN_ID = int(_env('WEBINAR_ADMIN_ID', '0')) ADMIN_ID = int(_env('WEBINAR_ADMIN_ID', '0'))
METRICS_PORT = int(_env('METRICS_PORT', '8000'))
# --- Prometheus metrics (stdlib only, no extra deps) ---
# Scraped by prometheus-stack via ServiceMonitor (edu_master/k8s/servicemonitor.yaml).
# Critical alerts in edu_master/k8s/alerts.yaml fire to Telegram via Alertmanager.
_METRICS_LOCK = threading.Lock()
_METRICS = {
'last_run': 0.0, # Unix ts of last check start
'last_success': 0.0, # Unix ts of last successful check
'last_duration': 0.0, # Duration of last check in seconds
'success_total': 0,
'failure_total': 0,
'consecutive_failures': 0,
'phpsessid_present': 1, # 1 if EDU_PHPSESSID found in redis, else 0
}
def _metric_check_start():
with _METRICS_LOCK:
_METRICS['last_run'] = time.time()
def _metric_check_ok(duration: float):
now = time.time()
with _METRICS_LOCK:
_METRICS['last_success'] = now
_METRICS['last_duration'] = duration
_METRICS['success_total'] += 1
_METRICS['consecutive_failures'] = 0
_METRICS['phpsessid_present'] = 1
def _metric_check_fail(duration: float, phpsessid_missing: bool = False):
with _METRICS_LOCK:
_METRICS['last_duration'] = duration
_METRICS['failure_total'] += 1
_METRICS['consecutive_failures'] += 1
_METRICS['phpsessid_present'] = 0 if phpsessid_missing else 1
def _metrics_render() -> bytes:
with _METRICS_LOCK:
m = dict(_METRICS)
lines = [
'# HELP webinar_check_last_run_timestamp_seconds Unix timestamp of last webinar check start.',
'# TYPE webinar_check_last_run_timestamp_seconds gauge',
f'webinar_check_last_run_timestamp_seconds {m["last_run"]}',
'# HELP webinar_check_last_success_timestamp_seconds Unix timestamp of last successful webinar check.',
'# TYPE webinar_check_last_success_timestamp_seconds gauge',
f'webinar_check_last_success_timestamp_seconds {m["last_success"]}',
'# HELP webinar_check_last_duration_seconds Duration of last webinar check in seconds.',
'# TYPE webinar_check_last_duration_seconds gauge',
f'webinar_check_last_duration_seconds {m["last_duration"]}',
'# HELP webinar_check_success_total Total successful webinar checks.',
'# TYPE webinar_check_success_total counter',
f'webinar_check_success_total {m["success_total"]}',
'# HELP webinar_check_failure_total Total failed webinar checks (timeout, playwright error, page error).',
'# TYPE webinar_check_failure_total counter',
f'webinar_check_failure_total {m["failure_total"]}',
'# HELP webinar_check_consecutive_failures Consecutive failed webinar checks (reset on success).',
'# TYPE webinar_check_consecutive_failures gauge',
f'webinar_check_consecutive_failures {m["consecutive_failures"]}',
'# HELP edu_phpsessid_present 1 if EDU_PHPSESSID exists in redis, 0 otherwise.',
'# TYPE edu_phpsessid_present gauge',
f'edu_phpsessid_present {m["phpsessid_present"]}',
]
return ('\n'.join(lines) + '\n').encode()
class _MetricsHandler(BaseHTTPRequestHandler):
def do_GET(self):
if self.path == '/metrics':
body = _metrics_render()
self.send_response(200)
self.send_header('Content-Type', 'text/plain; version=0.0.4')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
elif self.path in ('/healthz', '/health'):
body = b'ok\n'
self.send_response(200)
self.send_header('Content-Type', 'text/plain')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
else:
self.send_response(404)
self.end_headers()
def log_message(self, *args):
pass # keep bot logs clean
def start_metrics_server(port: int = METRICS_PORT):
server = HTTPServer(('0.0.0.0', port), _MetricsHandler) # noqa: S104 - k8s ServiceMonitor scrapes pod IP
thread = threading.Thread(target=server.serve_forever, name='metrics-server', daemon=True)
thread.start()
logger.info(f'Metrics server listening on :{port}/metrics')
return server
# Redis Keys # Redis Keys
KEY_WHITELIST = 'bot:whitelist' KEY_WHITELIST = 'bot:whitelist'
@@ -597,8 +700,9 @@ async def _collect_event_times(page) -> dict:
async def fetch_diary_data(phpsessid: str) -> dict | None: async def fetch_diary_data(phpsessid: str) -> dict | None:
logger.info('Fetching diary data via Playwright...') logger.info('Fetching diary data via Playwright...')
try: try:
async with asyncio.timeout(60):
async with async_playwright() as p: async with async_playwright() as p:
browser = await p.chromium.connect(PLAYWRIGHT_WS) browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
try: try:
context_browser = await browser.new_context(user_agent=USER_AGENT) context_browser = await browser.new_context(user_agent=USER_AGENT)
await context_browser.add_cookies( await context_browser.add_cookies(
@@ -607,7 +711,7 @@ async def fetch_diary_data(phpsessid: str) -> dict | None:
page = await context_browser.new_page() page = await context_browser.new_page()
try: try:
await page.goto(DIARY_URL, wait_until='domcontentloaded') await asyncio.wait_for(page.goto(DIARY_URL, wait_until='domcontentloaded'), timeout=30)
await page.wait_for_selector('table.calendar', timeout=10000) await page.wait_for_selector('table.calendar', timeout=10000)
await page.wait_for_timeout(1500) await page.wait_for_timeout(1500)
@@ -646,10 +750,16 @@ async def fetch_diary_data(phpsessid: str) -> dict | None:
logger.error(f'Error parsing diary: {e}') logger.error(f'Error parsing diary: {e}')
return None return None
finally: finally:
await page.close() with contextlib.suppress(Exception):
await context_browser.close() await asyncio.wait_for(page.close(), timeout=5)
with contextlib.suppress(Exception):
await asyncio.wait_for(context_browser.close(), timeout=5)
finally: finally:
await browser.close() with contextlib.suppress(Exception):
await asyncio.wait_for(browser.close(), timeout=5)
except TimeoutError:
logger.error('Diary fetch timed out (60s)')
return None
except Exception as e: except Exception as e:
logger.error(f'Playwright error in diary fetch: {e}') logger.error(f'Playwright error in diary fetch: {e}')
return None return None
@@ -931,8 +1041,9 @@ def _parse_schedule_html(table_html: str) -> dict:
async def fetch_schedule_data(phpsessid: str) -> dict | None: async def fetch_schedule_data(phpsessid: str) -> dict | None:
logger.info('Fetching schedule data via Playwright...') logger.info('Fetching schedule data via Playwright...')
try: try:
async with asyncio.timeout(60):
async with async_playwright() as p: async with async_playwright() as p:
browser = await p.chromium.connect(PLAYWRIGHT_WS) browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
try: try:
context_browser = await browser.new_context(user_agent=USER_AGENT) context_browser = await browser.new_context(user_agent=USER_AGENT)
await context_browser.add_cookies( await context_browser.add_cookies(
@@ -941,7 +1052,7 @@ async def fetch_schedule_data(phpsessid: str) -> dict | None:
page = await context_browser.new_page() page = await context_browser.new_page()
try: try:
await page.goto(SCHEDULE_URL, wait_until='domcontentloaded') await asyncio.wait_for(page.goto(SCHEDULE_URL, wait_until='domcontentloaded'), timeout=30)
await page.wait_for_selector('table.schedule-table', timeout=10000) await page.wait_for_selector('table.schedule-table', timeout=10000)
await page.wait_for_timeout(1500) await page.wait_for_timeout(1500)
@@ -969,10 +1080,16 @@ async def fetch_schedule_data(phpsessid: str) -> dict | None:
logger.error(f'Error parsing schedule: {e}') logger.error(f'Error parsing schedule: {e}')
return None return None
finally: finally:
await page.close() with contextlib.suppress(Exception):
await context_browser.close() await asyncio.wait_for(page.close(), timeout=5)
with contextlib.suppress(Exception):
await asyncio.wait_for(context_browser.close(), timeout=5)
finally: finally:
await browser.close() with contextlib.suppress(Exception):
await asyncio.wait_for(browser.close(), timeout=5)
except TimeoutError:
logger.error('Schedule fetch timed out (60s)')
return None
except Exception as e: except Exception as e:
logger.error(f'Playwright error in schedule fetch: {e}') logger.error(f'Playwright error in schedule fetch: {e}')
return None return None
@@ -1485,10 +1602,13 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
int: Number of webinars found, or None if check failed int: Number of webinars found, or None if check failed
""" """
logger.info('Running webinar check...') logger.info('Running webinar check...')
_t0 = time.time()
_metric_check_start()
phpsessid = redis_client.get(KEY_PHPSESSID) phpsessid = redis_client.get(KEY_PHPSESSID)
if not phpsessid: if not phpsessid:
logger.warning('PHPSESSID missing. Skipping check.') logger.warning('PHPSESSID missing. Skipping check.')
_metric_check_fail(time.time() - _t0, phpsessid_missing=True)
# --- DEBUG LOGGING --- # --- DEBUG LOGGING ---
try: try:
with open('phpsessid_missing.log', 'a') as f: with open('phpsessid_missing.log', 'a') as f:
@@ -1502,9 +1622,10 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
content = '' content = ''
try: try:
async with asyncio.timeout(90):
async with async_playwright() as p: async with async_playwright() as p:
# Connect to remote Playwright service # Connect to remote Playwright service
browser = await p.chromium.connect(PLAYWRIGHT_WS) browser = await asyncio.wait_for(p.chromium.connect(PLAYWRIGHT_WS), timeout=15)
try: try:
# Create browser context with user agent # Create browser context with user agent
@@ -1520,7 +1641,7 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
try: try:
# Navigate to webinar page # Navigate to webinar page
await page.goto(WEBINAR_URL, wait_until='domcontentloaded') await asyncio.wait_for(page.goto(WEBINAR_URL, wait_until='domcontentloaded'), timeout=30)
# Wait for the table to load # Wait for the table to load
await page.wait_for_selector('#meetings table', timeout=10000) await page.wait_for_selector('#meetings table', timeout=10000)
@@ -1564,16 +1685,25 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
if page and not content: if page and not content:
content = await page.content() content = await page.content()
_metric_check_fail(time.time() - _t0)
return None return None
finally: finally:
await page.close() with contextlib.suppress(Exception):
await context_browser.close() await asyncio.wait_for(page.close(), timeout=5)
with contextlib.suppress(Exception):
await asyncio.wait_for(context_browser.close(), timeout=5)
finally: finally:
await browser.close() with contextlib.suppress(Exception):
await asyncio.wait_for(browser.close(), timeout=5)
except TimeoutError:
logger.error('Webinar check timed out after 90s (playwright hang)')
_metric_check_fail(time.time() - _t0)
return None
except Exception as e: except Exception as e:
logger.error(f'Playwright error: {e}') logger.error(f'Playwright error: {e}')
_metric_check_fail(time.time() - _t0)
return None return None
# --- DEBUG LOGGING (Saving last response content) --- # --- DEBUG LOGGING (Saving last response content) ---
@@ -1637,6 +1767,7 @@ async def check_webinars_job(context: ContextTypes.DEFAULT_TYPE):
else: else:
logger.info(f'Found {len(current_webinars)} webinar(s), but all are already known') logger.info(f'Found {len(current_webinars)} webinar(s), but all are already known')
_metric_check_ok(time.time() - _t0)
return len(current_webinars) return len(current_webinars)
@@ -1680,6 +1811,8 @@ def main():
job_queue = app.job_queue job_queue = app.job_queue
job_queue.run_repeating(check_webinars_job, interval=WEBINAR_CHECK_INTERVAL, first=10) job_queue.run_repeating(check_webinars_job, interval=WEBINAR_CHECK_INTERVAL, first=10)
start_metrics_server()
logger.info('Bot started polling...') logger.info('Bot started polling...')
app.run_polling() app.run_polling()
+1 -1
View File
@@ -31,7 +31,7 @@ spec:
spec: spec:
containers: containers:
- name: gitea - name: gitea
image: docker.gitea.com/gitea:1.27.3 image: gitea/gitea:1.27.3
envFrom: envFrom:
- configMapRef: - configMapRef:
name: gitea-config name: gitea-config
+1
View File
@@ -3,3 +3,4 @@ AUTHENTIK_DB_PASSWORD=
GITEA_DB_PASSWORD= GITEA_DB_PASSWORD=
NETRONOME_DB_PASSWORD= NETRONOME_DB_PASSWORD=
PENPOT_DB_PASSWORD= PENPOT_DB_PASSWORD=
STATUSPAGE_DB_PASSWORD=
+13 -14
View File
@@ -1,23 +1,22 @@
# Shared PostgreSQL # Shared PostgreSQL
This directory contains a PostgreSQL 15 deployment draft for Authentik, Gitea, This directory contains the shared PostgreSQL 17 deployment for Authentik,
Netronome, and Penpot. It creates one database and one login role per service; Gitea, Netronome, and Statuspage. It creates one database and one login role
it does not migrate existing data or change application connection settings. per service. Per-service standalone databases were removed after the
migration (Sep 2026); Penpot stays on its own compose PostgreSQL (archived,
not part of the shared instance).
## Compatibility baseline ## Compatibility baseline
| Service | Current application | Current standalone PostgreSQL | Common PostgreSQL 15 | | Service | Current application | Shared PostgreSQL 17 |
| --------- | ------------------- | ----------------------------: | ------------------------------------------------------------------------------------ | | --------- | ------------------- | -------------------- |
| Authentik | 2025.10.2 | 15 | Supported (Authentik requires 14+) | | Authentik | 2025.10.x | Supported (Authentik requires 14+) |
| Gitea | 1.27.3 | 14 | Supported (Gitea requires 12+) | | Gitea | 1.27.3 | Supported (Gitea requires 12+) |
| Netronome | 0.14.0 | 17 | Validate in staging; upstream's example uses 17 but no 17-only feature is documented | | Netronome | 0.14.0 | Supported (upstream's example uses 17) |
| Penpot | 2.17.2 | 15 | Supported by the official deployment | | Statuspage| custom | Supported |
PostgreSQL 15 is the conservative common major. A major-version downgrade or A major-version change must use a logical dump/restore; changing only the
change must use a logical dump/restore; changing only the image tag while image tag while keeping a data directory is not supported.
keeping a data directory is not supported. Back up and migrate one application
at a time, starting with Netronome because its current standalone deployment
uses PostgreSQL 17.
For Compose, copy `.env.example` to `.env`, set all passwords, and start it with For Compose, copy `.env.example` to `.env`, set all passwords, and start it with
`docker compose -f shared-compose.yaml up -d`. This file is intentionally not `docker compose -f shared-compose.yaml up -d`. This file is intentionally not
+2 -1
View File
@@ -5,12 +5,12 @@ set -euo pipefail
: "${GITEA_DB_PASSWORD:?GITEA_DB_PASSWORD is required}" : "${GITEA_DB_PASSWORD:?GITEA_DB_PASSWORD is required}"
: "${NETRONOME_DB_PASSWORD:?NETRONOME_DB_PASSWORD is required}" : "${NETRONOME_DB_PASSWORD:?NETRONOME_DB_PASSWORD is required}"
: "${PENPOT_DB_PASSWORD:?PENPOT_DB_PASSWORD is required}" : "${PENPOT_DB_PASSWORD:?PENPOT_DB_PASSWORD is required}"
: "${STATUSPAGE_DB_PASSWORD:?STATUSPAGE_DB_PASSWORD is required}"
create_role_and_database() { create_role_and_database() {
local role="$1" local role="$1"
local database="$2" local database="$2"
local password="$3" local password="$3"
psql --username "$POSTGRES_USER" --dbname postgres \ psql --username "$POSTGRES_USER" --dbname postgres \
-v role="$role" -v database="$database" -v password="$password" \ -v role="$role" -v database="$database" -v password="$password" \
<<'SQL' <<'SQL'
@@ -25,3 +25,4 @@ create_role_and_database authentik authentik "$AUTHENTIK_DB_PASSWORD"
create_role_and_database gitea gitea "$GITEA_DB_PASSWORD" create_role_and_database gitea gitea "$GITEA_DB_PASSWORD"
create_role_and_database netronome netronome "$NETRONOME_DB_PASSWORD" create_role_and_database netronome netronome "$NETRONOME_DB_PASSWORD"
create_role_and_database penpot penpot "$PENPOT_DB_PASSWORD" create_role_and_database penpot penpot "$PENPOT_DB_PASSWORD"
create_role_and_database statuspage statuspage "$STATUSPAGE_DB_PASSWORD"
+12
View File
@@ -60,11 +60,23 @@ spec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"] command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
initialDelaySeconds: 10 initialDelaySeconds: 10
periodSeconds: 10 periodSeconds: 10
startupProbe:
exec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
failureThreshold: 30
periodSeconds: 10
livenessProbe: livenessProbe:
exec: exec:
command: ["pg_isready", "-U", "postgres", "-d", "postgres"] command: ["pg_isready", "-U", "postgres", "-d", "postgres"]
initialDelaySeconds: 30 initialDelaySeconds: 30
periodSeconds: 20 periodSeconds: 20
resources:
requests:
memory: "512Mi"
cpu: "500m"
limits:
memory: "2Gi"
cpu: "2000m"
volumes: volumes:
- name: postgres-data - name: postgres-data
persistentVolumeClaim: persistentVolumeClaim:
+1 -1
View File
@@ -1,6 +1,6 @@
services: services:
postgres: postgres:
image: postgres:15.19-alpine image: postgres:17.11-alpine
container_name: homelab-postgres container_name: homelab-postgres
restart: unless-stopped restart: unless-stopped
env_file: env_file:
+30 -1
View File
@@ -1,14 +1,43 @@
{ {
"$schema": "https://docs.renovatebot.com/renovate-schema.json", "$schema": "https://docs.renovatebot.com/renovate-schema.json",
"extends": ["config:recommended"], "extends": ["config:recommended"],
"enabledManagers": ["docker-compose", "kubernetes", "helm-values"], "enabledManagers": ["dockerfile", "docker-compose", "kubernetes", "helm-values", "custom.regex"],
"helm-values": { "helm-values": {
"managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"] "managerFilePatterns": ["/k8s/.+values\\.ya?ml$/"]
}, },
"kubernetes": { "kubernetes": {
"managerFilePatterns": ["/k8s/.+\\.ya?ml$/"] "managerFilePatterns": ["/k8s/.+\\.ya?ml$/"]
}, },
"customManagers": [
{
"customType": "regex",
"description": "singlesource: playwright npm version pinned in npx command (k8s + compose)",
"fileMatch": ["^edu_master/k8s/playwright\\.yaml$", "^edu_master/compose\\.yaml$"],
"matchStrings": ["playwright@(?<currentValue>\\d+\\.\\d+\\.\\d+)"],
"datasourceTemplate": "npm",
"depNameTemplate": "playwright"
},
{
"customType": "regex",
"description": "singlesource: PLAYWRIGHT_VERSION file",
"fileMatch": ["^edu_master/PLAYWRIGHT_VERSION$"],
"matchStrings": ["^(?<currentValue>\\d+\\.\\d+\\.\\d+)$"],
"datasourceTemplate": "pypi",
"depNameTemplate": "playwright"
}
],
"packageRules": [ "packageRules": [
{
"description": "singlesource playwright - use whichever version is found, keep docker+pypi+npm in sync",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"groupName": "playwright singlesource",
"groupSlug": "playwright"
},
{
"description": "playwright must not automerge - version skew breaks WS handshake (checker.py:1523 vs playwright.yaml:20)",
"matchPackageNames": ["playwright", "mcr.microsoft.com/playwright"],
"automerge": false
},
{ {
"description": "Keep private homelab images unchanged", "description": "Keep private homelab images unchanged",
"matchDatasources": ["docker"], "matchDatasources": ["docker"],