renovate-ci / validate-renovate (push) Skipped
ci / lint-compose (push) Successful in 14s
ci / lint-actionlint (push) Successful in 6s
ci / lint-shellcheck (push) Successful in 13s
ci / lint-prettier (push) Successful in 19s
ci / lint-ruff (push) Successful in 9s
ci / lint-yaml (push) Successful in 10s
ci / lint-dockerfiles (push) Successful in 6s
ci / validate (push) Successful in 10s
ci / build (push) Successful in 30s
116 lines
5.3 KiB
YAML
116 lines
5.3 KiB
YAML
apiVersion: monitoring.coreos.com/v1
|
|
kind: PrometheusRule
|
|
metadata:
|
|
name: edu-master-webinar
|
|
namespace: edu-master
|
|
labels:
|
|
release: prometheus-stack
|
|
spec:
|
|
groups:
|
|
- name: edu_master.webinar
|
|
rules:
|
|
# No successful webinar check for 5m (~2-3 missed 2-min checks).
|
|
# Catches: playwright hangs/timeouts, version skew, site changes, hung job.
|
|
# The last_success > 0 guard is mandatory: checker.py initialises
|
|
# last_success to 0, so without it `time() - 0` equals the current epoch
|
|
# and humanizeDuration renders ~20722d on every pod restart. Keep the
|
|
# duration expression on the left so $value stays the real gap.
|
|
- alert: WebinarCheckerNoSuccessfulCheck
|
|
expr: |
|
|
((time() - webinar_check_last_success_timestamp_seconds) > 300)
|
|
and (webinar_check_last_success_timestamp_seconds > 0)
|
|
and (webinar_check_last_run_timestamp_seconds > 0)
|
|
for: 2m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "Webinar checker has no successful check for 5m"
|
|
description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent."
|
|
|
|
# Checks are running but none has ever succeeded since pod start.
|
|
# Split out from the rule above so a zeroed gauge never feeds
|
|
# humanizeDuration.
|
|
- alert: WebinarCheckerNeverSucceeded
|
|
expr: |
|
|
(webinar_check_last_success_timestamp_seconds == 0)
|
|
and (webinar_check_last_run_timestamp_seconds > 0)
|
|
for: 10m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "Webinar checker has never completed a successful check"
|
|
description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
|
|
|
|
# Fast path: 3 consecutive failures (~6+ min at 2-min interval).
|
|
- alert: WebinarCheckerConsecutiveFailures
|
|
expr: |
|
|
webinar_check_consecutive_failures >= 3
|
|
for: 5m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "Webinar checker failing consecutively"
|
|
description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / http error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).'
|
|
|
|
- alert: WebinarCheckerNeverStarted
|
|
expr: |
|
|
(time() - edu_process_start > 120)
|
|
and (webinar_check_last_run_timestamp_seconds == 0)
|
|
for: 2m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "Webinar checker job has not started"
|
|
description: "The process exposes metrics but its webinar job has never started."
|
|
|
|
- alert: WebinarDeliveryPending
|
|
expr: edu_delivery_pending > 0
|
|
for: 5m
|
|
labels:
|
|
severity: warning
|
|
annotations:
|
|
summary: "Webinar notifications await delivery"
|
|
description: "Telegram delivery has pending recipients. Check delivery failures and retry status."
|
|
|
|
- alert: EduRedisUnavailable
|
|
expr: edu_redis_connected == 0
|
|
for: 2m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "EDU checker cannot reach Redis"
|
|
description: "Redis health checks are failing; checker commands and delivery may be unavailable."
|
|
|
|
# Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken.
|
|
- alert: WebinarCheckerScrapeDown
|
|
expr: |
|
|
absent(webinar_check_last_run_timestamp_seconds) == 1
|
|
for: 10m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "Webinar checker metrics missing"
|
|
description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved."
|
|
|
|
# EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped.
|
|
- alert: EduPhpsessidMissing
|
|
expr: |
|
|
edu_phpsessid_present == 0
|
|
for: 10m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "EDU_PHPSESSID missing"
|
|
description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials."
|
|
|
|
# Hard deps: checker deployment unavailable.
|
|
- alert: WebinarCheckerDeploymentDown
|
|
expr: |
|
|
kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0
|
|
for: 10m
|
|
labels:
|
|
severity: critical
|
|
annotations:
|
|
summary: "Webinar checker deployment unavailable"
|
|
description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m."
|