apiVersion: monitoring.coreos.com/v1 kind: PrometheusRule metadata: name: edu-master-webinar namespace: edu-master labels: release: prometheus-stack spec: groups: - name: edu_master.webinar rules: # No successful webinar check for 5m (~2-3 missed 2-min checks). # Catches: playwright hangs/timeouts, version skew, site changes, hung job. # The last_success > 0 guard is mandatory: checker.py initialises # last_success to 0, so without it `time() - 0` equals the current epoch # and humanizeDuration renders ~20722d on every pod restart. Keep the # duration expression on the left so $value stays the real gap. - alert: WebinarCheckerNoSuccessfulCheck expr: | ((time() - webinar_check_last_success_timestamp_seconds) > 300) and (webinar_check_last_success_timestamp_seconds > 0) and (webinar_check_last_run_timestamp_seconds > 0) for: 2m labels: severity: critical annotations: summary: "Webinar checker has no successful check for 5m" description: "edu-master/webinar-checker: last successful webinar check was {{ $value | humanizeDuration }} ago. Checks are failing or hanging (see consecutive failures alert). Notifications about new webinars are NOT being sent." # Checks are running but none has ever succeeded since pod start. # Split out from the rule above so a zeroed gauge never feeds # humanizeDuration. - alert: WebinarCheckerNeverSucceeded expr: | (webinar_check_last_success_timestamp_seconds == 0) and (webinar_check_last_run_timestamp_seconds > 0) for: 10m labels: severity: critical annotations: summary: "Webinar checker has never completed a successful check" description: 'edu-master/webinar-checker: checks have been running for 10m but not one has ever succeeded since the pod started, so every check is failing. Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).' # Fast path: 3 consecutive failures (~6+ min at 2-min interval). - alert: WebinarCheckerConsecutiveFailures expr: | webinar_check_consecutive_failures >= 3 for: 5m labels: severity: critical annotations: summary: "Webinar checker failing consecutively" description: 'edu-master/webinar-checker: {{ $value }} consecutive webinar check failures (timeout / playwright error / page error). Check pod logs (Loki: {namespace="edu-master", container="webinar-checker"}).' # Metrics endpoint not scraped for 10m: pod down, metrics server dead, or ServiceMonitor broken. - alert: WebinarCheckerScrapeDown expr: | absent(webinar_check_last_run_timestamp_seconds) == 1 for: 10m labels: severity: critical annotations: summary: "Webinar checker metrics missing" description: "edu-master/webinar-checker: no metrics series for 10m. Pod may be down, metrics server dead, or ServiceMonitor/Service broken. Webinar checks are unobserved." # EDU session lost: session-keeper down or credentials expired. Without PHPSESSID every check is skipped. - alert: EduPhpsessidMissing expr: | edu_phpsessid_present == 0 for: 10m labels: severity: critical annotations: summary: "EDU_PHPSESSID missing" description: "edu-master: EDU_PHPSESSID absent from redis for 10m. Webinar/diari/schedule checks are all skipped. Check session-keeper logs and EDU credentials." # Hard deps: checker and playwright deployments unavailable. - alert: WebinarCheckerDeploymentDown expr: | kube_deployment_status_replicas_unavailable{deployment="webinar-checker", namespace="edu-master"} > 0 for: 10m labels: severity: critical annotations: summary: "Webinar checker deployment unavailable" description: "edu-master/webinar-checker deployment has {{ $value }} unavailable replica(s) for 10m." - alert: PlaywrightServiceDown expr: | kube_deployment_status_replicas_unavailable{deployment="playwright-service", namespace="edu-master"} > 0 for: 10m labels: severity: critical annotations: summary: "Playwright service unavailable" description: "edu-master/playwright-service deployment has {{ $value }} unavailable replica(s) for 10m. All webinar/diari/schedule checks fail without it."