From 2f891a5d31d3405418f2a4b6a79e5cae25d377a7 Mon Sep 17 00:00:00 2001 From: mr-forust Date: Mon, 28 Sep 2026 12:35:16 +0200 Subject: [PATCH] fix(postgres): give probes room on an I/O-bound single node pg_isready with the 1s default times out under I/O stall and kubelet kills a healthy postgres mid-recovery; each kill restarts a multi-minute fsync from zero and loops forever. readiness/liveness timeout 5s, liveness threshold 5, startup budget 15min. --- postgres/k8s/postgres.yaml | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/postgres/k8s/postgres.yaml b/postgres/k8s/postgres.yaml index ae83540..51112c8 100644 --- a/postgres/k8s/postgres.yaml +++ b/postgres/k8s/postgres.yaml @@ -60,16 +60,26 @@ spec: command: ["pg_isready", "-U", "postgres", "-d", "postgres"] initialDelaySeconds: 10 periodSeconds: 10 + # Generous timeout: on an I/O-bound single node even exec can take + # seconds, and a 1s default kills a healthy postgres mid-recovery. + timeoutSeconds: 5 startupProbe: exec: command: ["pg_isready", "-U", "postgres", "-d", "postgres"] - failureThreshold: 30 + # Crash recovery on an I/O-starved single node can fsync for 10+ + # minutes; killing postgres mid-recovery restarts the fsync from + # zero and loops forever. 90x10s = 15 minutes of grace. + failureThreshold: 90 periodSeconds: 10 livenessProbe: exec: command: ["pg_isready", "-U", "postgres", "-d", "postgres"] initialDelaySeconds: 30 periodSeconds: 20 + # Same I/O reasoning as readiness, plus more misses before a kill: + # restarting postgres on a loaded node only makes recovery longer. + timeoutSeconds: 5 + failureThreshold: 5 resources: requests: memory: "512Mi"