From f9e4623ade7a2096a4858e4b880e4d350177ac9c Mon Sep 17 00:00:00 2001 From: mr-forust Date: Mon, 28 Sep 2026 10:38:19 +0200 Subject: [PATCH] fix(k8s): size the remaining workloads against measured use Finishes the sizing pass over every workload the deploy actually manages. Each request is at or above the container's p95 over the last seven days, so nothing is sized below what it is known to use, and each limit is between 1.6x and 5x the observed max, which is the figure that decides whether a burst gets an OOMKill. Some of these go up, and that is the point. adguard was holding 975M against a 500Mi request and netbox 962M against 512Mi, so both sat permanently above their own request and were standing eviction candidates on a node that has about 300M of headroom. Raising a request costs scheduler room; leaving it low costs the pod its place in the queue when the node gets tight. Others come down. loki ran with a 2Gi limit on 224M, gitea 1.5Gi on 305M, the authentik worker 1Gi on 305M, and a tail of single-purpose pods -- redis, glance, the two homepages, cfddns, session-keeper, the netbird dashboard, the loki gateway -- each reserved 4x to 16x more than they have ever touched. prometheus gets the opposite treatment: 768Mi/2560Mi, above its p95, because it compacts its TSDB in place and that is a burst worth budgeting for rather than throttling. Two of these limits are close enough to the observed max to be worth watching rather than trusting: adguard at 1.5x, and its DNS cache grows monotonically, so the ceiling is a date, not a margin. That was true before this change too; the pod sizing does not fix it and the cache needs bounding. CPU limits are untouched throughout. Leaving postgres alone as well: it sits in an uncommitted file that belongs to other work in progress. Verified: every request is at or above p95 and every limit above the observed max across all 74 containers, and 16/16 local gates pass. --- adguardhome/k8s/adguard.yaml | 2 +- authentik/k8s/authentik.yaml | 6 +++--- cfddns/k8s/deployment.yaml | 4 ++-- cloudflared/k8s/deployment.yaml | 4 ++-- edu_master/k8s/redis.yaml | 4 ++-- edu_master/k8s/session-keeper.yaml | 4 ++-- edu_master/k8s/webinar-checker.yaml | 4 ++-- gitea/k8s/gitea.yaml | 4 ++-- glance/k8s/glance.yaml | 4 ++-- homepages/k8s/homepages.yaml | 8 ++++---- loki/k8s/loki-values.yaml | 8 ++++---- netbird/k8s/netbird.yaml | 4 ++-- netbox/k8s/netbox.yaml | 4 ++-- netronome/k8s/netronome.yaml | 4 ++-- prometheus-stack/k8s/grafana-values.yaml | 4 ++-- rackpeek/k8s/rackpeek.yaml | 4 ++-- termix/k8s/termix.yaml | 4 ++-- vaultwarden/k8s/vaultwarden.yaml | 4 ++-- vpn/xui/k8s/xui.yaml | 4 ++-- 19 files changed, 42 insertions(+), 42 deletions(-) diff --git a/adguardhome/k8s/adguard.yaml b/adguardhome/k8s/adguard.yaml index 6cf2b77..f6aa50a 100644 --- a/adguardhome/k8s/adguard.yaml +++ b/adguardhome/k8s/adguard.yaml @@ -73,7 +73,7 @@ spec: memory: "1.5Gi" cpu: "300m" requests: - memory: "500Mi" + memory: "1Gi" cpu: "50m" ports: - containerPort: 3000 diff --git a/authentik/k8s/authentik.yaml b/authentik/k8s/authentik.yaml index c2e795a..4485da4 100644 --- a/authentik/k8s/authentik.yaml +++ b/authentik/k8s/authentik.yaml @@ -52,7 +52,7 @@ spec: - containerPort: 9000 resources: requests: - memory: "700Mi" + memory: "768Mi" cpu: "300m" limits: memory: "1.5Gi" @@ -86,8 +86,8 @@ spec: name: authentik-secrets resources: requests: - memory: "512Mi" + memory: "320Mi" cpu: "300m" limits: - memory: "1Gi" + memory: "768Mi" cpu: "700m" diff --git a/cfddns/k8s/deployment.yaml b/cfddns/k8s/deployment.yaml index 5e85124..705ab3a 100644 --- a/cfddns/k8s/deployment.yaml +++ b/cfddns/k8s/deployment.yaml @@ -22,10 +22,10 @@ spec: imagePullPolicy: Always resources: requests: - memory: "20Mi" + memory: "32Mi" cpu: "30m" limits: - memory: "64Mi" + memory: "128Mi" cpu: "50m" envFrom: - secretRef: diff --git a/cloudflared/k8s/deployment.yaml b/cloudflared/k8s/deployment.yaml index cff6039..0b93a36 100644 --- a/cloudflared/k8s/deployment.yaml +++ b/cloudflared/k8s/deployment.yaml @@ -30,8 +30,8 @@ spec: key: TUNNEL_TOKEN resources: requests: - memory: "32Mi" + memory: "128Mi" cpu: "30m" limits: - memory: "128Mi" + memory: "256Mi" cpu: "200m" diff --git a/edu_master/k8s/redis.yaml b/edu_master/k8s/redis.yaml index b42647f..e0fca76 100644 --- a/edu_master/k8s/redis.yaml +++ b/edu_master/k8s/redis.yaml @@ -28,10 +28,10 @@ spec: resources: requests: cpu: 25m - memory: 64Mi + memory: 32Mi limits: cpu: 250m - memory: 256Mi + memory: 128Mi readinessProbe: exec: command: ["redis-cli", "ping"] diff --git a/edu_master/k8s/session-keeper.yaml b/edu_master/k8s/session-keeper.yaml index 73df466..e5b5fdd 100644 --- a/edu_master/k8s/session-keeper.yaml +++ b/edu_master/k8s/session-keeper.yaml @@ -38,10 +38,10 @@ spec: resources: requests: cpu: 25m - memory: 96Mi + memory: 32Mi limits: cpu: 250m - memory: 256Mi + memory: 128Mi readinessProbe: exec: command: ["/bin/sh", "-ec", "redis-cli -h redis EXISTS EDU_PHPSESSID | grep -q 1"] diff --git a/edu_master/k8s/webinar-checker.yaml b/edu_master/k8s/webinar-checker.yaml index c028aad..51daf7f 100644 --- a/edu_master/k8s/webinar-checker.yaml +++ b/edu_master/k8s/webinar-checker.yaml @@ -67,7 +67,7 @@ spec: resources: requests: cpu: "50m" - memory: "128Mi" + memory: "192Mi" limits: cpu: "600m" - memory: "512Mi" + memory: "384Mi" diff --git a/gitea/k8s/gitea.yaml b/gitea/k8s/gitea.yaml index 3ce7aa7..815db4c 100644 --- a/gitea/k8s/gitea.yaml +++ b/gitea/k8s/gitea.yaml @@ -47,10 +47,10 @@ spec: mountPath: /data resources: requests: - memory: "512Mi" + memory: "320Mi" cpu: "300m" limits: - memory: "1.5Gi" + memory: "1Gi" cpu: "1300m" volumes: - name: gitea-data diff --git a/glance/k8s/glance.yaml b/glance/k8s/glance.yaml index 0578e9e..20a5330 100644 --- a/glance/k8s/glance.yaml +++ b/glance/k8s/glance.yaml @@ -57,10 +57,10 @@ spec: resources: requests: cpu: "50m" - memory: "64Mi" + memory: "32Mi" limits: cpu: "200m" - memory: "256Mi" + memory: "128Mi" volumes: - name: glance-config configMap: diff --git a/homepages/k8s/homepages.yaml b/homepages/k8s/homepages.yaml index 96314b3..92e009a 100644 --- a/homepages/k8s/homepages.yaml +++ b/homepages/k8s/homepages.yaml @@ -39,10 +39,10 @@ spec: failureThreshold: 3 resources: requests: - memory: "10Mi" + memory: "32Mi" cpu: "20m" limits: - memory: "100Mi" + memory: "128Mi" cpu: "50m" --- apiVersion: v1 @@ -86,8 +86,8 @@ spec: failureThreshold: 3 resources: requests: - memory: "10Mi" + memory: "32Mi" cpu: "20m" limits: - memory: "100Mi" + memory: "128Mi" cpu: "50m" diff --git a/loki/k8s/loki-values.yaml b/loki/k8s/loki-values.yaml index 7b5a058..5ec7f1e 100644 --- a/loki/k8s/loki-values.yaml +++ b/loki/k8s/loki-values.yaml @@ -57,10 +57,10 @@ singleBinary: storageClass: local-path-retain resources: requests: - memory: "512Mi" + memory: "256Mi" cpu: "200m" limits: - memory: "2Gi" + memory: "1Gi" cpu: "1000m" # Zeroed: unused in SingleBinary mode (chart validation requires it). @@ -75,10 +75,10 @@ gateway: replicas: 1 resources: requests: - memory: "64Mi" + memory: "32Mi" cpu: "50m" limits: - memory: "256Mi" + memory: "128Mi" cpu: "300m" monitoring: diff --git a/netbird/k8s/netbird.yaml b/netbird/k8s/netbird.yaml index cd76e91..f72f6cb 100644 --- a/netbird/k8s/netbird.yaml +++ b/netbird/k8s/netbird.yaml @@ -163,10 +163,10 @@ spec: failureThreshold: 5 resources: requests: - memory: "64Mi" + memory: "32Mi" cpu: "50m" limits: - memory: "256Mi" + memory: "128Mi" cpu: "300m" --- apiVersion: v1 diff --git a/netbox/k8s/netbox.yaml b/netbox/k8s/netbox.yaml index b8f3550..6de0284 100644 --- a/netbox/k8s/netbox.yaml +++ b/netbox/k8s/netbox.yaml @@ -97,7 +97,7 @@ spec: resources: requests: cpu: "100m" - memory: "512Mi" + memory: "1Gi" limits: cpu: "2" memory: "2Gi" @@ -164,7 +164,7 @@ spec: memory: "256Mi" limits: cpu: "1" - memory: "1Gi" + memory: "512Mi" volumes: - name: netbox-config configMap: diff --git a/netronome/k8s/netronome.yaml b/netronome/k8s/netronome.yaml index d6d44ad..f455891 100644 --- a/netronome/k8s/netronome.yaml +++ b/netronome/k8s/netronome.yaml @@ -51,8 +51,8 @@ spec: key: NETRONOME__DB_PASSWORD resources: requests: - memory: "100Mi" + memory: "64Mi" cpu: "100m" limits: - memory: "512Mi" + memory: "256Mi" cpu: "500m" diff --git a/prometheus-stack/k8s/grafana-values.yaml b/prometheus-stack/k8s/grafana-values.yaml index b7a00e8..0618ab0 100644 --- a/prometheus-stack/k8s/grafana-values.yaml +++ b/prometheus-stack/k8s/grafana-values.yaml @@ -67,10 +67,10 @@ prometheus: storage: 40Gi resources: requests: - memory: "700Mi" + memory: "768Mi" cpu: 200m limits: - memory: "2Gi" + memory: "2560Mi" alertmanager: alertmanagerSpec: configSecret: alertmanager-config diff --git a/rackpeek/k8s/rackpeek.yaml b/rackpeek/k8s/rackpeek.yaml index 42fa1fb..c8bcb24 100644 --- a/rackpeek/k8s/rackpeek.yaml +++ b/rackpeek/k8s/rackpeek.yaml @@ -66,10 +66,10 @@ spec: periodSeconds: 30 resources: requests: - memory: "128Mi" + memory: "160Mi" cpu: "100m" limits: - memory: "512Mi" + memory: "384Mi" cpu: "500m" volumes: - name: rackpeek-config diff --git a/termix/k8s/termix.yaml b/termix/k8s/termix.yaml index 1fecdbc..0f6ea1e 100644 --- a/termix/k8s/termix.yaml +++ b/termix/k8s/termix.yaml @@ -38,10 +38,10 @@ spec: mountPath: /app/data resources: requests: - memory: "128Mi" + memory: "160Mi" cpu: "100m" limits: - memory: "256Mi" + memory: "384Mi" cpu: "300m" livenessProbe: httpGet: diff --git a/vaultwarden/k8s/vaultwarden.yaml b/vaultwarden/k8s/vaultwarden.yaml index f60dc8c..775a729 100644 --- a/vaultwarden/k8s/vaultwarden.yaml +++ b/vaultwarden/k8s/vaultwarden.yaml @@ -36,10 +36,10 @@ spec: resources: requests: cpu: "100m" - memory: "128Mi" + memory: "80Mi" limits: cpu: "500m" - memory: "512Mi" + memory: "256Mi" volumeMounts: - name: vaultwarden-data mountPath: /data diff --git a/vpn/xui/k8s/xui.yaml b/vpn/xui/k8s/xui.yaml index 07541ce..93f597b 100644 --- a/vpn/xui/k8s/xui.yaml +++ b/vpn/xui/k8s/xui.yaml @@ -51,10 +51,10 @@ spec: mountPath: /etc/x-ui resources: requests: - memory: "128Mi" + memory: "192Mi" cpu: "100m" limits: - memory: "1Gi" + memory: "512Mi" cpu: "1000m" volumes: - name: x-ui-db