diff --git a/edu_master/k8s/playwright.yaml b/edu_master/k8s/playwright.yaml index 6d0a4e3..ebf71d8 100644 --- a/edu_master/k8s/playwright.yaml +++ b/edu_master/k8s/playwright.yaml @@ -20,6 +20,15 @@ spec: # renovate: datasource=docker depName=mcr.microsoft.com/playwright versioning=docker image: mcr.microsoft.com/playwright:v1.56.0-jammy imagePullPolicy: IfNotPresent + # p95 412M, max 478M over 7 days, no limit before. Request is set at p95 + # so the pod is not an eviction candidate; the limit stays above 2x the + # request because browser page lifetimes are unpredictable. + resources: + requests: + cpu: "200m" + memory: "416Mi" + limits: + memory: "1Gi" command: - npx - -y diff --git a/errorpages/k8s/error-pages.yaml b/errorpages/k8s/error-pages.yaml index 0f7b993..4d1845d 100644 --- a/errorpages/k8s/error-pages.yaml +++ b/errorpages/k8s/error-pages.yaml @@ -28,6 +28,13 @@ spec: containers: - name: error-pages image: gcr.forust.xyz/forust/error-pages:prod + # p95 6M, max 10M, no limit before. + resources: + requests: + cpu: "10m" + memory: "32Mi" + limits: + memory: "128Mi" ports: - containerPort: 80 readinessProbe: diff --git a/loki/k8s/alloy-values.yaml b/loki/k8s/alloy-values.yaml index 2f88adc..7e8307b 100644 --- a/loki/k8s/alloy-values.yaml +++ b/loki/k8s/alloy-values.yaml @@ -7,18 +7,32 @@ controller: type: daemonset + +# config-reloader sidecar: p95 33M, max 43M. The chart keeps it at the top level, +# not under `alloy:`. +configReloader: resources: requests: - memory: "128Mi" - cpu: "50m" + memory: "32Mi" + cpu: "10m" limits: - memory: "512Mi" - cpu: "500m" + memory: "128Mi" image: tag: "v1.19.2" alloy: + # p95 275M, max 287M. Alloy tails every pod log and ships it to Loki, so it sits + # on the same IronWolf read path the node is I/O bound on. Request is set at p95. + # The chart key is `alloy.resources`. `controller.resources` is ignored silently, + # which is why this pod shipped with no limits at all. + resources: + requests: + memory: "288Mi" + cpu: "50m" + limits: + memory: "512Mi" + configMap: create: true content: | diff --git a/loki/k8s/loki-values.yaml b/loki/k8s/loki-values.yaml index d22c4f3..7b5a058 100644 --- a/loki/k8s/loki-values.yaml +++ b/loki/k8s/loki-values.yaml @@ -39,6 +39,16 @@ loki: local: directory: /var/loki/rules +# p95 84M, max 85M for the rules sidecar that shares the singleBinary pod. +# The chart exposes it as `sidecar.resources`, shared with any other sidecar. +sidecar: + resources: + requests: + memory: "96Mi" + cpu: "10m" + limits: + memory: "192Mi" + singleBinary: replicas: 1 persistence: diff --git a/prometheus-stack/k8s/grafana-values.yaml b/prometheus-stack/k8s/grafana-values.yaml index 8feddb9..b7a00e8 100644 --- a/prometheus-stack/k8s/grafana-values.yaml +++ b/prometheus-stack/k8s/grafana-values.yaml @@ -26,6 +26,25 @@ grafana: service: port: 80 + # p95 391M, observed max 1046M with no limit at all. Request is set at p95 so the + # scheduler sees reality; the limit is a manual exception above the 1.3x max + # formula, because a single query burst reached 1046M. + resources: + requests: + memory: "416Mi" + cpu: 100m + limits: + memory: "1536Mi" + + # One block covers both the dashboards and datasources sidecars (p95 91M / 80M). + sidecar: + resources: + requests: + memory: "96Mi" + cpu: 10m + limits: + memory: "192Mi" + additionalDataSources: - name: Loki type: loki @@ -55,6 +74,14 @@ prometheus: alertmanager: alertmanagerSpec: configSecret: alertmanager-config + # p95 66M, max 68M. Silences and notification state live here, so the request + # stays above p95 to keep the pod out of the eviction candidates. + resources: + requests: + memory: "96Mi" + cpu: 10m + limits: + memory: "192Mi" storage: volumeClaimTemplate: spec: @@ -66,6 +93,39 @@ alertmanager: requests: storage: 20Gi +# p95 103M, max 107M, and it grows with the number of cluster objects. +kube-state-metrics: + # Values key is the dependency name from Chart.yaml, not the `kubeStateMetrics` + # condition key. Setting resources under `kubeStateMetrics:` is silently ignored. + resources: + requests: + memory: "128Mi" + cpu: 50m + limits: + memory: "256Mi" + +# p95 39M, max 40M. One per node, so it scales with node count. +prometheus-node-exporter: + resources: + requests: + memory: "32Mi" + cpu: 20m + limits: + memory: "128Mi" + +# p95 75M, max 75M. Creates and reconciles every PrometheusRule in the cluster. +prometheusOperator: + resources: + requests: + memory: "96Mi" + cpu: 50m + limits: + memory: "192Mi" + +# config-reloader sidecars (p95 33M, max 43M) are not covered: the chart does not +# template `prometheusSpec.configReloader.resources`, so there is no values key +# for them. They keep shipping with requests only. + defaultRules: disabled: CPUThrottlingHigh: true diff --git a/reloader/k8s/reloader-values.yaml b/reloader/k8s/reloader-values.yaml index 656e939..f26caea 100644 --- a/reloader/k8s/reloader-values.yaml +++ b/reloader/k8s/reloader-values.yaml @@ -11,10 +11,13 @@ reloader: replicas: 1 # The chart defaults to no requests or limits, so the pod is evictable under node # pressure and the restarts go with it. + # Memory was raised from 64Mi: measured p95 over 7 days is 73M, so the pod was + # running above its own request and sitting in the eviction candidates. This pod + # is the one that restarts every other pod, so it must not be evicted. resources: requests: cpu: "10m" - memory: "64Mi" + memory: "96Mi" limits: cpu: "100m" - memory: "128Mi" + memory: "192Mi" diff --git a/searxng/k8s/valkey.yaml b/searxng/k8s/valkey.yaml index 3ec5b9d..08f9933 100644 --- a/searxng/k8s/valkey.yaml +++ b/searxng/k8s/valkey.yaml @@ -30,6 +30,13 @@ spec: containers: - name: valkey image: docker.io/valkey/valkey:9.1.2-alpine + # p95 15M, max 17M, no limit before. Matches the netbox valkey pod. + resources: + requests: + cpu: "50m" + memory: "64Mi" + limits: + memory: "256Mi" command: - valkey-server - --save