feat(monitoring): add VictoriaMetrics trial stack
ci / lint-prettier (push) Skipped
ci / lint-ruff (push) Skipped
ci / lint-yaml (push) Skipped
ci / lint-dockerfiles (push) Skipped
ci / validate (push) Skipped
renovate-ci / validate-renovate (push) Skipped
renovate-ci / validate-renovate (pull_request) Skipped
ci / lint-compose (pull_request) Successful in 12s
ci / lint-actionlint (pull_request) Successful in 6s
ci / lint-shellcheck (pull_request) Successful in 13s
ci / lint-prettier (pull_request) Successful in 21s
ci / lint-ruff (pull_request) Successful in 7s
ci / lint-yaml (pull_request) Successful in 12s
ci / lint-dockerfiles (pull_request) Successful in 7s
ci / validate (pull_request) Successful in 7s
ci / build (pull_request) Skipped

This commit is contained in:
forust committed 2026-10-06 17:45:44 +02:00
1 parent 4510394531
commit 18c633c242
18 files changed
+3482

No files matched your search

+8
View File
@@ -38,6 +38,9 @@ grafana:
# One block covers both the dashboards and datasources sidecars (p95 91M / 80M).
sidecar:
datasources:
# Chart built-in Prometheus DS disabled: VictoriaMetrics (below) is the default.
defaultDatasourceEnabled: false
resources:
requests:
memory: "96Mi"
@@ -50,6 +53,11 @@ grafana:
type: loki
url: http://loki-gateway.prometheus.svc.cluster.local
access: proxy
- name: VictoriaMetrics
type: prometheus
url: http://victoria-metrics.prometheus.svc.cluster.local:8428
access: proxy
isDefault: true
prometheus:
prometheusSpec:
+102
View File
@@ -31,3 +31,105 @@ spec:
port: 80
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: prometheus-local
namespace: prometheus
spec:
entryPoints:
- websecure
routes:
- match: Host(`prom.workstation.internal`) || Host(`prom.gigaforust.internal`)
kind: Rule
services:
- name: prometheus-stack-kube-prom-prometheus
port: 9090
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: alertmanager-local
namespace: prometheus
spec:
entryPoints:
- websecure
routes:
- match: Host(`am.workstation.internal`) || Host(`am.gigaforust.internal`)
kind: Rule
services:
- name: prometheus-stack-kube-prom-alertmanager
port: 9093
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: loki-local
namespace: prometheus
spec:
entryPoints:
- websecure
routes:
- match: Host(`loki.workstation.internal`) || Host(`loki.gigaforust.internal`)
kind: Rule
services:
- name: loki-gateway
port: 80
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: alloy-local
namespace: prometheus
spec:
entryPoints:
- websecure
routes:
- match: Host(`alloy.workstation.internal`) || Host(`alloy.gigaforust.internal`)
kind: Rule
services:
- name: alloy
port: 12345
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: victoria-local
namespace: prometheus
spec:
entryPoints:
- websecure
routes:
- match: Host(`victoria.workstation.internal`) || Host(`victoria.gigaforust.internal`)
kind: Rule
services:
- name: victoria-metrics
port: 8428
tls:
secretName: internal-wildcard-tls
---
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: vmalert-local
namespace: prometheus
spec:
entryPoints:
- websecure
routes:
- match: Host(`vmalert.workstation.internal`) || Host(`vmalert.gigaforust.internal`)
kind: Rule
services:
- name: vmalert
port: 8880
tls:
secretName: internal-wildcard-tls
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,10 @@
# Create the victoria-secrets Secret before deploying VictoriaMetrics.
# Copy the metrics password from uptime-kuma/k8s/secrets.yaml into the value below.
apiVersion: v1
kind: Secret
metadata:
name: victoria-secrets
namespace: prometheus
type: Opaque
stringData:
uptime-kuma-password: "REPLACE_ME"
+122
View File
@@ -0,0 +1,122 @@
apiVersion: v1
kind: ServiceAccount
metadata:
name: victoria
namespace: prometheus
---
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: victoria
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: prometheus-stack-kube-prom-prometheus
subjects:
- kind: ServiceAccount
name: victoria
namespace: prometheus
---
apiVersion: v1
kind: Service
metadata:
name: victoria-metrics
namespace: prometheus
spec:
selector:
app: victoria-metrics
ports:
- port: 8428
targetPort: 8428
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: victoria-pvc
namespace: prometheus
spec:
resources:
requests:
storage: 10Gi
volumeMode: Filesystem
accessModes:
- ReadWriteOnce
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: victoria-deployment
namespace: prometheus
spec:
replicas: 1
selector:
matchLabels:
app: victoria-metrics
strategy:
type: Recreate
template:
metadata:
labels:
app: victoria-metrics
spec:
serviceAccountName: victoria
containers:
- name: victoria
image: victoriametrics/victoria-metrics:v1.153.0-scratch
args:
- -storageDataPath=/vmdata
- -retentionPeriod=30d
- -httpListenAddr=:8428
- -promscrape.config=/etc/vm/conf/scrape.yaml
- -promscrape.configCheckInterval=60s
ports:
- containerPort: 8428
readinessProbe:
httpGet:
path: /health
port: 8428
initialDelaySeconds: 15
periodSeconds: 10
failureThreshold: 6
livenessProbe:
httpGet:
path: /health
port: 8428
initialDelaySeconds: 60
periodSeconds: 30
failureThreshold: 3
volumeMounts:
- name: vmdata
mountPath: /vmdata
- name: scrape-config
mountPath: /etc/vm/conf
readOnly: true
- name: vm-secrets
mountPath: /etc/vm/secrets
readOnly: true
- name: prom-admission-ca
mountPath: /etc/prometheus/certs
readOnly: true
resources:
requests:
cpu: "100m"
memory: "256Mi"
limits:
cpu: "1000m"
memory: "1Gi"
volumes:
- name: vmdata
persistentVolumeClaim:
claimName: victoria-pvc
- name: scrape-config
configMap:
name: victoria-scrape
- name: vm-secrets
secret:
secretName: victoria-secrets
- name: prom-admission-ca
secret:
secretName: prometheus-stack-kube-prom-admission
items:
- key: ca
path: 0_prometheus_prometheus-stack-kube-prom-admission_ca
+70
View File
@@ -0,0 +1,70 @@
apiVersion: v1
kind: Service
metadata:
name: vmalert
namespace: prometheus
spec:
selector:
app: vmalert
ports:
- port: 8880
targetPort: 8880
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: vmalert-deployment
namespace: prometheus
spec:
replicas: 1
selector:
matchLabels:
app: vmalert
strategy:
type: Recreate
template:
metadata:
labels:
app: vmalert
spec:
containers:
- name: vmalert
image: victoriametrics/vmalert:v1.153.0
args:
- -datasource.url=http://victoria-metrics.prometheus.svc.cluster.local:8428
- -remoteWrite.url=http://victoria-metrics.prometheus.svc.cluster.local:8428
- -notifier.url=http://prometheus-stack-kube-prom-alertmanager.prometheus.svc.cluster.local:9093
- -rule=/etc/vm/rules/*.yaml
- -evaluationInterval=60s
- -httpListenAddr=:8880
ports:
- containerPort: 8880
readinessProbe:
httpGet:
path: /metrics
port: 8880
initialDelaySeconds: 15
periodSeconds: 10
failureThreshold: 6
livenessProbe:
httpGet:
path: /metrics
port: 8880
initialDelaySeconds: 60
periodSeconds: 30
failureThreshold: 3
volumeMounts:
- name: rules
mountPath: /etc/vm/rules
readOnly: true
resources:
requests:
cpu: "50m"
memory: "64Mi"
limits:
cpu: "200m"
memory: "256Mi"
volumes:
- name: rules
configMap:
name: prometheus-prometheus-stack-kube-prom-prometheus-rulefiles-0
+31
View File
@@ -0,0 +1,31 @@
# One-shot history backfill Prometheus -> VictoriaMetrics via vmctl remote-read.
# Safe to keep applied: a completed Job is a no-op on re-apply.
apiVersion: batch/v1
kind: Job
metadata:
name: vmctl-backfill
namespace: prometheus
spec:
backoffLimit: 2
ttlSecondsAfterFinished: 3600
template:
spec:
restartPolicy: OnFailure
containers:
- name: vmctl
image: victoriametrics/vmctl:v1.153.0
args:
- remote-read
- -s
- --disable-progress-bar
- --remote-read-src-addr=http://prometheus-stack-kube-prom-prometheus.prometheus.svc.cluster.local:9090
- --remote-read-filter-time-start=2026-09-06T10:26:14Z
- --remote-read-step-interval=day
- --vm-addr=http://victoria-metrics.prometheus.svc.cluster.local:8428
resources:
requests:
cpu: "200m"
memory: "512Mi"
limits:
cpu: "1000m"
memory: "1Gi"
+543
View File
@@ -0,0 +1,543 @@
apiVersion: apps/v1
kind: Deployment
metadata:
# name: unique within the namespace
name: nginx-deployment
# namespace: logical isolation
namespace: example
labels:
# free-form key/value tags used for selection and grouping.
app: nginx
app.kubernetes.io/name: nginx
app.kubernetes.io/instance: nginx-example
app.kubernetes.io/version: "1.27"
app.kubernetes.io/component: server
app.kubernetes.io/part-of: example
app.kubernetes.io/managed-by: kubectl
annotations:
# Key/value, but never used for selection
# Only for metadata (descriptions, owners, timestamps).
description: "deployment example"
owner: team-platform
# finalizers: identifiers that block deletion until some controller removes
# them after cleanup. RARE on workloads.
# finalizers:
# - example.com/cleanup
spec:
# replicas: how many pod copies to keep running. Default 1.
replicas: 3
# revisionHistoryLimit: how many old ReplicaSets are kept so you can roll
# back. Default 10.
revisionHistoryLimit: 5
# progressDeadlineSeconds: if a rollout makes no progress for this long it
# is marked ProgressDeadlineExceeded. Default 600.
progressDeadlineSeconds: 600
# minReadySeconds: a new pod must stay Ready for this long before it counts
# as available. Protects against pods that flap right after start.
minReadySeconds: 10
# paused: freezes the rollout controller. RARE - used to accumulate several
# changes and release them as a single rollout.
paused: false
# selector: defines which pods belong to this Deployment. Must match the
# pod template labels exactly. Immutable after creation.
selector:
matchLabels:
app: nginx
# matchExpressions: set-based selection (In, NotIn, Exists, DoesNotExist).
# RARE on Deployments.
# matchExpressions:
# - key: tier
# operator: In
# values: [frontend]
strategy:
# type: RollingUpdate (replace gradually, default) or Recreate (kill all
# old pods first). Recreate fits state that cannot have two writers at
# once (SQLite file, exclusive lock).
type: RollingUpdate
rollingUpdate:
# maxSurge: how many pods above `replicas` may exist mid-rollout.
# Number or percentage.
maxSurge: 1
# maxUnavailable: how many pods may be simultaneously down mid-rollout.
# Number or percentage.
maxUnavailable: 1
template:
metadata:
labels:
app: nginx
app.kubernetes.io/name: nginx
annotations:
description: "nginx pod"
spec:
# serviceAccountName: the identity pods use against the API server.
serviceAccountName: default
# automountServiceAccountToken: mount the API token into pods. Set false
# for pods that never call the API to shrink the escape blast radius.
automountServiceAccountToken: true
# schedulerName: which scheduler places the pod. The default scheduler
# handles virtually everything.
schedulerName: default-scheduler
# nodeName: pin the pod to one node, bypassing the scheduler. RARE and
# brittle - nodeSelector/affinity express intent better.
# nodeName: node-1
# nodeSelector: hard requirement on node labels.
nodeSelector:
kubernetes.io/os: linux
# hostname/subdomain: give the pod a stable hostname and DNS entry
# <hostname>.<subdomain>.<namespace>.svc.cluster.local. Mostly a
# StatefulSet concern (which gets this automatically).
hostname: nginx
subdomain: example-subdomain
# setHostnameAsFQDN: use the FQDN above as the hostname. Default false.
setHostnameAsFQDN: false
# priorityClassName: scheduling priority; higher values preempt lower.
# priorityClassName: high-priority
# preemptionPolicy: Never stops this pod from preempting others.
# Default PreemptLowerPriority.
preemptionPolicy: PreemptLowerPriority
# runtimeClassName: alternate container runtime (gVisor, Kata).
# Omit for the default runtime.
# runtimeClassName: gvisor
# enableServiceLinks: inject <SVC>_SERVICE_HOST style env vars. True by
# default; false keeps the environment clean when you use DNS only.
enableServiceLinks: true
# hostAliases: extra /etc/hosts lines. RARE - usually means DNS should
# have been fixed instead.
hostAliases:
- ip: 192.168.1.10
hostnames:
- legacy-db.example.com
# hostNetwork/hostPID/hostIPC: share the node's network/process/IPC
# namespaces. Needed for node-level agents; dangerous for apps.
hostNetwork: false
hostPID: false
hostIPC: false
# shareProcessNamespace: all containers in the pod see each other's
# processes. Handy for sidecar debuggers; off by default.
shareProcessNamespace: false
# dnsPolicy: ClusterFirst (default, .svc names resolve), Default
# (inherit the node's resolver), ClusterFirstWithHostNet (for
# hostNetwork pods), None (dnsConfig takes over completely).
dnsPolicy: ClusterFirst
dnsConfig:
nameservers:
- 1.1.1.1
searches:
- example.com
options:
- name: ndots
value: "2"
# readinessGates: custom conditions (reported by external controllers)
# that must be true before the pod counts as Ready. RARE.
# readinessGates:
# - conditionType: example.com/lb-attached
# topologySpreadConstraints: spread pods across zones/hosts with skew
# control. The modern, expressive successor of bare podAntiAffinity.
topologySpreadConstraints:
- maxSkew: 1
topologyKey: kubernetes.io/hostname
whenUnsatisfiable: ScheduleAnyway
labelSelector:
matchLabels:
app: nginx
affinity:
nodeAffinity:
# requiredDuringSchedulingIgnoredDuringExecution: hard node rule -
# pods that violate it are never scheduled there.
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/arch
operator: In
values: [amd64]
# preferredDuringSchedulingIgnoredDuringExecution: soft node rule -
# the scheduler tries, but schedules anyway if impossible.
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 10
preference:
matchExpressions:
- key: node-role.kubernetes.io/worker
operator: Exists
podAffinity:
# Attract to nodes already running matching pods (data locality).
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 50
podAffinityTerm:
labelSelector:
matchLabels:
app: cache
topologyKey: kubernetes.io/hostname
podAntiAffinity:
# Repel from nodes running matching pods (spread replicas).
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
podAffinityTerm:
labelSelector:
matchLabels:
app: nginx
topologyKey: kubernetes.io/hostname
tolerations:
# Tolerate a node taint to allow scheduling there. Without a matching
# toleration, a tainted node rejects the pod.
- key: dedicated
operator: Equal
value: "true"
effect: NoSchedule
# tolerationSeconds: with effect NoExecute, tolerate for this long
# before eviction. Omit for infinite tolerance.
tolerationSeconds: 3600
# imagePullSecrets: credentials for private registries.
# imagePullSecrets:
# - name: registry-credentials
# securityContext (pod-level): defaults inherited by every container
# unless a container overrides them.
securityContext:
runAsUser: 101
runAsGroup: 101
# runAsNonRoot: refuse to start as uid 0.
runAsNonRoot: true
# fsGroup: group that owns mounted volumes; kubelet chowns on mount.
fsGroup: 101
# fsGroupChangePolicy: Always (chown on every mount, slow on large
# volumes) or OnRootMismatch (chown only when needed).
fsGroupChangePolicy: OnRootMismatch
# supplementalGroups: extra groups granted for volume access.
supplementalGroups: [102]
# sysctls: namespaced kernel parameters. Unsafe ones need explicit
# kubelet opt-in.
sysctls:
- name: net.core.somaxconn
value: "1024"
# seLinuxOptions: SELinux user/role/type/level. RARE outside MLS.
seLinuxOptions:
level: s0:c123,c456
# seccompProfile: syscall sandbox. RuntimeDefault is the sane
# baseline; Localhost loads a custom profile from the node.
seccompProfile:
type: RuntimeDefault
# windowsOptions: GMSA / runAsUserName for Windows nodes.
initContainers:
# Run strictly in order, each to completion, before app containers
# start. Used for migrations, permission fixes, dependency waits.
- name: init-permissions
image: busybox:1.36
command: ["sh", "-c", "chown -R 101:101 /data"]
volumeMounts:
- name: html
mountPath: /data
resources:
requests:
cpu: "10m"
memory: "16Mi"
limits:
cpu: "50m"
memory: "64Mi"
# restartPolicy on a container (not the pod): Always turns it into
# a native sidecar that keeps running next to the app (1.28+).
# restartPolicy: Always
containers:
- name: nginx
# image: repository plus tag. Pin tags - `latest` moves under you.
image: nginx:1.27.3
# imagePullPolicy: Always (re-pull even pinned tags), IfNotPresent
# (use cache, works offline), Never (cache only, fails otherwise).
imagePullPolicy: IfNotPresent
# command: overrides the image ENTRYPOINT. args: overrides CMD.
# command: ["nginx"]
# args: ["-g", "daemon off;"]
# workingDir: overrides the image WORKDIR.
workingDir: /usr/share/nginx/html
# stdin/stdinOnce/tty: interactive input. For debug shells and
# one-shot runs, never for servers.
stdin: false
stdinOnce: false
tty: false
ports:
- name: http
containerPort: 80
protocol: TCP
# hostPort: expose straight on the node, bypassing Services.
# RARE - allows only one such pod per node per port.
# hostPort: 8080
# hostIP: which node address hostPort binds to.
# hostIP: 127.0.0.1
env:
- name: NGINX_PORT
value: "80"
- name: POD_NAME
valueFrom:
fieldRef:
# fieldPath exposes pod metadata: metadata.name,
# metadata.namespace, metadata.uid, spec.nodeName,
# spec.serviceAccountName, status.podIP(s), etc.
fieldPath: metadata.name
- name: NODE_NAME
valueFrom:
fieldRef:
fieldPath: spec.nodeName
- name: POD_MEMORY_LIMIT
valueFrom:
# resourceFieldRef exposes this container's own
# requests/limits. divisor formats the value.
resourceFieldRef:
resource: limits.memory
divisor: 1Mi
- name: API_PASSWORD
valueFrom:
secretKeyRef:
name: example-secrets
key: api-password
# optional: tolerate a missing key (variable stays unset).
optional: false
- name: LOG_LEVEL
valueFrom:
configMapKeyRef:
name: example-config
key: log-level
optional: false
envFrom:
# Bulk-inject every key of a ConfigMap/Secret as env vars.
- configMapRef:
name: example-config
optional: false
# prefix: prepended to every injected key, avoids collisions.
prefix: APP_
- secretRef:
name: example-secrets
optional: false
resources:
# requests: guaranteed reservation used for scheduling. Set at
# measured idle/p95 - over-requesting starves neighboring pods.
requests:
cpu: "100m"
memory: "128Mi"
# limits: hard ceiling. Breaching memory kills the container
# (OOMKilled); breaching CPU only throttles it (slow, not dead).
limits:
cpu: "500m"
memory: "512Mi"
# claims: reference a ResourceClaim for dynamic resources
# (GPUs via DRA, 1.26+). RARE.
# claims:
# - name: gpu
# resizePolicy: what happens on in-place container resize (1.27+).
# NotRequired keeps running; RestartContainer restarts to apply.
resizePolicy:
- resourceName: cpu
restartPolicy: NotRequired
- resourceName: memory
restartPolicy: NotRequired
volumeMounts:
- name: html
mountPath: /usr/share/nginx/html
readOnly: false
# subPath: mount a single file/dir of the volume instead of
# its root. Typical for single-file config mounts.
# subPath: index.html
# subPathExpr: subPath assembled from env variables.
# subPathExpr: $(POD_NAME)/data
# mountPropagation: share mounts back with the host
# (HostToContainer, Bidirectional). Storage-driver territory.
mountPropagation: None
- name: tmp
mountPath: /tmp
# volumeDevices: raw block devices without a filesystem. RARE -
# databases on local PVs with volumeMode: Block.
# volumeDevices:
# - name: blockvol
# devicePath: /dev/xvda
livenessProbe:
# Exactly one handler per probe: httpGet, tcpSocket, exec, grpc.
httpGet:
path: /healthz
port: http
scheme: HTTP
# httpHeaders: extra headers sent with the probe request.
httpHeaders:
- name: Host
value: example.com
# initialDelaySeconds: wait after start before first probe.
initialDelaySeconds: 15
# periodSeconds: interval between probes.
periodSeconds: 20
# timeoutSeconds: when a single probe counts as failed.
timeoutSeconds: 5
# successThreshold: consecutive successes to count as healthy.
# Keep 1.
successThreshold: 1
# failureThreshold: consecutive failures to trigger the action.
failureThreshold: 3
readinessProbe:
# Failing readiness removes the pod from Services (no traffic)
# without restarting it. Failing liveness restarts it.
httpGet:
path: /readyz
port: 80
initialDelaySeconds: 5
periodSeconds: 10
timeoutSeconds: 3
successThreshold: 1
failureThreshold: 3
startupProbe:
# Disables liveness/readiness until it first succeeds. Total
# budget = failureThreshold * periodSeconds (here 30 * 10s).
# The cure for slow-starting apps that otherwise get
# restart-looped before they finish booting.
tcpSocket:
port: 80
# host: probe a different host than the pod IP. RARE.
failureThreshold: 30
periodSeconds: 10
timeoutSeconds: 5
lifecycle:
# postStart runs right after start; preStop runs before SIGTERM.
# Slow hooks stall the pod transition - keep them fast.
postStart:
exec:
command: ["sh", "-c", "echo started > /tmp/started"]
# Classic preStop: sleep so endpoints are removed everywhere
# before the process receives SIGTERM.
preStop:
exec:
command: ["sh", "-c", "sleep 5"]
# terminationMessagePath: file whose content becomes the container's
# final status message. FallbackToLogsOnError appends log tail when
# the file is empty.
terminationMessagePath: /dev/termination-log
terminationMessagePolicy: File
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: false
runAsNonRoot: true
runAsUser: 101
runAsGroup: 101
capabilities:
add:
- NET_BIND_SERVICE
drop:
- ALL
seccompProfile:
type: RuntimeDefault
# ephemeralContainers: troubleshooting shells injected into a RUNNING
# pod with kubectl debug. Declared ad-hoc in practice, never committed.
# ephemeralContainers:
# - name: debugger
# image: busybox:1.36
# command: ["sh"]
# stdin: true
# tty: true
# targetContainerName: nginx
volumes:
- name: html
persistentVolumeClaim:
claimName: example-pvc
# readOnly: mount the claim read-only in this pod.
readOnly: false
- name: tmp
emptyDir:
# medium "" (node disk) or Memory (tmpfs). sizeLimit evicts the
# pod when exceeded. Dies with the pod either way.
medium: ""
sizeLimit: 256Mi
- name: config-files
configMap:
# Each key becomes a file under the mount path.
name: example-config-files
defaultMode: 0644
optional: false
items:
- key: nginx.conf
path: nginx.conf
mode: 0644
- name: tls
secret:
# Secret volumes are tmpfs-backed, never touch node disk.
secretName: example-tls
defaultMode: 0644
optional: false
items:
- key: tls.crt
path: tls.crt
mode: 0644
- name: host-time
hostPath:
path: /etc/localtime
# type: DirectoryOrCreate, Directory, FileOrCreate, File,
# Socket, CharDevice, BlockDevice. Always set it: a missing
# path then fails loudly instead of silently creating the
# wrong filesystem object.
type: File
- name: podinfo
downwardAPI:
# Expose pod metadata as files.
items:
- path: labels
fieldRef:
fieldPath: metadata.labels
- path: cpu-request
resourceFieldRef:
resource: requests.cpu
containerName: nginx
divisor: 1m
- name: all-config
projected:
# Merge several sources into a single directory.
defaultMode: 0644
sources:
- configMap:
name: example-config
items:
- key: log-level
path: log-level
- secret:
name: example-secrets
items:
- key: api-password
path: password
- downwardAPI:
items:
- path: podname
fieldRef:
fieldPath: metadata.name
# Further volume types share the same `name:` + type shape:
# nfs: { server, path }, csi: (storage drivers), persistentVolumeClaim
# shown above, plus legacy in-tree plugins (fc, iscsi, rbd, glusterfs)
# and gitRepo (deprecated - use initContainers + emptyDir instead).
# restartPolicy: only Always is valid for Deployments. (Jobs use
# OnFailure/Never; bare pods accept all three.)
restartPolicy: Always
terminationGracePeriodSeconds: 30
# activeDeadlineSeconds: kill the pod after this long no matter what.
# A Job concern, not a server concern - shown only for completeness.
# activeDeadlineSeconds: 3600
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: example-pvc
namespace: example
spec:
# accessModes: ReadWriteOnce (one node), ReadOnlyMany, ReadWriteMany
# (needs a shared filesystem), ReadWriteOncePod (single pod, strictest).
accessModes:
- ReadWriteOnce
# storageClassName: selects the provisioner. "" (empty) disables dynamic
# provisioning and binds a pre-created volume instead.
storageClassName: standard
# volumeMode: Filesystem (default) or Block (used with volumeDevices).
volumeMode: Filesystem
resources:
requests:
storage: 5Gi
# limits: accepted on paper, almost no provisioner enforces them.
# selector: bind a specific pre-created PV by labels. RARE with dynamic
# provisioning.
# selector:
# matchLabels:
# disk: ssd
# dataSource: clone this PVC from another PVC or a snapshot at creation.
# dataSource:
# name: example-snapshot
# kind: VolumeSnapshot
# apiGroup: snapshot.storage.k8s.io
# dataSourceRef: cross-namespace-capable successor of dataSource.
+63
View File
@@ -0,0 +1,63 @@
# Exhaustive EndpointSlice reference (discovery.k8s.io/v1).
# EndpointSlices are the address lists behind a Service: each entry says
# "IP X serves port Y, ready or not". Normally the endpoint controller writes
# them automatically from the Service selector. You write one by hand in a
# single case: a selector-less Service pointing OUTSIDE the cluster (a host
# daemon, a LAN appliance, an external database). The pair looks like:
# Service (no selector, same port names) + this EndpointSlice.
apiVersion: discovery.k8s.io/v1
kind: EndpointSlice
metadata:
name: app-external-abc123
namespace: example
labels:
# kubernetes.io/service-name: THE binding label. It must equal the
# selector-less Service name - this is what attaches the slice to it.
# The endpoint controller owns slices it creates; hand-written ones
# just need this label to be picked up.
kubernetes.io/service-name: app-external-service
app: app
annotations:
description: "hand-written slice for an off-cluster backend"
# addressType: IPv4, IPv6, or FQDN. All endpoints in one slice share it -
# mix families with one slice per family.
addressType: IPv4
ports:
- name: http
# name: MUST match the Service port name it serves.
protocol: TCP
# port: the REAL backend port (may differ from the Service port - the
# Service port is the in-cluster alias, this is where packets go).
port: 8080
# appProtocol: payload hint, mirrors the Service field.
appProtocol: http
endpoints:
# One entry per backend address. kube-proxy load-balances across the ones
# whose conditions say ready+serving+terminating=false.
- addresses:
# addresses: one or more IPs (or a single DNS name for FQDN slices).
- 192.168.1.50
conditions:
# ready: backend accepts traffic. False removes it from rotation
# without deleting the entry (flap-friendly).
ready: true
# serving: the process is up. Differs from ready during shutdown:
# serving=false + terminating=true = draining.
serving: true
# terminating: the endpoint is going away. Draining traffic, not dead.
terminating: false
# hostname: DNS name published for this endpoint (headless Services).
# hostname: backend-1
# targetRef: link back to the pod/node object (set automatically on
# controller-managed slices; omit on hand-written ones).
# targetRef:
# kind: Pod
# namespace: example
# name: app-67890abcde-fghij
# uid: 12345678-1234-1234-1234-123456789abc
# nodeName: the node hosting this endpoint (topology-aware routing).
# zone: override the endpoint zone (defaults from nodeName).
# hints: topology hints for zone-aware routing (PreferClose).
# hints:
# forZones:
# - name: zone-a
+100
View File
@@ -0,0 +1,100 @@
# Exhaustive Gateway reference (gateway.networking.k8s.io/v1).
# A Gateway is the entry door: it owns listener ports/protocols/hostnames and
# delegates actual routing to Route objects (HTTPRoute, TCPRoute, ...), which
# attach via parentRefs. One Gateway usually fronts many Routes.
apiVersion: gateway.networking.k8s.io/v1
kind: Gateway
metadata:
name: example
namespace: example
labels:
app: example
annotations:
description: "exhaustive gateway example"
spec:
# gatewayClassName: which controller implements this Gateway
# (kubectl get gatewayclass). The controller only touches Gateways naming
# its own class; anything else stays Ignored.
gatewayClassName: example-class
# addresses: VIPs/hostnames to request for the Gateway. Most controllers
# (including cloud LBs) allocate and fill status.addresses automatically;
# setting this pins a static IP. Omit for auto-assignment.
# addresses:
# - type: IPAddress
# value: 203.0.113.10
# - type: Hostname
# value: lb.example.com
# infrastructure: controller-specific settings for the provisioned data
# plane (labels/annotations propagated to it). RARE - most setups never
# need it.
# infrastructure:
# labels:
# environment: prod
# annotations:
# example.com/keep: "true"
listeners:
# Each listener = one port + protocol + optional hostname + TLS + which
# Routes may attach. Listener names are referenced by Route parentRefs
# via sectionName.
- name: http
# port: 1-65535. Must be free on the data plane (controllers often
# require 80/443 to match their own entrypoints, otherwise the
# listener is marked Invalid/Conflicted).
port: 80
# protocol: HTTP, HTTPS, TLS, TCP, UDP.
protocol: HTTP
# hostname: restrict this listener to one DNS name. Omit to accept all
# (Routes then narrow via their own hostnames). Listener hostname and
# Route hostnames must intersect or the Route is rejected.
hostname: app.example.com
# allowedRoutes: which Routes may bind here.
allowedRoutes:
namespaces:
# from: Same (only this namespace), All (any namespace), or
# Selector (namespaces matching the selector below).
from: Same
# selector: used only with from: Selector.
# selector:
# matchLabels:
# shared-gateway-access: "true"
# kinds: restrict by Route kind. Default allows whatever the
# listener protocol supports (HTTPRoute on HTTP, etc.).
kinds:
- kind: HTTPRoute
- name: https
port: 443
protocol: HTTPS
hostname: app.example.com
# tls: termination settings for HTTPS/TLS listeners.
tls:
# mode: Terminate (decrypt here, default) or Passthrough (forward
# encrypted bytes to the backend - the backend holds the key).
mode: Terminate
# certificateRefs: TLS Secrets (or other kinds) in the SAME namespace
# (cross-namespace needs a ReferenceGrant). SNI picks among them.
certificateRefs:
- name: app-prod-tls
# kind/group default to Secret / core. Other kinds (e.g. a
# cert-manager Certificate via a plugin) set kind + group.
kind: Secret
group: ""
# options: controller-specific TLS knobs, referenced by name
# (cipher suites, min version). RARE.
# options:
# name: tls-options
- name: tcp
port: 2222
protocol: TCP
allowedRoutes:
namespaces:
from: Same
kinds:
- kind: TCPRoute
- name: udp
port: 3478
protocol: UDP
allowedRoutes:
namespaces:
from: Same
kinds:
- kind: UDPRoute
+181
View File
@@ -0,0 +1,181 @@
# Exhaustive HTTPRoute reference (gateway.networking.k8s.io/v1).
# Attaches to a Gateway listener via parentRefs and routes HTTP(S) by
# hostname + path/method/headers/query. Rules are evaluated in order; the
# first matching rule wins.
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: app
namespace: example
labels:
app: app
spec:
parentRefs:
# Every entry picks one listener to bind to. Omit sectionName/port to
# attach to ALL listeners of the Gateway (common for simple setups).
- name: example
# namespace: Gateway's namespace. Omit when same-namespace (the norm;
# cross-namespace needs the Gateway to allow it).
# namespace: example
# kind/group: default Gateway / gateway.networking.k8s.io. Set
# explicitly only for non-Gateway parents (mesh service parents).
kind: Gateway
group: gateway.networking.k8s.io
# sectionName: the listener name from the Gateway (http/https/...).
sectionName: https
# port: narrow further to one listener port. RARE when sectionName is
# already set.
port: 443
# hostnames: which Host headers this Route serves. Must intersect the
# listener hostname; otherwise the Route is rejected as incompatible.
# Omit to match every hostname on the listener.
hostnames:
- app.example.com
- www.example.com
rules:
# Rule 1: API traffic with header manipulation and canary split.
- matches:
# All conditions inside one match are ANDed; several matches in the
# list are ORed.
- path:
# type: Exact (one URL), PathPrefix (subtree), RegularExpression.
type: PathPrefix
value: /api
method: POST
headers:
# type: Exact or RegularExpression. Names are case-insensitive
# per HTTP spec; values are case-sensitive.
- type: Exact
name: X-Api-Version
value: v2
queryParams:
# Match on ?debug=true style parameters.
- type: Exact
name: debug
value: "true"
filters:
# Filters run in order and transform the request/response.
- type: RequestHeaderModifier
requestHeaderModifier:
# add: append even if present (duplicates allowed). set:
# overwrite-or-add. remove: delete by name.
add:
- name: X-Gateway
value: example
set:
- name: X-Forwarded-Proto
value: https
remove:
- X-Internal-Token
- type: ResponseHeaderModifier
responseHeaderModifier:
set:
- name: X-Frame-Options
value: DENY
remove:
- Server
# backendRefs below carry per-backend filters too; rule-level filters
# apply to every backend of this rule.
backendRefs:
- name: app-service
# port: the Service port (number). Required - unlike backendRef in
# Ingress, there is no default.
port: 80
# group/kind: default Service / core (""). Other kinds (e.g. a
# ServiceImport for multi-cluster) set kind + group explicitly.
kind: Service
group: ""
# weight: traffic share. 90/10 below = canary: 90% stable, 10% new.
weight: 90
filters:
# Per-backend filter: only this backend's requests get it.
- type: RequestHeaderModifier
requestHeaderModifier:
set:
- name: X-Backend
value: stable
- name: app-canary-service
port: 80
weight: 10
filters:
- type: RequestHeaderModifier
requestHeaderModifier:
set:
- name: X-Backend
value: canary
# timeouts: per-attempt deadlines. request = whole gateway-to-client
# exchange; backendRequest = single backend try.
timeouts:
request: 30s
backendRequest: 10s
# sessionPersistence: stick a client to one backend (cookie-based).
# Type Cookie or Header; absoluteTimeout caps the stickiness.
sessionPersistence:
sessionName: route-session
type: Cookie
absoluteTimeout: 1h
cookieConfig:
lifetimeType: Session
# Rule 2: redirect old path to a new URL.
- matches:
- path:
type: PathPrefix
value: /old-docs
filters:
- type: RequestRedirect
requestRedirect:
# Any combination: scheme/host/port/path/statusCode. Unset fields
# keep the original value.
scheme: https
hostname: docs.example.com
path:
# type: ReplaceFullPath or ReplacePrefixMatch (rewrite the
# matched prefix, keep the remainder).
type: ReplacePrefixMatch
replacePrefixMatch: /docs
port: 443
# statusCode: 301 (permanent) or 302 (temporary).
statusCode: 301
# Rule 3: rewrite the URL but still proxy (client sees no redirect).
- matches:
- path:
type: PathPrefix
value: /shop
filters:
- type: URLRewrite
urlRewrite:
path:
type: ReplacePrefixMatch
replacePrefixMatch: /store
# hostname: also rewrite the Host header sent upstream.
# hostname: store-internal.example.com
backendRefs:
- name: app-service
port: 80
# Rule 4: mirror (shadow) traffic to a second backend for testing.
# The mirror gets a copy; its response is discarded.
- matches:
- path:
type: Exact
value: /checkout
filters:
- type: RequestMirror
requestMirror:
backendRef:
name: app-shadow-service
port: 80
backendRefs:
- name: app-service
port: 80
# Rule 5: delegate to an implementation-specific filter (auth plugin,
# rate limit, wasm). The controller documents the group/kind it honors.
# - filters:
# - type: ExtensionRef
# extensionRef:
# group: example.com
# kind: AuthPolicy
# name: app-auth
# Catch-all rule (no matches): everything not matched above lands here.
- backendRefs:
- name: app-service
port: 80
+52
View File
@@ -0,0 +1,52 @@
# Exhaustive Traefik IngressRouteTCP reference (traefik.io/v1alpha1).
# Routes raw TCP: SSH, databases, or TLS-passthrough where Traefik never
# decrypts. Two TLS modes exist - termination (Traefik holds the cert) and
# passthrough (backend holds the cert) - and they are mutually exclusive.
apiVersion: traefik.io/v1alpha1
kind: IngressRouteTCP
metadata:
name: app-ssh
namespace: example
labels:
app: app
spec:
entryPoints:
- ssh
routes:
# Plain TCP (SSH here): no TLS block at all, bytes flow as-is.
- match: HostSNI(`*`)
# HostSNI matches the TLS Server Name Indication. `*` accepts anything
# (required for non-TLS protocols like SSH that send no SNI).
# With TLS + a real hostname: HostSNI(`db.example.com`).
# middlewares: TCP middleware chain (IP allowlist, rate limit...).
# middlewares:
# - name: ssh-allowlist
# priority: same semantics as HTTP - higher wins.
# priority: 10
services:
- name: app-service
port: 2222
# weight: share of connections across backends.
# weight: 1
# terminationDelay: linger after backend close to drain in-flight
# data. Default 100ms; raise for slow protocols.
terminationDelay: 100
# proxyProtocol: PROXY header toward the backend (v1/v2) so it
# learns real client IPs.
# proxyProtocol:
# version: 2
# TLS termination: Traefik decrypts with its own cert, forwards plaintext.
# - match: HostSNI(`db.example.com`)
# services:
# - name: app-service
# port: 5432
# tls: enable TLS handling on this route. Omit entirely for plain TCP.
# tls:
# Either termination...
# secretName: app-tcp-tls
# options:
# name: modern-tls
# domains:
# - main: db.example.com
# ...or passthrough (Traefik never sees plaintext; needs SNI routing):
# passthrough: true
+19
View File
@@ -0,0 +1,19 @@
# Exhaustive Traefik IngressRouteUDP reference (traefik.io/v1alpha1).
# Routes UDP datagrams (DNS, STUN/TURN, syslog...). No match rules exist -
# UDP has no hostname/SNI to route on, so one route per entrypoint simply
# forwards everything it receives.
apiVersion: traefik.io/v1alpha1
kind: IngressRouteUDP
metadata:
name: app-stun
namespace: example
labels:
app: app
spec:
entryPoints:
- stun
services:
- name: app-service
port: 3478
# weight: share of datagrams when several backends are listed.
weight: 1
+93
View File
@@ -0,0 +1,93 @@
# Exhaustive Traefik IngressRoute reference (traefik.io/v1alpha1, HTTP).
# Routes are evaluated top to bottom by priority, then by rule length: the
# first matching route handles the request. Keep specific rules above the
# catch-all.
apiVersion: traefik.io/v1alpha1
kind: IngressRoute
metadata:
name: app-prod
namespace: example
labels:
app: app
annotations:
description: "exhaustive traefik http route example"
spec:
# entryPoints: static entrypoints the route listens on (ports Traefik was
# started with: web=:80, websecure=:443, plus any custom ones).
entryPoints:
- websecure
routes:
# Rule 1: API subtree with middleware chain and weighted backends.
- match: Host(`app.example.com`) && PathPrefix(`/api`)
# kind: Rule (match traffic) or the same match used for redirections.
kind: Rule
# priority: explicit precedence. Higher wins regardless of position.
# Default is the rule length in characters - explicit numbers beat
# clever ordering.
priority: 100
# middlewares: request pipeline, in order (auth, headers, rate limit,
# redirect, strip prefix...). Same-namespace by name; cross-namespace
# as name@namespace (never across providers without the suffix).
middlewares:
- name: app-auth
- name: security-headers
services:
- name: app-service
# port: Service port number or name.
port: 80
# scheme: http (default), https (TLS backend), h2c (cleartext
# HTTP/2, e.g. gRPC without TLS).
scheme: http
# weight: traffic share for canary/blue-green splits.
weight: 90
# serversTransport: ServersTransport CRD with TLS/forwarding
# tuning for this backend (rootCAs, insecureSkipVerify...).
# serversTransport: app-transport
# responseForwarding:
# flushInterval: 100ms
# passHostHeader: forward the original Host header (default true).
# passHostHeader: true
# proxyProtocol: speak PROXY protocol to the backend so it sees
# real client IPs. Version v1 or v2; backend must understand it.
# proxyProtocol:
# version: 2
- name: app-canary-service
port: 80
weight: 10
# Rule 2: multiple hosts, regex path, external backend by URL.
- match: (Host(`app.example.com`) || Host(`www.example.com`)) && PathRegexp(`^/files/.*$`)
kind: Rule
priority: 50
services:
# servers: bypass the Service and address backends directly.
# Only one of name/port (cluster Service) or servers (explicit
# URLs) may be set.
- name: app-service
port: 80
# Catch-all rule: everything not matched above.
- match: Host(`app.example.com`)
kind: Rule
priority: 1
services:
- name: app-service
port: 80
# tls: terminate TLS on this route. Omit the whole block for plain HTTP.
tls:
# secretName: TLS Secret (tls.crt/tls.key) in THIS namespace.
secretName: app-prod-tls
# options: TLSOption CRD (minVersion, cipherSuites, sniStrict...).
# options:
# name: modern-tls
# certResolver: ACME resolver name (letsencrypt-style) that issues the
# certificate on demand. Use EITHER certResolver OR secretName, not both:
# resolver for auto-issued certs, secretName for pre-made ones.
# certResolver: letsencrypt
# store: custom TLSStore for the certificate. Default store otherwise.
# store:
# name: default
# domains: certificates to request/serve (main + SANs). With secretName
# this documents intent; with certResolver it drives issuance.
domains:
- main: app.example.com
sans:
- www.example.com
+100
View File
@@ -0,0 +1,100 @@
# Exhaustive Service reference (v1).
# A Service is a stable virtual endpoint in front of pods: one DNS name and
# one cluster IP that load-balances to the currently Ready pods behind it.
# Pods come and go; the Service name (app-service.example.svc.cluster.local)
# never changes.
apiVersion: v1
kind: Service
metadata:
name: app-service
namespace: example
labels:
app: app
annotations:
description: "exhaustive service example"
spec:
# selector: pods carrying these labels receive traffic. Empty selector =
# no automatic endpoints (pair with a hand-written EndpointSlice to aim at
# an external address - see endpointslice.yaml).
selector:
app: app
ports:
- name: http
# protocol: TCP (default), UDP, or SCTP. Each port entry needs one.
protocol: TCP
# port: the port clients connect to on the Service IP.
port: 80
# targetPort: the port on the pod. Number or container port NAME
# (names decouple the Service from container port renumbering).
# Omit when it equals `port`.
targetPort: http
# nodePort: fixed node port for type NodePort/LoadBalancer (range
# 30000-32767). Omit for auto-assignment.
# nodePort: 30080
# appProtocol: protocol hint for the payload (http, https, grpc, h2c,
# ws...). Used by meshes and LBs, ignored by plain kube-proxy routing.
appProtocol: http
- name: metrics
protocol: TCP
port: 9090
targetPort: 9090
# type: ClusterIP (default, internal VIP), NodePort (also open a fixed port
# on every node), LoadBalancer (NodePort + cloud LB in front),
# ExternalName (DNS alias, no proxying - see below).
type: ClusterIP
# clusterIP: pin the virtual IP. "None" makes the Service headless: no VIP,
# DNS returns pod IPs directly (required base for StatefulSets).
# clusterIP: None
# clusterIPs / ipFamilies / ipFamilyPolicy: dual-stack control.
# ipFamilies: [IPv4] (default), [IPv6], or [IPv4, IPv6].
# ipFamilyPolicy: SingleStack (default), PreferDualStack, RequireDualStack.
# ipFamilies:
# - IPv4
# ipFamilyPolicy: SingleStack
# sessionAffinity: None (default, spread every connection) or ClientIP
# (same client IP sticks to one pod).
sessionAffinity: None
# sessionAffinityConfig: stickiness TTL for ClientIP affinity.
# sessionAffinityConfig:
# clientIP:
# timeoutSeconds: 10800
# publishNotReadyAddresses: send traffic to not-Ready pods too. Needed for
# peer discovery where members must find each other before anyone is Ready.
publishNotReadyAddresses: false
# internalTrafficPolicy: Cluster (default, route to pods on any node) or
# Local (only pods on the receiving node - preserves source IP, drops
# traffic on nodes without local pods).
internalTrafficPolicy: Cluster
# --- NodePort / LoadBalancer extras (ignored by pure ClusterIP) ---
# externalTrafficPolicy: Cluster (default) or Local. Local preserves the
# client source IP but drops traffic arriving on nodes with no local pod.
# externalTrafficPolicy: Cluster
# healthCheckNodePort: fixed node port for the LB health check (Local
# policy). Omit for auto-assignment.
# healthCheckNodePort: 32111
# allocateLoadBalancerNodePorts: set false to skip node-port allocation on
# a LoadBalancer (when the LB routes straight to pods). Default true.
# allocateLoadBalancerNodePorts: true
# loadBalancerIP: request a specific IP from the cloud provider. Provider
# support varies; most now prefer annotations or IP pools.
# loadBalancerIP: 203.0.113.10
# loadBalancerSourceRanges: client CIDRs allowed through the cloud LB.
# Unset = world-open. This is the cloud firewall in front of the Service.
# loadBalancerSourceRanges:
# - 203.0.113.0/24
# loadBalancerClass: use a custom LB implementation instead of the cloud
# default (e.g. MetalLB speaker). The named controller must be installed.
# loadBalancerClass: example.com/custom-lb
# trafficDistribution: hint how to prefer endpoints (PreferClose = same
# zone first). Best-effort, kube-proxy dependent.
# trafficDistribution: PreferClose
# --- other types ---
# externalIPs: extra IPs (already routed to nodes) that also serve this
# Service. Traffic arriving there is proxied like ClusterIP traffic.
# externalIPs:
# - 203.0.113.20
# ExternalName type: no proxying at all - DNS CNAME to the target.
# Ports are informational. Used to reference outside names under a stable
# in-cluster name.
# type: ExternalName
# externalName: db.example.com
+166
View File
@@ -0,0 +1,166 @@
# Exhaustive StatefulSet reference (apps/v1), shown with its headless Service.
# StatefulSets give each pod a stable name and stable storage:
# web-0, web-1, ... each reattached to its own volume after rescheduling.
# The classic use is databases and anything with an identity (postgres,
# redis, kafka). For the full container/pod field catalog see
# deployment.yaml - only StatefulSet-specific fields are expanded here.
apiVersion: v1
kind: Service
metadata:
name: example-sts
namespace: example
labels:
app: example-sts
spec:
# clusterIP: None makes the Service headless: no virtual IP, DNS returns
# the pod IPs directly (web-0.example-sts...). Required for StatefulSets -
# it is how stable network identity works.
clusterIP: None
# publishNotReadyAddresses: include not-Ready pods in DNS. Needed for
# peer discovery when members must find each other before anyone is Ready
# (etcd, clustered databases).
publishNotReadyAddresses: false
selector:
app: example-sts
ports:
- name: db
port: 5432
targetPort: db
# sessionAffinity: None (default, spread connections) or ClientIP (sticky
# sessions to one pod). Stateful apps sometimes want ClientIP.
sessionAffinity: None
---
apiVersion: apps/v1
kind: StatefulSet
metadata:
name: example-sts
namespace: example
labels:
app: example-sts
spec:
# serviceName: the headless Service above. Must exist; governs the pods'
# DNS domain. Changing it requires recreating the StatefulSet.
serviceName: example-sts
# replicas: pod count. Pods start in order 0..N-1 and stop in reverse.
replicas: 3
# revisionHistoryLimit: old ControllerRevisions kept for rollback.
revisionHistoryLimit: 5
# minReadySeconds: pod must stay Ready this long to count as available.
minReadySeconds: 10
# podManagementPolicy: OrderedReady (default - strict 0,1,2 startup order,
# each predecessor must be Ready) or Parallel (start/stop all at once,
# faster for stateless-ish sets that only want stable names).
podManagementPolicy: OrderedReady
# persistentVolumeClaimRetentionPolicy: what happens to per-pod PVCs on
# scale-down (whenDeleted) and StatefulSet deletion (whenScaled). Retain
# keeps data (safe default); Delete wipes it. Set explicitly - the default
# Retain surprises people who expected cleanup.
persistentVolumeClaimRetentionPolicy:
whenDeleted: Retain
whenScaled: Retain
# ordinals: first ordinal (default 0). RARE - used when migrating an
# existing cluster whose numbering starts elsewhere.
# ordinals:
# start: 0
selector:
matchLabels:
app: example-sts
updateStrategy:
# type: RollingUpdate (default) or OnDelete (new pods only replace old
# ones when you delete them manually - full control for databases).
type: RollingUpdate
rollingUpdate:
# partition: only ordinals >= partition are updated. Lets you canary:
# partition 2 updates web-2 first, then lower to 0 for the rest.
partition: 0
# maxUnavailable: how many pods may be down during the update (1.25+).
# StatefulSets traditionally allowed exactly 1; now configurable.
maxUnavailable: 1
template:
metadata:
labels:
app: example-sts
spec:
serviceAccountName: default
automountServiceAccountToken: true
terminationGracePeriodSeconds: 60
# Databases want a long grace: 30s kills a checkpointing postmaster
# mid-write. 60-120 is typical for postgres.
containers:
- name: db
image: postgres:17
imagePullPolicy: IfNotPresent
ports:
- name: db
containerPort: 5432
env:
- name: POSTGRES_USER
value: example
- name: POSTGRES_DB
value: example
- name: POSTGRES_PASSWORD
valueFrom:
secretKeyRef:
name: example-secrets
key: db-password
# Probes for a database: pg_isready via exec is the standard.
# Budgets are generous - killing a recovering database only buys
# another full replay.
startupProbe:
exec:
command: ["sh", "-c", "pg_isready -U example -d example"]
# failureThreshold * periodSeconds = total startup budget
# (60 * 10s = 10 minutes here).
failureThreshold: 60
periodSeconds: 10
timeoutSeconds: 5
readinessProbe:
exec:
command: ["sh", "-c", "pg_isready -U example -d example"]
periodSeconds: 10
timeoutSeconds: 5
livenessProbe:
exec:
command: ["sh", "-c", "pg_isready -U example -d example"]
initialDelaySeconds: 30
periodSeconds: 30
timeoutSeconds: 5
failureThreshold: 3
resources:
requests:
cpu: "100m"
memory: "256Mi"
limits:
cpu: "1000m"
memory: "1Gi"
volumeMounts:
- name: data
mountPath: /var/lib/postgresql/data
# subPath: mounting the volume root directly breaks postgres
# (it wants an empty dir); subPath gives it a subdirectory.
subPath: pgdata
volumes: []
# No shared volumes here: each pod gets its own PVC from the template
# below, mounted under the same `data` name.
volumeClaimTemplates:
# One template entry per volume. Pod web-N gets PVC data-web-N,
# created on first scheduling and (per the retention policy) kept after.
- metadata:
name: data
labels:
app: example-sts
annotations:
description: "per-pod database storage"
spec:
accessModes:
# Must be ReadWriteOnce: one pod owns its volume. (Shared RWX
# defeats the whole point of per-pod volumes.)
- ReadWriteOnce
storageClassName: standard
volumeMode: Filesystem
resources:
requests:
storage: 10Gi
# selector/dataSource/dataSourceRef: same semantics as a plain PVC
# (bind a specific PV, clone, restore from snapshot). See
# deployment.yaml.
+35
View File
@@ -0,0 +1,35 @@
# Exhaustive TCPRoute reference (gateway.networking.k8s.io/v1alpha2).
# Routes raw TCP (SSH, databases, any non-HTTP protocol) from a TCP listener
# to backends. No hostname/path matching exists at this layer - there is
# nothing but the destination port to route on. For TLS with SNI-based
# routing see TLSRoute; for HTTP see HTTPRoute.
apiVersion: gateway.networking.k8s.io/v1alpha2
kind: TCPRoute
metadata:
name: app-ssh
namespace: example
labels:
app: app
spec:
parentRefs:
# Bind to the TCP listener of the Gateway. Same fields as HTTPRoute
# parentRefs: name/kind/group plus sectionName and/or port narrowing.
- name: example
kind: Gateway
group: gateway.networking.k8s.io
sectionName: tcp
port: 2222
rules:
- backendRefs:
- name: app-service
# port: backend Service port. Required.
port: 2222
kind: Service
group: ""
# weight: share of connections when several backends are listed.
weight: 1
# Second backend: every new connection goes 3:1 here. Weights are
# the only traffic-shaping knob TCP routing has.
- name: app-replica-service
port: 2222
weight: 3
+29
View File
@@ -0,0 +1,29 @@
# Exhaustive UDPRoute reference (gateway.networking.k8s.io/v1alpha2).
# Routes raw UDP datagrams (DNS, STUN/TURN, game servers, QUIC-before-TLS)
# from a UDP listener to backends. Same minimal shape as TCPRoute: UDP has
# no sessions or headers to match on, so backendRefs carry the whole rule.
apiVersion: gateway.networking.k8s.io/v1alpha2
kind: UDPRoute
metadata:
name: app-stun
namespace: example
labels:
app: app
spec:
parentRefs:
# Bind to the UDP listener of the Gateway. Same fields as HTTPRoute
# parentRefs: name/kind/group plus sectionName and/or port narrowing.
- name: example
kind: Gateway
group: gateway.networking.k8s.io
sectionName: udp
port: 3478
rules:
- backendRefs:
- name: app-service
# port: backend Service port. Required.
port: 3478
kind: Service
group: ""
# weight: share of traffic when several backends are listed.
weight: 1