# Exhaustive StatefulSet reference (apps/v1), shown with its headless Service. # StatefulSets give each pod a stable name and stable storage: # web-0, web-1, ... each reattached to its own volume after rescheduling. # The classic use is databases and anything with an identity (postgres, # redis, kafka). For the full container/pod field catalog see # deployment.yaml - only StatefulSet-specific fields are expanded here. apiVersion: v1 kind: Service metadata: name: example-sts namespace: example labels: app: example-sts spec: # clusterIP: None makes the Service headless: no virtual IP, DNS returns # the pod IPs directly (web-0.example-sts...). Required for StatefulSets - # it is how stable network identity works. clusterIP: None # publishNotReadyAddresses: include not-Ready pods in DNS. Needed for # peer discovery when members must find each other before anyone is Ready # (etcd, clustered databases). publishNotReadyAddresses: false selector: app: example-sts ports: - name: db port: 5432 targetPort: db # sessionAffinity: None (default, spread connections) or ClientIP (sticky # sessions to one pod). Stateful apps sometimes want ClientIP. sessionAffinity: None --- apiVersion: apps/v1 kind: StatefulSet metadata: name: example-sts namespace: example labels: app: example-sts spec: # serviceName: the headless Service above. Must exist; governs the pods' # DNS domain. Changing it requires recreating the StatefulSet. serviceName: example-sts # replicas: pod count. Pods start in order 0..N-1 and stop in reverse. replicas: 3 # revisionHistoryLimit: old ControllerRevisions kept for rollback. revisionHistoryLimit: 5 # minReadySeconds: pod must stay Ready this long to count as available. minReadySeconds: 10 # podManagementPolicy: OrderedReady (default - strict 0,1,2 startup order, # each predecessor must be Ready) or Parallel (start/stop all at once, # faster for stateless-ish sets that only want stable names). podManagementPolicy: OrderedReady # persistentVolumeClaimRetentionPolicy: what happens to per-pod PVCs on # scale-down (whenDeleted) and StatefulSet deletion (whenScaled). Retain # keeps data (safe default); Delete wipes it. Set explicitly - the default # Retain surprises people who expected cleanup. persistentVolumeClaimRetentionPolicy: whenDeleted: Retain whenScaled: Retain # ordinals: first ordinal (default 0). RARE - used when migrating an # existing cluster whose numbering starts elsewhere. # ordinals: # start: 0 selector: matchLabels: app: example-sts updateStrategy: # type: RollingUpdate (default) or OnDelete (new pods only replace old # ones when you delete them manually - full control for databases). type: RollingUpdate rollingUpdate: # partition: only ordinals >= partition are updated. Lets you canary: # partition 2 updates web-2 first, then lower to 0 for the rest. partition: 0 # maxUnavailable: how many pods may be down during the update (1.25+). # StatefulSets traditionally allowed exactly 1; now configurable. maxUnavailable: 1 template: metadata: labels: app: example-sts spec: serviceAccountName: default automountServiceAccountToken: true terminationGracePeriodSeconds: 60 # Databases want a long grace: 30s kills a checkpointing postmaster # mid-write. 60-120 is typical for postgres. containers: - name: db image: postgres:17 imagePullPolicy: IfNotPresent ports: - name: db containerPort: 5432 env: - name: POSTGRES_USER value: example - name: POSTGRES_DB value: example - name: POSTGRES_PASSWORD valueFrom: secretKeyRef: name: example-secrets key: db-password # Probes for a database: pg_isready via exec is the standard. # Budgets are generous - killing a recovering database only buys # another full replay. startupProbe: exec: command: ["sh", "-c", "pg_isready -U example -d example"] # failureThreshold * periodSeconds = total startup budget # (60 * 10s = 10 minutes here). failureThreshold: 60 periodSeconds: 10 timeoutSeconds: 5 readinessProbe: exec: command: ["sh", "-c", "pg_isready -U example -d example"] periodSeconds: 10 timeoutSeconds: 5 livenessProbe: exec: command: ["sh", "-c", "pg_isready -U example -d example"] initialDelaySeconds: 30 periodSeconds: 30 timeoutSeconds: 5 failureThreshold: 3 resources: requests: cpu: "100m" memory: "256Mi" limits: cpu: "1000m" memory: "1Gi" volumeMounts: - name: data mountPath: /var/lib/postgresql/data # subPath: mounting the volume root directly breaks postgres # (it wants an empty dir); subPath gives it a subdirectory. subPath: pgdata volumes: [] # No shared volumes here: each pod gets its own PVC from the template # below, mounted under the same `data` name. volumeClaimTemplates: # One template entry per volume. Pod web-N gets PVC data-web-N, # created on first scheduling and (per the retention policy) kept after. - metadata: name: data labels: app: example-sts annotations: description: "per-pod database storage" spec: accessModes: # Must be ReadWriteOnce: one pod owns its volume. (Shared RWX # defeats the whole point of per-pod volumes.) - ReadWriteOnce storageClassName: standard volumeMode: Filesystem resources: requests: storage: 10Gi # selector/dataSource/dataSourceRef: same semantics as a plain PVC # (bind a specific PV, clone, restore from snapshot). See # deployment.yaml.