apiVersion: apps/v1 kind: Deployment metadata: # name: unique within the namespace name: nginx-deployment # namespace: logical isolation namespace: example labels: # free-form key/value tags used for selection and grouping. app: nginx app.kubernetes.io/name: nginx app.kubernetes.io/instance: nginx-example app.kubernetes.io/version: "1.27" app.kubernetes.io/component: server app.kubernetes.io/part-of: example app.kubernetes.io/managed-by: kubectl annotations: # Key/value, but never used for selection # Only for metadata (descriptions, owners, timestamps). description: "deployment example" owner: team-platform # finalizers: identifiers that block deletion until some controller removes # them after cleanup. RARE on workloads. # finalizers: # - example.com/cleanup spec: # replicas: how many pod copies to keep running. Default 1. replicas: 3 # revisionHistoryLimit: how many old ReplicaSets are kept so you can roll # back. Default 10. revisionHistoryLimit: 5 # progressDeadlineSeconds: if a rollout makes no progress for this long it # is marked ProgressDeadlineExceeded. Default 600. progressDeadlineSeconds: 600 # minReadySeconds: a new pod must stay Ready for this long before it counts # as available. Protects against pods that flap right after start. minReadySeconds: 10 # paused: freezes the rollout controller. RARE - used to accumulate several # changes and release them as a single rollout. paused: false # selector: defines which pods belong to this Deployment. Must match the # pod template labels exactly. Immutable after creation. selector: matchLabels: app: nginx # matchExpressions: set-based selection (In, NotIn, Exists, DoesNotExist). # RARE on Deployments. # matchExpressions: # - key: tier # operator: In # values: [frontend] strategy: # type: RollingUpdate (replace gradually, default) or Recreate (kill all # old pods first). Recreate fits state that cannot have two writers at # once (SQLite file, exclusive lock). type: RollingUpdate rollingUpdate: # maxSurge: how many pods above `replicas` may exist mid-rollout. # Number or percentage. maxSurge: 1 # maxUnavailable: how many pods may be simultaneously down mid-rollout. # Number or percentage. maxUnavailable: 1 template: metadata: labels: app: nginx app.kubernetes.io/name: nginx annotations: description: "nginx pod" spec: # serviceAccountName: the identity pods use against the API server. serviceAccountName: default # automountServiceAccountToken: mount the API token into pods. Set false # for pods that never call the API to shrink the escape blast radius. automountServiceAccountToken: true # schedulerName: which scheduler places the pod. The default scheduler # handles virtually everything. schedulerName: default-scheduler # nodeName: pin the pod to one node, bypassing the scheduler. RARE and # brittle - nodeSelector/affinity express intent better. # nodeName: node-1 # nodeSelector: hard requirement on node labels. nodeSelector: kubernetes.io/os: linux # hostname/subdomain: give the pod a stable hostname and DNS entry # ...svc.cluster.local. Mostly a # StatefulSet concern (which gets this automatically). hostname: nginx subdomain: example-subdomain # setHostnameAsFQDN: use the FQDN above as the hostname. Default false. setHostnameAsFQDN: false # priorityClassName: scheduling priority; higher values preempt lower. # priorityClassName: high-priority # preemptionPolicy: Never stops this pod from preempting others. # Default PreemptLowerPriority. preemptionPolicy: PreemptLowerPriority # runtimeClassName: alternate container runtime (gVisor, Kata). # Omit for the default runtime. # runtimeClassName: gvisor # enableServiceLinks: inject _SERVICE_HOST style env vars. True by # default; false keeps the environment clean when you use DNS only. enableServiceLinks: true # hostAliases: extra /etc/hosts lines. RARE - usually means DNS should # have been fixed instead. hostAliases: - ip: 192.168.1.10 hostnames: - legacy-db.example.com # hostNetwork/hostPID/hostIPC: share the node's network/process/IPC # namespaces. Needed for node-level agents; dangerous for apps. hostNetwork: false hostPID: false hostIPC: false # shareProcessNamespace: all containers in the pod see each other's # processes. Handy for sidecar debuggers; off by default. shareProcessNamespace: false # dnsPolicy: ClusterFirst (default, .svc names resolve), Default # (inherit the node's resolver), ClusterFirstWithHostNet (for # hostNetwork pods), None (dnsConfig takes over completely). dnsPolicy: ClusterFirst dnsConfig: nameservers: - 1.1.1.1 searches: - example.com options: - name: ndots value: "2" # readinessGates: custom conditions (reported by external controllers) # that must be true before the pod counts as Ready. RARE. # readinessGates: # - conditionType: example.com/lb-attached # topologySpreadConstraints: spread pods across zones/hosts with skew # control. The modern, expressive successor of bare podAntiAffinity. topologySpreadConstraints: - maxSkew: 1 topologyKey: kubernetes.io/hostname whenUnsatisfiable: ScheduleAnyway labelSelector: matchLabels: app: nginx affinity: nodeAffinity: # requiredDuringSchedulingIgnoredDuringExecution: hard node rule - # pods that violate it are never scheduled there. requiredDuringSchedulingIgnoredDuringExecution: nodeSelectorTerms: - matchExpressions: - key: kubernetes.io/arch operator: In values: [amd64] # preferredDuringSchedulingIgnoredDuringExecution: soft node rule - # the scheduler tries, but schedules anyway if impossible. preferredDuringSchedulingIgnoredDuringExecution: - weight: 10 preference: matchExpressions: - key: node-role.kubernetes.io/worker operator: Exists podAffinity: # Attract to nodes already running matching pods (data locality). preferredDuringSchedulingIgnoredDuringExecution: - weight: 50 podAffinityTerm: labelSelector: matchLabels: app: cache topologyKey: kubernetes.io/hostname podAntiAffinity: # Repel from nodes running matching pods (spread replicas). preferredDuringSchedulingIgnoredDuringExecution: - weight: 100 podAffinityTerm: labelSelector: matchLabels: app: nginx topologyKey: kubernetes.io/hostname tolerations: # Tolerate a node taint to allow scheduling there. Without a matching # toleration, a tainted node rejects the pod. - key: dedicated operator: Equal value: "true" effect: NoSchedule # tolerationSeconds: with effect NoExecute, tolerate for this long # before eviction. Omit for infinite tolerance. tolerationSeconds: 3600 # imagePullSecrets: credentials for private registries. # imagePullSecrets: # - name: registry-credentials # securityContext (pod-level): defaults inherited by every container # unless a container overrides them. securityContext: runAsUser: 101 runAsGroup: 101 # runAsNonRoot: refuse to start as uid 0. runAsNonRoot: true # fsGroup: group that owns mounted volumes; kubelet chowns on mount. fsGroup: 101 # fsGroupChangePolicy: Always (chown on every mount, slow on large # volumes) or OnRootMismatch (chown only when needed). fsGroupChangePolicy: OnRootMismatch # supplementalGroups: extra groups granted for volume access. supplementalGroups: [102] # sysctls: namespaced kernel parameters. Unsafe ones need explicit # kubelet opt-in. sysctls: - name: net.core.somaxconn value: "1024" # seLinuxOptions: SELinux user/role/type/level. RARE outside MLS. seLinuxOptions: level: s0:c123,c456 # seccompProfile: syscall sandbox. RuntimeDefault is the sane # baseline; Localhost loads a custom profile from the node. seccompProfile: type: RuntimeDefault # windowsOptions: GMSA / runAsUserName for Windows nodes. initContainers: # Run strictly in order, each to completion, before app containers # start. Used for migrations, permission fixes, dependency waits. - name: init-permissions image: busybox:1.36 command: ["sh", "-c", "chown -R 101:101 /data"] volumeMounts: - name: html mountPath: /data resources: requests: cpu: "10m" memory: "16Mi" limits: cpu: "50m" memory: "64Mi" # restartPolicy on a container (not the pod): Always turns it into # a native sidecar that keeps running next to the app (1.28+). # restartPolicy: Always containers: - name: nginx # image: repository plus tag. Pin tags - `latest` moves under you. image: nginx:1.27.3 # imagePullPolicy: Always (re-pull even pinned tags), IfNotPresent # (use cache, works offline), Never (cache only, fails otherwise). imagePullPolicy: IfNotPresent # command: overrides the image ENTRYPOINT. args: overrides CMD. # command: ["nginx"] # args: ["-g", "daemon off;"] # workingDir: overrides the image WORKDIR. workingDir: /usr/share/nginx/html # stdin/stdinOnce/tty: interactive input. For debug shells and # one-shot runs, never for servers. stdin: false stdinOnce: false tty: false ports: - name: http containerPort: 80 protocol: TCP # hostPort: expose straight on the node, bypassing Services. # RARE - allows only one such pod per node per port. # hostPort: 8080 # hostIP: which node address hostPort binds to. # hostIP: 127.0.0.1 env: - name: NGINX_PORT value: "80" - name: POD_NAME valueFrom: fieldRef: # fieldPath exposes pod metadata: metadata.name, # metadata.namespace, metadata.uid, spec.nodeName, # spec.serviceAccountName, status.podIP(s), etc. fieldPath: metadata.name - name: NODE_NAME valueFrom: fieldRef: fieldPath: spec.nodeName - name: POD_MEMORY_LIMIT valueFrom: # resourceFieldRef exposes this container's own # requests/limits. divisor formats the value. resourceFieldRef: resource: limits.memory divisor: 1Mi - name: API_PASSWORD valueFrom: secretKeyRef: name: example-secrets key: api-password # optional: tolerate a missing key (variable stays unset). optional: false - name: LOG_LEVEL valueFrom: configMapKeyRef: name: example-config key: log-level optional: false envFrom: # Bulk-inject every key of a ConfigMap/Secret as env vars. - configMapRef: name: example-config optional: false # prefix: prepended to every injected key, avoids collisions. prefix: APP_ - secretRef: name: example-secrets optional: false resources: # requests: guaranteed reservation used for scheduling. Set at # measured idle/p95 - over-requesting starves neighboring pods. requests: cpu: "100m" memory: "128Mi" # limits: hard ceiling. Breaching memory kills the container # (OOMKilled); breaching CPU only throttles it (slow, not dead). limits: cpu: "500m" memory: "512Mi" # claims: reference a ResourceClaim for dynamic resources # (GPUs via DRA, 1.26+). RARE. # claims: # - name: gpu # resizePolicy: what happens on in-place container resize (1.27+). # NotRequired keeps running; RestartContainer restarts to apply. resizePolicy: - resourceName: cpu restartPolicy: NotRequired - resourceName: memory restartPolicy: NotRequired volumeMounts: - name: html mountPath: /usr/share/nginx/html readOnly: false # subPath: mount a single file/dir of the volume instead of # its root. Typical for single-file config mounts. # subPath: index.html # subPathExpr: subPath assembled from env variables. # subPathExpr: $(POD_NAME)/data # mountPropagation: share mounts back with the host # (HostToContainer, Bidirectional). Storage-driver territory. mountPropagation: None - name: tmp mountPath: /tmp # volumeDevices: raw block devices without a filesystem. RARE - # databases on local PVs with volumeMode: Block. # volumeDevices: # - name: blockvol # devicePath: /dev/xvda livenessProbe: # Exactly one handler per probe: httpGet, tcpSocket, exec, grpc. httpGet: path: /healthz port: http scheme: HTTP # httpHeaders: extra headers sent with the probe request. httpHeaders: - name: Host value: example.com # initialDelaySeconds: wait after start before first probe. initialDelaySeconds: 15 # periodSeconds: interval between probes. periodSeconds: 20 # timeoutSeconds: when a single probe counts as failed. timeoutSeconds: 5 # successThreshold: consecutive successes to count as healthy. # Keep 1. successThreshold: 1 # failureThreshold: consecutive failures to trigger the action. failureThreshold: 3 readinessProbe: # Failing readiness removes the pod from Services (no traffic) # without restarting it. Failing liveness restarts it. httpGet: path: /readyz port: 80 initialDelaySeconds: 5 periodSeconds: 10 timeoutSeconds: 3 successThreshold: 1 failureThreshold: 3 startupProbe: # Disables liveness/readiness until it first succeeds. Total # budget = failureThreshold * periodSeconds (here 30 * 10s). # The cure for slow-starting apps that otherwise get # restart-looped before they finish booting. tcpSocket: port: 80 # host: probe a different host than the pod IP. RARE. failureThreshold: 30 periodSeconds: 10 timeoutSeconds: 5 lifecycle: # postStart runs right after start; preStop runs before SIGTERM. # Slow hooks stall the pod transition - keep them fast. postStart: exec: command: ["sh", "-c", "echo started > /tmp/started"] # Classic preStop: sleep so endpoints are removed everywhere # before the process receives SIGTERM. preStop: exec: command: ["sh", "-c", "sleep 5"] # terminationMessagePath: file whose content becomes the container's # final status message. FallbackToLogsOnError appends log tail when # the file is empty. terminationMessagePath: /dev/termination-log terminationMessagePolicy: File securityContext: allowPrivilegeEscalation: false readOnlyRootFilesystem: false runAsNonRoot: true runAsUser: 101 runAsGroup: 101 capabilities: add: - NET_BIND_SERVICE drop: - ALL seccompProfile: type: RuntimeDefault # ephemeralContainers: troubleshooting shells injected into a RUNNING # pod with kubectl debug. Declared ad-hoc in practice, never committed. # ephemeralContainers: # - name: debugger # image: busybox:1.36 # command: ["sh"] # stdin: true # tty: true # targetContainerName: nginx volumes: - name: html persistentVolumeClaim: claimName: example-pvc # readOnly: mount the claim read-only in this pod. readOnly: false - name: tmp emptyDir: # medium "" (node disk) or Memory (tmpfs). sizeLimit evicts the # pod when exceeded. Dies with the pod either way. medium: "" sizeLimit: 256Mi - name: config-files configMap: # Each key becomes a file under the mount path. name: example-config-files defaultMode: 0644 optional: false items: - key: nginx.conf path: nginx.conf mode: 0644 - name: tls secret: # Secret volumes are tmpfs-backed, never touch node disk. secretName: example-tls defaultMode: 0644 optional: false items: - key: tls.crt path: tls.crt mode: 0644 - name: host-time hostPath: path: /etc/localtime # type: DirectoryOrCreate, Directory, FileOrCreate, File, # Socket, CharDevice, BlockDevice. Always set it: a missing # path then fails loudly instead of silently creating the # wrong filesystem object. type: File - name: podinfo downwardAPI: # Expose pod metadata as files. items: - path: labels fieldRef: fieldPath: metadata.labels - path: cpu-request resourceFieldRef: resource: requests.cpu containerName: nginx divisor: 1m - name: all-config projected: # Merge several sources into a single directory. defaultMode: 0644 sources: - configMap: name: example-config items: - key: log-level path: log-level - secret: name: example-secrets items: - key: api-password path: password - downwardAPI: items: - path: podname fieldRef: fieldPath: metadata.name # Further volume types share the same `name:` + type shape: # nfs: { server, path }, csi: (storage drivers), persistentVolumeClaim # shown above, plus legacy in-tree plugins (fc, iscsi, rbd, glusterfs) # and gitRepo (deprecated - use initContainers + emptyDir instead). # restartPolicy: only Always is valid for Deployments. (Jobs use # OnFailure/Never; bare pods accept all three.) restartPolicy: Always terminationGracePeriodSeconds: 30 # activeDeadlineSeconds: kill the pod after this long no matter what. # A Job concern, not a server concern - shown only for completeness. # activeDeadlineSeconds: 3600 --- apiVersion: v1 kind: PersistentVolumeClaim metadata: name: example-pvc namespace: example spec: # accessModes: ReadWriteOnce (one node), ReadOnlyMany, ReadWriteMany # (needs a shared filesystem), ReadWriteOncePod (single pod, strictest). accessModes: - ReadWriteOnce # storageClassName: selects the provisioner. "" (empty) disables dynamic # provisioning and binds a pre-created volume instead. storageClassName: standard # volumeMode: Filesystem (default) or Block (used with volumeDevices). volumeMode: Filesystem resources: requests: storage: 5Gi # limits: accepted on paper, almost no provisioner enforces them. # selector: bind a specific pre-created PV by labels. RARE with dynamic # provisioning. # selector: # matchLabels: # disk: ssd # dataSource: clone this PVC from another PVC or a snapshot at creation. # dataSource: # name: example-snapshot # kind: VolumeSnapshot # apiGroup: snapshot.storage.k8s.io # dataSourceRef: cross-namespace-capable successor of dataSource.