apiVersion: apps/v1 kind: Deployment metadata: name: prop namespace: prop spec: # replicas is owned by the HPA in hpa.yaml (currently pinned at 1). # Leaving it out here so the GitOps reconciler doesn't overwrite # whatever replica count the autoscaler has settled on. minReadySeconds: 5 strategy: type: RollingUpdate rollingUpdate: # Zero-downtime rollout with a single replica: surge to 2 first, # wait for the new pod to pass its readinessProbe, THEN tear down # the old one. maxUnavailable: 1 (the K8s default) was killing the # only pod before the replacement was ready, taking the site down # for the entire image pull + boot window on every deploy. maxSurge: 1 maxUnavailable: 0 selector: matchLabels: app: prop template: metadata: labels: # `tier: hot` distinguishes these pods from prop-backfill (same # app: prop label) so the Service selector can exclude backfill # from its endpoint pool — backfill runs PHX_SERVER=false and # doesn't listen on :5000. app: prop tier: hot spec: tolerations: - key: node-role.kubernetes.io/control-plane operator: Exists effect: NoSchedule serviceAccountName: prop imagePullSecrets: - name: forgejo-registry securityContext: runAsUser: 65534 runAsNonRoot: true fsGroup: 65534 seccompProfile: type: RuntimeDefault containers: - name: prop image: git.mcintire.me/graham/prop:main-1777300373-610892e imagePullPolicy: IfNotPresent env: - name: POD_IP valueFrom: fieldRef: fieldPath: status.podIP - name: PHX_SERVER value: "true" - name: PHX_HOST value: "prop.w5isp.com" - name: PORT value: "5000" - name: HRRR_BASE_URL value: "http://skippy.w5isp.com:8080" envFrom: - secretRef: name: prop-secrets ports: - name: http containerPort: 5000 protocol: TCP startupProbe: httpGet: path: /health port: 5000 initialDelaySeconds: 5 periodSeconds: 3 failureThreshold: 10 timeoutSeconds: 3 readinessProbe: httpGet: path: /health port: 5000 initialDelaySeconds: 5 periodSeconds: 5 failureThreshold: 2 timeoutSeconds: 2 # /live skips the DB — a saturated Ecto pool (e.g. a PromEx # poller holding a connection >15s) would otherwise cause # /health's `SELECT 1` to time out and the kubelet to SIGKILL # every replica at once. Readiness still points at /health so # DB outages drain the pod from Service endpoints gracefully. # timeoutSeconds raised to 5s: under an IEM 429 retry storm # Bandit's acceptor loop can stall briefly as the scheduler # chews through retrying HTTP sleeps, and 3s was tripping # failureThreshold across all 3 replicas at once. livenessProbe: httpGet: path: /live port: 5000 periodSeconds: 10 failureThreshold: 3 timeoutSeconds: 5 resources: requests: cpu: 100m memory: 512Mi limits: # Bumped 2 → 3 on 2026-04-23: BEAM was getting 2 schedulers # online on a 2-core limit, and the :05 propagation chain # plus ScoreCache warming spiked the second run queue to # ~10, starving /live and /health probe handlers and # tripping liveness/readiness timeouts. cpu: "3" memory: 6Gi securityContext: allowPrivilegeEscalation: false runAsNonRoot: true runAsUser: 65534 capabilities: drop: - ALL seccompProfile: type: RuntimeDefault volumeMounts: - name: data mountPath: /data volumes: - name: data nfs: server: 10.0.15.103 path: /data