apiVersion: apps/v1 kind: Deployment metadata: name: prop namespace: prop spec: # replicas is owned by the HPA in hpa.yaml (min 2 / max 5). # Leaving it out here so Flux's periodic reconcile doesn't overwrite # whatever replica count the autoscaler has settled on. minReadySeconds: 5 strategy: type: RollingUpdate rollingUpdate: maxSurge: 1 maxUnavailable: 1 selector: matchLabels: app: prop template: metadata: labels: # `tier: hot` distinguishes these pods from prop-backfill (same # app: prop label) so the Service selector can exclude backfill # from its endpoint pool — backfill runs PHX_SERVER=false and # doesn't listen on :5000. app: prop tier: hot spec: tolerations: - key: node-role.kubernetes.io/control-plane operator: Exists effect: NoSchedule serviceAccountName: prop imagePullSecrets: - name: forgejo-registry securityContext: runAsUser: 65534 runAsNonRoot: true fsGroup: 65534 seccompProfile: type: RuntimeDefault containers: - name: prop image: git.mcintire.me/graham/prop:main-1777235105-c0a884d # {"$imagepolicy": "flux-system:prop"} imagePullPolicy: IfNotPresent env: - name: POD_IP valueFrom: fieldRef: fieldPath: status.podIP - name: PHX_SERVER value: "true" - name: PHX_HOST value: "prop.w5isp.com" - name: PORT value: "5000" - name: HRRR_BASE_URL value: "http://skippy.w5isp.com:8080" envFrom: - secretRef: name: prop-secrets ports: - name: http containerPort: 5000 protocol: TCP startupProbe: httpGet: path: /health port: 5000 initialDelaySeconds: 5 periodSeconds: 3 failureThreshold: 10 timeoutSeconds: 3 readinessProbe: httpGet: path: /health port: 5000 initialDelaySeconds: 5 periodSeconds: 5 failureThreshold: 2 timeoutSeconds: 2 # /live skips the DB — a saturated Ecto pool (e.g. a PromEx # poller holding a connection >15s) would otherwise cause # /health's `SELECT 1` to time out and the kubelet to SIGKILL # every replica at once. Readiness still points at /health so # DB outages drain the pod from Service endpoints gracefully. # timeoutSeconds raised to 5s: under an IEM 429 retry storm # Bandit's acceptor loop can stall briefly as the scheduler # chews through retrying HTTP sleeps, and 3s was tripping # failureThreshold across all 3 replicas at once. livenessProbe: httpGet: path: /live port: 5000 periodSeconds: 10 failureThreshold: 3 timeoutSeconds: 5 resources: requests: cpu: 100m memory: 512Mi limits: # Bumped 2 → 3 on 2026-04-23: BEAM was getting 2 schedulers # online on a 2-core limit, and the :05 propagation chain # plus ScoreCache warming spiked the second run queue to # ~10, starving /live and /health probe handlers and # tripping liveness/readiness timeouts. cpu: "3" memory: 6Gi securityContext: allowPrivilegeEscalation: false runAsNonRoot: true runAsUser: 65534 capabilities: drop: - ALL seccompProfile: type: RuntimeDefault volumeMounts: - name: data mountPath: /data volumes: - name: data nfs: server: 10.0.15.103 path: /data