apiVersion: apps/v1 kind: Deployment metadata: name: prop-grid-rs namespace: prop # Phase 2 cutover: Rust writes directly to /data/scores. Elixir only # runs the f00 analysis-hour step (which carries enrichment Rust # doesn't own yet — native-level duct merge, NEXRAD composite, # commercial-link degradation, ProfilesFile write). f01..f18 land # here from the Rust worker. spec: # Replicas claim from grid_tasks via FOR UPDATE SKIP LOCKED with no # coordination; anti-affinity spreads them across distinct hosts so # any single node going down leaves the hourly chain with other # workers still draining. # Replica count is HPA-managed (hpa.yaml, min 1 / max 4). minReadySeconds: 5 # Zero-downtime rollout for the propagation chain. With HPA min=1 # and the previous maxSurge:0/maxUnavailable:1 config, every deploy # tore the single replica down before bringing the new one up, # leaving a multi-minute window where the :05 hourly cron had no # worker to claim tasks. Surge-first means the new pod is ready # before the old one drains; the two briefly overlap and race for # `grid_tasks` rows via FOR UPDATE SKIP LOCKED, which is safe. strategy: type: RollingUpdate rollingUpdate: maxSurge: 1 maxUnavailable: 0 selector: matchLabels: app: prop-grid-rs template: metadata: labels: app: prop-grid-rs tier: grid-rs annotations: # Scraped by the Prometheus at 10.0.15.31 via the # `kubernetes-pods` job in prometheus.yml.j2 (apiserver proxy). # Metrics endpoint is on container port 9100 (see METRICS_ADDR # env below). `prometheus.io/job` makes the dashboard's # `up{job="prop-grid-rs"}` selector work without per-pod # relabel. prometheus.io/scrape: "true" prometheus.io/port: "9100" prometheus.io/path: "/metrics" prometheus.io/job: "prop-grid-rs" spec: # No node pinning — scheduler picks any host. Anti-affinity was # removed so HPA can stack multiple replicas on the same node if # that's where the scheduler wants them; grid-rs is CPU-bound but # not memory-bound so colocating a couple of replicas is fine. # Wide window for an in-flight chain step (typically 1–3 min) to # finish before SIGKILL. The worker handles SIGTERM by stopping # task pickup; releasing the current claim is the cron's job # (visibility timeout reclaims abandoned `grid_tasks` rows). terminationGracePeriodSeconds: 300 imagePullSecrets: - name: forgejo-registry securityContext: runAsUser: 65534 runAsNonRoot: true fsGroup: 65534 seccompProfile: type: RuntimeDefault containers: - name: prop-grid-rs # Image built by a new CI pipeline from rust/prop_grid_rs/. # Multi-arch (amd64 + arm64) — reuses the repo's existing # buildx wiring. image: git.mcintire.me/graham/prop-grid-rs:main-1781279001-e879f29 imagePullPolicy: IfNotPresent env: # Elixir secrets bundle carries DATABASE_URL already. Rust # reads it directly (sqlx connects with the same URL). - name: HRRR_FALLBACK_BASE_URL value: "https://noaa-hrrr-bdp-pds.s3.amazonaws.com" - name: PROP_SCORES_DIR value: "/data/scores" # Shared idx-file cache on NFS. All grid-rs replicas read/write # through the same directory; idx bodies are immutable per HRRR # run, so cross-replica hits eliminate redundant NOAA S3 calls. - name: HRRR_IDX_CACHE_DIR value: "/data/hrrr_idx" - name: RUST_LOG value: "info" # Was "1": a decoded grid used to be ~95 k nested HashMaps # (~200-400 MB per in-flight task), so two concurrent tasks # plus the wgrib2 working set crossed the 3 Gi limit. The # dense `FieldGrid` representation stores the same data as # ~48 f32 planes (~18 MB), and the profile artifact is now a # flat f32 write instead of an rmpv tree + gzip -9, so the # per-task footprint is roughly an order of magnitude lower. # # Raising to 3 first and holding the memory limit; drop the # limit only after observing steady-state RSS at this # parallelism (one knob per deploy). - name: PROP_GRID_RS_PARALLELISM value: "3" # Pool size = parallelism + 2 (NOTIFY + retry slack). Kept in # sync with PROP_GRID_RS_PARALLELISM when either changes. - name: PROP_GRID_RS_PG_CONNS value: "5" - name: METRICS_ADDR value: "0.0.0.0:9100" envFrom: - secretRef: name: prop-secrets ports: - name: metrics containerPort: 9100 protocol: TCP # Liveness is "the process is still running"; let the kubelet # restart us if we crash. Claiming work from grid_tasks is the # implicit readiness signal — an idle pod is still healthy # (the queue is simply empty). A minimal readiness check on # /health confirms the metrics listener is up, which doubles # as evidence the tokio runtime is alive. readinessProbe: httpGet: path: /health port: 9100 # Generous initialDelaySeconds so a slow DB connect during # startup (Postgres acquire_timeout is 10s) can't race the # first probe. failureThreshold=6 keeps the pod from # flapping if a single scrape times out during a long # chain step. initialDelaySeconds: 15 periodSeconds: 10 timeoutSeconds: 3 failureThreshold: 6 resources: requests: cpu: 100m memory: 768Mi limits: cpu: "4" # After the dense-grid rewrite (commit 63f25a96) the # per-task grid footprint dropped from ~200-400 MB of # nested HashMaps to ~18 MB of dense f32 planes. # PARALLELISM went 1 → 3 in the same commit. The remaining # large term is the f00 native-level decode (~530 MB GRIB2 # + wgrib2 working set). 1.5Gi leaves headroom for three # concurrent chain steps plus one analysis step. memory: 1.5Gi securityContext: allowPrivilegeEscalation: false runAsNonRoot: true runAsUser: 65534 capabilities: drop: - ALL seccompProfile: type: RuntimeDefault volumeMounts: - name: data mountPath: /data volumes: - name: data nfs: server: 10.0.19.103 path: /data