apiVersion: apps/v1 kind: Deployment metadata: name: prop-grid-rs namespace: prop # Phase 2 cutover: Rust writes directly to /data/scores. Elixir only # runs the f00 analysis-hour step (which carries enrichment Rust # doesn't own yet — native-level duct merge, NEXRAD composite, # commercial-link degradation, ProfilesFile write). f01..f18 land # here from the Rust worker. spec: # Three replicas claim from grid_tasks via FOR UPDATE SKIP LOCKED # with no coordination. Anti-affinity spreads them across distinct # hosts so any single node going down leaves the hourly chain with # two workers still draining. replicas: 3 minReadySeconds: 5 strategy: type: RollingUpdate rollingUpdate: maxSurge: 0 maxUnavailable: 1 selector: matchLabels: app: prop-grid-rs template: metadata: labels: app: prop-grid-rs tier: grid-rs annotations: # Scraped by the Prometheus at 10.0.15.25. Metrics endpoint is # on container port 9100 (see METRICS_ADDR env below). prometheus.io/scrape: "true" prometheus.io/port: "9100" prometheus.io/path: "/metrics" spec: # No node pinning — let the scheduler place replicas on any host # the anti-affinity rule allows. talos5 is reserved for Postgres. affinity: podAntiAffinity: requiredDuringSchedulingIgnoredDuringExecution: - labelSelector: matchLabels: app: prop-grid-rs topologyKey: kubernetes.io/hostname imagePullSecrets: - name: forgejo-registry securityContext: runAsUser: 65534 runAsNonRoot: true fsGroup: 65534 seccompProfile: type: RuntimeDefault containers: - name: prop-grid-rs # Image built by a new CI pipeline from rust/prop_grid_rs/. # Multi-arch (amd64 + arm64) — reuses the repo's existing # buildx wiring. image: git.mcintire.me/graham/prop-grid-rs:main-1776809182-e9a3862 # {"$imagepolicy": "flux-system:prop-grid-rs"} imagePullPolicy: IfNotPresent env: # Elixir secrets bundle carries DATABASE_URL already. Rust # reads it directly (sqlx connects with the same URL). - name: HRRR_BASE_URL value: "http://skippy.w5isp.com:8080" # When the LAN caching proxy at skippy is unreachable, fetches # retry once against the NOAA S3 public bucket so a proxy # outage degrades to direct S3 reads instead of stalling the # hourly pipeline. - name: HRRR_FALLBACK_BASE_URL value: "https://noaa-hrrr-bdp-pds.s3.amazonaws.com" - name: PROP_SCORES_DIR value: "/data/scores" # Shared idx-file cache on NFS. All grid-rs replicas read/write # through the same directory; idx bodies are immutable per HRRR # run, so cross-replica hits eliminate redundant NOAA S3 calls. - name: HRRR_IDX_CACHE_DIR value: "/data/hrrr_idx" - name: RUST_LOG value: "info" # Single concurrent task per pod. 2 replicas → 2 concurrent # cluster-wide. Raised from 2 after the new 3 Gi limit still # OOMed under 2-parallel forecast load — the NFS-backed # score-file writes (23 files × ~2 MB each) keep cgroup page # cache high enough that two concurrent tasks plus the # wgrib2 working set regularly crosses 3 Gi. Single-lane # parallelism stays comfortably under 2 Gi and lets the # analysis step (native duct + NEXRAD + commercial merges) # run safely within the same budget. - name: PROP_GRID_RS_PARALLELISM value: "1" # Pool size = parallelism + 2 (NOTIFY + retry slack). Kept in # sync with PROP_GRID_RS_PARALLELISM when either changes. - name: PROP_GRID_RS_PG_CONNS value: "3" - name: METRICS_ADDR value: "0.0.0.0:9100" # OTLP/gRPC endpoint for the cluster OTel Collector. Unset # disables span export entirely (fmt layer still logs to # stdout). See vntx-infra/observability/ for the collector # + Tempo backend. - name: OTEL_EXPORTER_OTLP_ENDPOINT value: "http://otel-collector.observability.svc.cluster.local:4317" envFrom: - secretRef: name: prop-secrets ports: - name: metrics containerPort: 9100 protocol: TCP # Liveness is "the process is still running"; let the kubelet # restart us if we crash. Claiming work from grid_tasks is the # implicit readiness signal — an idle pod is still healthy # (the queue is simply empty). A minimal readiness check on # /health confirms the metrics listener is up, which doubles # as evidence the tokio runtime is alive. readinessProbe: httpGet: path: /health port: 9100 # Generous initialDelaySeconds so a slow DB connect during # startup (Postgres acquire_timeout is 10s) can't race the # first probe. failureThreshold=6 keeps the pod from # flapping if a single scrape times out during a long # chain step. initialDelaySeconds: 15 periodSeconds: 10 timeoutSeconds: 3 failureThreshold: 6 resources: requests: cpu: 100m memory: 768Mi limits: cpu: "4" # Analysis tasks (f00) run the native-level HRRR decode # (~530 MB GRIB2 + wgrib2 working set) on top of 2 concurrent # forecast tasks. 2 Gi OOM'd reliably in the 13 h after # Stream A cutover. 3 Gi leaves room for the analysis # wgrib2 peak without eliminating the forecast-lane # parallelism headroom. memory: 3Gi securityContext: allowPrivilegeEscalation: false runAsNonRoot: true runAsUser: 65534 capabilities: drop: - ALL seccompProfile: type: RuntimeDefault volumeMounts: - name: data mountPath: /data volumes: - name: data nfs: server: 10.0.15.103 path: /data