prop/k8s/deployment-grid-rs.yaml
Graham McIntire 38cf1336b7
chore(k8s): point deployments at codeberg.org images
The image tags here are placeholders — the live tag is set by the
ArgoCD Application's spec.source.kustomize.images field, same as
before. Only the registry hostname and path changed.

Pull secret 'forgejo-registry' in the prop namespace now carries
both git.mcintire.me and codeberg.org auths during the transition.
2026-05-05 10:38:47 -05:00

161 lines
6.8 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

apiVersion: apps/v1
kind: Deployment
metadata:
name: prop-grid-rs
namespace: prop
# Phase 2 cutover: Rust writes directly to /data/scores. Elixir only
# runs the f00 analysis-hour step (which carries enrichment Rust
# doesn't own yet — native-level duct merge, NEXRAD composite,
# commercial-link degradation, ProfilesFile write). f01..f18 land
# here from the Rust worker.
spec:
# Replicas claim from grid_tasks via FOR UPDATE SKIP LOCKED with no
# coordination; anti-affinity spreads them across distinct hosts so
# any single node going down leaves the hourly chain with other
# workers still draining.
# Replica count is HPA-managed (hpa.yaml, min 1 / max 4).
minReadySeconds: 5
# Zero-downtime rollout for the propagation chain. With HPA min=1
# and the previous maxSurge:0/maxUnavailable:1 config, every deploy
# tore the single replica down before bringing the new one up,
# leaving a multi-minute window where the :05 hourly cron had no
# worker to claim tasks. Surge-first means the new pod is ready
# before the old one drains; the two briefly overlap and race for
# `grid_tasks` rows via FOR UPDATE SKIP LOCKED, which is safe.
strategy:
type: RollingUpdate
rollingUpdate:
maxSurge: 1
maxUnavailable: 0
selector:
matchLabels:
app: prop-grid-rs
template:
metadata:
labels:
app: prop-grid-rs
tier: grid-rs
annotations:
# Scraped by the Prometheus at 10.0.15.25. Metrics endpoint is
# on container port 9100 (see METRICS_ADDR env below).
prometheus.io/scrape: "true"
prometheus.io/port: "9100"
prometheus.io/path: "/metrics"
spec:
# No node pinning — scheduler picks any host. Anti-affinity was
# removed so HPA can stack multiple replicas on the same node if
# that's where the scheduler wants them; grid-rs is CPU-bound but
# not memory-bound so colocating a couple of replicas is fine.
# Wide window for an in-flight chain step (typically 13 min) to
# finish before SIGKILL. The worker handles SIGTERM by stopping
# task pickup; releasing the current claim is the cron's job
# (visibility timeout reclaims abandoned `grid_tasks` rows).
terminationGracePeriodSeconds: 300
imagePullSecrets:
- name: forgejo-registry
securityContext:
runAsUser: 65534
runAsNonRoot: true
fsGroup: 65534
seccompProfile:
type: RuntimeDefault
containers:
- name: prop-grid-rs
# Image built by a new CI pipeline from rust/prop_grid_rs/.
# Multi-arch (amd64 + arm64) — reuses the repo's existing
# buildx wiring.
image: codeberg.org/gmcintire/prop-grid-rs:main-1777132414-6aa91e7
imagePullPolicy: IfNotPresent
env:
# Elixir secrets bundle carries DATABASE_URL already. Rust
# reads it directly (sqlx connects with the same URL).
- name: HRRR_BASE_URL
value: "http://skippy.w5isp.com:8080"
# When the LAN caching proxy at skippy is unreachable, fetches
# retry once against the NOAA S3 public bucket so a proxy
# outage degrades to direct S3 reads instead of stalling the
# hourly pipeline.
- name: HRRR_FALLBACK_BASE_URL
value: "https://noaa-hrrr-bdp-pds.s3.amazonaws.com"
- name: PROP_SCORES_DIR
value: "/data/scores"
# Shared idx-file cache on NFS. All grid-rs replicas read/write
# through the same directory; idx bodies are immutable per HRRR
# run, so cross-replica hits eliminate redundant NOAA S3 calls.
- name: HRRR_IDX_CACHE_DIR
value: "/data/hrrr_idx"
- name: RUST_LOG
value: "info"
# Single concurrent task per pod. 2 replicas → 2 concurrent
# cluster-wide. Raised from 2 after the new 3 Gi limit still
# OOMed under 2-parallel forecast load — the NFS-backed
# score-file writes (23 files × ~2 MB each) keep cgroup page
# cache high enough that two concurrent tasks plus the
# wgrib2 working set regularly crosses 3 Gi. Single-lane
# parallelism stays comfortably under 2 Gi and lets the
# analysis step (native duct + NEXRAD + commercial merges)
# run safely within the same budget.
- name: PROP_GRID_RS_PARALLELISM
value: "1"
# Pool size = parallelism + 2 (NOTIFY + retry slack). Kept in
# sync with PROP_GRID_RS_PARALLELISM when either changes.
- name: PROP_GRID_RS_PG_CONNS
value: "3"
- name: METRICS_ADDR
value: "0.0.0.0:9100"
envFrom:
- secretRef:
name: prop-secrets
ports:
- name: metrics
containerPort: 9100
protocol: TCP
# Liveness is "the process is still running"; let the kubelet
# restart us if we crash. Claiming work from grid_tasks is the
# implicit readiness signal — an idle pod is still healthy
# (the queue is simply empty). A minimal readiness check on
# /health confirms the metrics listener is up, which doubles
# as evidence the tokio runtime is alive.
readinessProbe:
httpGet:
path: /health
port: 9100
# Generous initialDelaySeconds so a slow DB connect during
# startup (Postgres acquire_timeout is 10s) can't race the
# first probe. failureThreshold=6 keeps the pod from
# flapping if a single scrape times out during a long
# chain step.
initialDelaySeconds: 15
periodSeconds: 10
timeoutSeconds: 3
failureThreshold: 6
resources:
requests:
cpu: 100m
memory: 768Mi
limits:
cpu: "4"
# Analysis tasks (f00) run the native-level HRRR decode
# (~530 MB GRIB2 + wgrib2 working set) on top of 2 concurrent
# forecast tasks. 2 Gi OOM'd reliably in the 13 h after
# Stream A cutover. 3 Gi leaves room for the analysis
# wgrib2 peak without eliminating the forecast-lane
# parallelism headroom.
memory: 3Gi
securityContext:
allowPrivilegeEscalation: false
runAsNonRoot: true
runAsUser: 65534
capabilities:
drop:
- ALL
seccompProfile:
type: RuntimeDefault
volumeMounts:
- name: data
mountPath: /data
volumes:
- name: data
nfs:
server: 10.0.15.103
path: /data