prop-grid-rs already exposed prop_grid_rs_chain_step_duration_seconds
+ tasks_in_flight on :9100 but the comment still pointed at the old
prom server (10.0.15.25). hrrr-point-rs had no metrics listener at all.
- metrics.rs: new POINT_BATCH_DURATION / POINT_BATCHES_TOTAL /
POINTS_PROCESSED_TOTAL series + record_point_batch helper, plus unit
tests serialized via a static Mutex (the prometheus crate's global
registry races otherwise)
- hrrr_point_worker.rs: spawn metrics::serve on METRICS_ADDR (default
0.0.0.0:9100), wrap process_batch in InFlightGuard, record outcome +
per-point counts on every batch
- deployment-hrrr-point-rs.yaml: prometheus.io/{scrape,port,path,job}
annotations, METRICS_ADDR env, ports.metrics, /health readiness probe
- deployment-grid-rs.yaml: refresh stale 10.0.15.25 comment, add
prometheus.io/job for clean dashboard up{} selectors
Also fix Microwaveprop.PromExTest: `assert plugins != []` triggered
Elixir 1.19 type-checker warning ("non_empty_list != empty_list always
true"); replaced with membership assertions for the InstrumentPlugin +
Plugins.Beam.
133 lines
5.1 KiB
YAML
133 lines
5.1 KiB
YAML
apiVersion: apps/v1
|
|
kind: Deployment
|
|
metadata:
|
|
name: hrrr-point-rs
|
|
namespace: prop
|
|
# Phase 3 Stream C: per-QSO HRRR fetches move off the BEAM pods.
|
|
# Elixir writes to hrrr_fetch_tasks (one row per valid_time with a
|
|
# points JSONB array); this deployment drains via FOR UPDATE SKIP
|
|
# LOCKED, fetches each HRRR cycle once, and upserts hrrr_profiles.
|
|
spec:
|
|
# Parallelism on QSO-submit spikes comes from multiple replicas
|
|
# claiming rows from hrrr_fetch_tasks via FOR UPDATE SKIP LOCKED —
|
|
# no coordination needed between workers.
|
|
# Replica count is HPA-managed (hpa.yaml, min 1 / max 4).
|
|
# Zero-downtime rollout: surge first so the new pod is serving
|
|
# before the old one drains, otherwise a deploy during a QSO-submit
|
|
# burst leaves the per-QSO fetch lane empty until the new pod is
|
|
# ready. Both pods race for hrrr_fetch_tasks via FOR UPDATE SKIP
|
|
# LOCKED, so brief overlap is safe.
|
|
strategy:
|
|
type: RollingUpdate
|
|
rollingUpdate:
|
|
maxSurge: 1
|
|
maxUnavailable: 0
|
|
selector:
|
|
matchLabels:
|
|
app: hrrr-point-rs
|
|
template:
|
|
metadata:
|
|
labels:
|
|
app: hrrr-point-rs
|
|
tier: hrrr-points
|
|
annotations:
|
|
# Scraped by the Prometheus at 10.0.15.31 via the
|
|
# `kubernetes-pods` job in prometheus.yml.j2 (apiserver proxy).
|
|
# The hrrr_point_worker binary spawns the same metrics::serve
|
|
# listener as prop-grid-rs (METRICS_ADDR, default 0.0.0.0:9100)
|
|
# and reports prop_grid_rs_point_batch_* + points_processed_*
|
|
# series.
|
|
prometheus.io/scrape: "true"
|
|
prometheus.io/port: "9100"
|
|
prometheus.io/path: "/metrics"
|
|
prometheus.io/job: "hrrr-point-rs"
|
|
spec:
|
|
# Anti-affinity against prop-grid-rs was removed; the scheduler
|
|
# is free to colocate. Contention on wgrib2 + NFS hasn't been an
|
|
# observed bottleneck in practice.
|
|
# Per-QSO point fetches are short (~5-15 s), but a stack of
|
|
# claimed rows can take a couple of minutes to drain. 120 s lets
|
|
# the worker finish in-flight tasks before SIGKILL.
|
|
terminationGracePeriodSeconds: 120
|
|
imagePullSecrets:
|
|
- name: forgejo-registry
|
|
securityContext:
|
|
runAsUser: 65534
|
|
runAsNonRoot: true
|
|
fsGroup: 65534
|
|
seccompProfile:
|
|
type: RuntimeDefault
|
|
containers:
|
|
- name: hrrr-point-rs
|
|
# Shares the prop-grid-rs image so one CI pipeline produces
|
|
# both binaries. k8s `command:` overrides the default
|
|
# ENTRYPOINT to launch the per-QSO point worker.
|
|
image: codeberg.org/gmcintire/prop-grid-rs:main-1777132414-6aa91e7
|
|
imagePullPolicy: IfNotPresent
|
|
command: ["/usr/bin/tini", "--", "/usr/local/bin/hrrr-point-worker"]
|
|
env:
|
|
- name: HRRR_BASE_URL
|
|
value: "http://skippy.w5isp.com:8080"
|
|
- name: HRRR_FALLBACK_BASE_URL
|
|
value: "https://noaa-hrrr-bdp-pds.s3.amazonaws.com"
|
|
- name: RUST_LOG
|
|
value: "info"
|
|
- name: PROP_GRID_RS_PG_CONNS
|
|
value: "4"
|
|
# Shared NFS idx cache — same directory as prop-grid-rs so a
|
|
# chain-step idx fetch warms both workers.
|
|
- name: HRRR_IDX_CACHE_DIR
|
|
value: "/data/hrrr_idx"
|
|
# /metrics + /health listener; same default as prop-grid-rs.
|
|
- name: METRICS_ADDR
|
|
value: "0.0.0.0:9100"
|
|
envFrom:
|
|
- secretRef:
|
|
name: prop-secrets
|
|
ports:
|
|
- name: metrics
|
|
containerPort: 9100
|
|
protocol: TCP
|
|
# Readiness mirrors prop-grid-rs: /health returns 200 once the
|
|
# axum metrics listener is bound. Liveness intentionally
|
|
# omitted — an idle worker (no claimable tasks) is healthy,
|
|
# and the kubelet will restart on process crash anyway.
|
|
readinessProbe:
|
|
httpGet:
|
|
path: /health
|
|
port: 9100
|
|
initialDelaySeconds: 10
|
|
periodSeconds: 10
|
|
timeoutSeconds: 3
|
|
failureThreshold: 6
|
|
resources:
|
|
requests:
|
|
cpu: 100m
|
|
memory: 256Mi
|
|
limits:
|
|
# Per-batch peak is a single CONUS surface + pressure
|
|
# decode (~40 MB blob, ~200 MB wgrib2 working set plus
|
|
# the full 92k-cell merged CellValues map kept in memory
|
|
# until all requested points are drained). 1 Gi OOMKilled
|
|
# repeatedly — four restarts in ten minutes — so bumped
|
|
# to 2 Gi. The scheduler-on-worker2 placement has plenty
|
|
# of headroom.
|
|
cpu: "2"
|
|
memory: 2Gi
|
|
securityContext:
|
|
allowPrivilegeEscalation: false
|
|
runAsNonRoot: true
|
|
runAsUser: 65534
|
|
capabilities:
|
|
drop:
|
|
- ALL
|
|
seccompProfile:
|
|
type: RuntimeDefault
|
|
volumeMounts:
|
|
- name: data
|
|
mountPath: /data
|
|
volumes:
|
|
- name: data
|
|
nfs:
|
|
server: 10.0.15.103
|
|
path: /data
|