fix(path): single grid decode + zero-downtime rust-worker rollouts
This commit is contained in:
parent
055be81442
commit
b1fe077863
4 changed files with 85 additions and 23 deletions
|
|
@ -15,11 +15,18 @@ spec:
|
||||||
# workers still draining.
|
# workers still draining.
|
||||||
# Replica count is HPA-managed (hpa.yaml, min 1 / max 4).
|
# Replica count is HPA-managed (hpa.yaml, min 1 / max 4).
|
||||||
minReadySeconds: 5
|
minReadySeconds: 5
|
||||||
|
# Zero-downtime rollout for the propagation chain. With HPA min=1
|
||||||
|
# and the previous maxSurge:0/maxUnavailable:1 config, every deploy
|
||||||
|
# tore the single replica down before bringing the new one up,
|
||||||
|
# leaving a multi-minute window where the :05 hourly cron had no
|
||||||
|
# worker to claim tasks. Surge-first means the new pod is ready
|
||||||
|
# before the old one drains; the two briefly overlap and race for
|
||||||
|
# `grid_tasks` rows via FOR UPDATE SKIP LOCKED, which is safe.
|
||||||
strategy:
|
strategy:
|
||||||
type: RollingUpdate
|
type: RollingUpdate
|
||||||
rollingUpdate:
|
rollingUpdate:
|
||||||
maxSurge: 0
|
maxSurge: 1
|
||||||
maxUnavailable: 1
|
maxUnavailable: 0
|
||||||
selector:
|
selector:
|
||||||
matchLabels:
|
matchLabels:
|
||||||
app: prop-grid-rs
|
app: prop-grid-rs
|
||||||
|
|
@ -39,6 +46,11 @@ spec:
|
||||||
# removed so HPA can stack multiple replicas on the same node if
|
# removed so HPA can stack multiple replicas on the same node if
|
||||||
# that's where the scheduler wants them; grid-rs is CPU-bound but
|
# that's where the scheduler wants them; grid-rs is CPU-bound but
|
||||||
# not memory-bound so colocating a couple of replicas is fine.
|
# not memory-bound so colocating a couple of replicas is fine.
|
||||||
|
# Wide window for an in-flight chain step (typically 1–3 min) to
|
||||||
|
# finish before SIGKILL. The worker handles SIGTERM by stopping
|
||||||
|
# task pickup; releasing the current claim is the cron's job
|
||||||
|
# (visibility timeout reclaims abandoned `grid_tasks` rows).
|
||||||
|
terminationGracePeriodSeconds: 300
|
||||||
imagePullSecrets:
|
imagePullSecrets:
|
||||||
- name: forgejo-registry
|
- name: forgejo-registry
|
||||||
securityContext:
|
securityContext:
|
||||||
|
|
|
||||||
|
|
@ -12,11 +12,16 @@ spec:
|
||||||
# claiming rows from hrrr_fetch_tasks via FOR UPDATE SKIP LOCKED —
|
# claiming rows from hrrr_fetch_tasks via FOR UPDATE SKIP LOCKED —
|
||||||
# no coordination needed between workers.
|
# no coordination needed between workers.
|
||||||
# Replica count is HPA-managed (hpa.yaml, min 1 / max 4).
|
# Replica count is HPA-managed (hpa.yaml, min 1 / max 4).
|
||||||
|
# Zero-downtime rollout: surge first so the new pod is serving
|
||||||
|
# before the old one drains, otherwise a deploy during a QSO-submit
|
||||||
|
# burst leaves the per-QSO fetch lane empty until the new pod is
|
||||||
|
# ready. Both pods race for hrrr_fetch_tasks via FOR UPDATE SKIP
|
||||||
|
# LOCKED, so brief overlap is safe.
|
||||||
strategy:
|
strategy:
|
||||||
type: RollingUpdate
|
type: RollingUpdate
|
||||||
rollingUpdate:
|
rollingUpdate:
|
||||||
maxSurge: 0
|
maxSurge: 1
|
||||||
maxUnavailable: 1
|
maxUnavailable: 0
|
||||||
selector:
|
selector:
|
||||||
matchLabels:
|
matchLabels:
|
||||||
app: hrrr-point-rs
|
app: hrrr-point-rs
|
||||||
|
|
@ -29,6 +34,10 @@ spec:
|
||||||
# Anti-affinity against prop-grid-rs was removed; the scheduler
|
# Anti-affinity against prop-grid-rs was removed; the scheduler
|
||||||
# is free to colocate. Contention on wgrib2 + NFS hasn't been an
|
# is free to colocate. Contention on wgrib2 + NFS hasn't been an
|
||||||
# observed bottleneck in practice.
|
# observed bottleneck in practice.
|
||||||
|
# Per-QSO point fetches are short (~5-15 s), but a stack of
|
||||||
|
# claimed rows can take a couple of minutes to drain. 120 s lets
|
||||||
|
# the worker finish in-flight tasks before SIGKILL.
|
||||||
|
terminationGracePeriodSeconds: 120
|
||||||
imagePullSecrets:
|
imagePullSecrets:
|
||||||
- name: forgejo-registry
|
- name: forgejo-registry
|
||||||
securityContext:
|
securityContext:
|
||||||
|
|
|
||||||
|
|
@ -323,7 +323,13 @@ defmodule Microwaveprop.Propagation.ProfilesFile do
|
||||||
:ok
|
:ok
|
||||||
end
|
end
|
||||||
|
|
||||||
defp snap(lat, lon) do
|
@doc """
|
||||||
|
Snap a lat/lon pair to the nearest grid cell key used in the decoded
|
||||||
|
profile map. Public so callers that have already loaded the full grid
|
||||||
|
via `read/1` can do their own keyed lookups without re-fetching.
|
||||||
|
"""
|
||||||
|
@spec snap(float(), float()) :: {float(), float()}
|
||||||
|
def snap(lat, lon) do
|
||||||
step = Grid.step()
|
step = Grid.step()
|
||||||
snapped_lat = Float.round(Float.round(lat / step) * step, 3)
|
snapped_lat = Float.round(Float.round(lat / step) * step, 3)
|
||||||
snapped_lon = Float.round(Float.round(lon / step) * step, 3)
|
snapped_lon = Float.round(Float.round(lon / step) * step, 3)
|
||||||
|
|
|
||||||
|
|
@ -296,17 +296,23 @@ defmodule MicrowavepropWeb.PathLive do
|
||||||
midlon = (src.lon + dst.lon) / 2
|
midlon = (src.lon + dst.lon) / 2
|
||||||
|
|
||||||
profile_valid_time = latest_profile_valid_time(now)
|
profile_valid_time = latest_profile_valid_time(now)
|
||||||
|
# Load the full profile grid ONCE (single NFS read + gunzip + decode)
|
||||||
|
# instead of letting 9 async_stream workers each re-decode the same
|
||||||
|
# ~30 MB file. Each per-point lookup becomes a Map.get.
|
||||||
|
profile_grid = profile_grid_for(profile_valid_time)
|
||||||
|
|
||||||
sample_points = path_sample_points(src, dst, 9)
|
sample_points = path_sample_points(src, dst, 9)
|
||||||
|
|
||||||
hrrr_points =
|
# Fast path: in-memory Map.get against the single decoded grid.
|
||||||
sample_points
|
# Only points that miss the grid (e.g. just-rolled-over chain) take
|
||||||
|> Task.async_stream(&label_hrrr_point(&1, now, profile_valid_time),
|
# the slow DB-fallback path concurrently below.
|
||||||
max_concurrency: 9,
|
{grid_hits, misses} = partition_grid_hits(sample_points, profile_grid, profile_valid_time)
|
||||||
timeout: 10_000,
|
|
||||||
on_timeout: :kill_task
|
fallback_hits =
|
||||||
)
|
misses
|
||||||
|> Enum.zip(sample_points)
|
|> Enum.reverse()
|
||||||
|
|> Task.async_stream(&fallback_hrrr_point(&1, now), max_concurrency: 4, timeout: 5_000, on_timeout: :kill_task)
|
||||||
|
|> Enum.zip(Enum.reverse(misses))
|
||||||
|> Enum.flat_map(fn
|
|> Enum.flat_map(fn
|
||||||
{{:ok, nil}, _} ->
|
{{:ok, nil}, _} ->
|
||||||
[]
|
[]
|
||||||
|
|
@ -316,12 +322,14 @@ defmodule MicrowavepropWeb.PathLive do
|
||||||
|
|
||||||
{{:exit, reason}, {label, lat, lon}} ->
|
{{:exit, reason}, {label, lat, lon}} ->
|
||||||
Logger.error(
|
Logger.error(
|
||||||
"PathLive HRRR point lookup failed: label=#{inspect(label)} lat=#{lat} lon=#{lon} valid_time=#{inspect(profile_valid_time)} reason=#{inspect(reason)}"
|
"PathLive HRRR fallback lookup failed: label=#{inspect(label)} lat=#{lat} lon=#{lon} reason=#{inspect(reason)}"
|
||||||
)
|
)
|
||||||
|
|
||||||
[]
|
[]
|
||||||
end)
|
end)
|
||||||
|
|
||||||
|
hrrr_points = Enum.reverse(grid_hits, fallback_hits)
|
||||||
|
|
||||||
hrrr_profiles = Enum.map(hrrr_points, & &1.profile)
|
hrrr_profiles = Enum.map(hrrr_points, & &1.profile)
|
||||||
|
|
||||||
# Independent duct signal: the nearest RAOB within 300 km / ±3 h of
|
# Independent duct signal: the nearest RAOB within 300 km / ±3 h of
|
||||||
|
|
@ -455,16 +463,43 @@ defmodule MicrowavepropWeb.PathLive do
|
||||||
defp humidity_effect_string(%{humidity_effect: :harmful}), do: "harmful"
|
defp humidity_effect_string(%{humidity_effect: :harmful}), do: "harmful"
|
||||||
defp humidity_effect_string(_), do: "neutral"
|
defp humidity_effect_string(_), do: "neutral"
|
||||||
|
|
||||||
defp label_hrrr_point({label, lat, lon}, now, profile_valid_time) do
|
# The grid is the single decoded output of ProfilesFile.read/1 — keys are
|
||||||
case profile_valid_time && ProfilesFile.read_point(profile_valid_time, lat, lon) do
|
# `{snapped_lat, snapped_lon}` tuples produced by ProfilesFile.snap/2.
|
||||||
%{} = cell ->
|
# We mirror the same snapping locally instead of round-tripping through
|
||||||
%{label: label, profile: profile_from_cell(cell, lat, lon, profile_valid_time)}
|
# ProfilesFile.read_point/3 (which would re-fetch the cached grid).
|
||||||
|
defp profile_grid_for(nil), do: nil
|
||||||
|
|
||||||
_ ->
|
defp profile_grid_for(valid_time) do
|
||||||
case Weather.find_nearest_hrrr(lat, lon, now) do
|
case ProfilesFile.read(valid_time) do
|
||||||
nil -> nil
|
{:ok, grid} -> grid
|
||||||
profile -> %{label: label, profile: profile}
|
_ -> nil
|
||||||
end
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
defp lookup_profile_grid(grid, lat, lon) do
|
||||||
|
{snapped_lat, snapped_lon} = ProfilesFile.snap(lat, lon)
|
||||||
|
Map.get(grid, {snapped_lat, snapped_lon})
|
||||||
|
end
|
||||||
|
|
||||||
|
defp partition_grid_hits(sample_points, profile_grid, profile_valid_time) do
|
||||||
|
Enum.reduce(sample_points, {[], []}, fn pt, acc ->
|
||||||
|
classify_grid_hit(pt, profile_grid, profile_valid_time, acc)
|
||||||
|
end)
|
||||||
|
end
|
||||||
|
|
||||||
|
defp classify_grid_hit(pt, nil, _vt, {hits, mss}), do: {hits, [pt | mss]}
|
||||||
|
|
||||||
|
defp classify_grid_hit({label, lat, lon} = pt, grid, vt, {hits, mss}) do
|
||||||
|
case lookup_profile_grid(grid, lat, lon) do
|
||||||
|
nil -> {hits, [pt | mss]}
|
||||||
|
cell -> {[%{label: label, profile: profile_from_cell(cell, lat, lon, vt)} | hits], mss}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
defp fallback_hrrr_point({label, lat, lon}, now) do
|
||||||
|
case Weather.find_nearest_hrrr(lat, lon, now) do
|
||||||
|
nil -> nil
|
||||||
|
profile -> %{label: label, profile: profile}
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue