perf(propagation): shrink hourly chain wall time
Four changes sized by measured prod telemetry (83m of spans): 1. propagation queue: 2 → 1 slot per pod. Two concurrent forecast-hour steps per pod stacked HRRR grid + native duct grid + scored band map into ~5-6 GiB RSS, OOM-killing every ~15 min. 3-way parallelism cluster-wide still finishes the chain inside the hourly interval. 2. weather queue: 3 → 1 slot per pod. ASOS backfill was 429-thrashing IEM (1,296 retryable jobs; logs were nothing but 429 backoffs). 3. PropagationGridWorker: skip native-level duct fetch on f01..f18. At ~7-11 min/fh and 18 forecast hours, this was the largest single cost per chain. Forecast hours fall back to derived[:min_refractivity_gradient] from the pressure-level profile. f00 still gets full native-level duct analysis. 4. HrrrClient.download_grib_ranges_to_file: parallelize with Task.async_stream (max_concurrency 8). The file-backed variant was sequential, dominating native-duct fetch time on the remaining f00 path. ~20s → ~3s per call.
This commit is contained in:
parent
95a769bf0e
commit
01b181b1e8
4 changed files with 100 additions and 24 deletions
|
|
@ -163,12 +163,23 @@ if config_env() == :prod do
|
||||||
engine: Oban.Pro.Engines.Smart,
|
engine: Oban.Pro.Engines.Smart,
|
||||||
# Per-pod concurrency (×3 pods = effective cluster total)
|
# Per-pod concurrency (×3 pods = effective cluster total)
|
||||||
queues: [
|
queues: [
|
||||||
# 2 slots so PropagationPruneWorker can run alongside the
|
# 1 slot per pod. Two concurrent forecast-hour steps in a single
|
||||||
# forecast-hour chain job without blocking.
|
# pod stack HRRR grid + native duct grid + scored band map into
|
||||||
propagation: 2,
|
# ~5-6 GiB RSS, hitting the 6 GiB limit and OOM-killing the pod
|
||||||
|
# mid-chain. With 3 pods we still get 3-way parallel fan-out
|
||||||
|
# cluster-wide, which is enough to finish the f00–f18 chain
|
||||||
|
# comfortably inside the hourly interval. PropagationPruneWorker
|
||||||
|
# shares this queue at a lower priority; if a prune job blocks
|
||||||
|
# the grid chain briefly, the chain's unique window protects
|
||||||
|
# the next hourly fire.
|
||||||
|
propagation: 1,
|
||||||
commercial: 1,
|
commercial: 1,
|
||||||
solar: 1,
|
solar: 1,
|
||||||
weather: 3,
|
# 1 slot per pod (3 cluster-wide). Any higher than this and the
|
||||||
|
# ASOS backfill hammers IEM hard enough to get HTTP 429s, which
|
||||||
|
# fill the retryable queue and thrash pod CPU without making
|
||||||
|
# forward progress.
|
||||||
|
weather: 1,
|
||||||
gefs: 1,
|
gefs: 1,
|
||||||
hrrr: 2,
|
hrrr: 2,
|
||||||
terrain: 3,
|
terrain: 3,
|
||||||
|
|
|
||||||
|
|
@ -468,26 +468,48 @@ defmodule Microwaveprop.Weather.HrrrClient do
|
||||||
def download_grib_ranges_to_file(_url, [], _dest_path), do: :ok
|
def download_grib_ranges_to_file(_url, [], _dest_path), do: :ok
|
||||||
|
|
||||||
def download_grib_ranges_to_file(url, ranges, dest_path) do
|
def download_grib_ranges_to_file(url, ranges, dest_path) do
|
||||||
merged = ranges |> merge_ranges() |> Enum.sort_by(&elem(&1, 0))
|
# Fetch ranges concurrently — Range requests against the HRRR
|
||||||
|
# mirror are independent and typically complete in ~200-500ms each
|
||||||
|
# when sequential. With 40+ ranges per native-duct fetch, serial
|
||||||
|
# download dominated the per-forecast-hour wall time. 8-way
|
||||||
|
# parallelism drops a 20s download to ~3s while still respecting
|
||||||
|
# the mirror's modest connection concurrency.
|
||||||
|
merged = merge_ranges(ranges)
|
||||||
|
|
||||||
file = File.open!(dest_path, [:write, :binary])
|
results =
|
||||||
|
merged
|
||||||
try do
|
|> Task.async_stream(
|
||||||
Enum.reduce_while(merged, :ok, fn {start, stop}, :ok ->
|
fn {start, stop} ->
|
||||||
case Req.get(url, [{:headers, [{"Range", "bytes=#{start}-#{stop}"}]} | req_options()]) do
|
case Req.get(url, [{:headers, [{"Range", "bytes=#{start}-#{stop}"}]} | req_options()]) do
|
||||||
{:ok, %{status: 206, body: body}} ->
|
{:ok, %{status: 206, body: body}} -> {:ok, start, body}
|
||||||
IO.binwrite(file, body)
|
{:ok, %{status: status}} -> {:error, "HRRR grib HTTP #{status}"}
|
||||||
{:cont, :ok}
|
{:error, reason} -> {:error, reason}
|
||||||
|
end
|
||||||
{:ok, %{status: status}} ->
|
end,
|
||||||
{:halt, {:error, "HRRR grib HTTP #{status}"}}
|
max_concurrency: 8,
|
||||||
|
timeout: 120_000
|
||||||
{:error, reason} ->
|
)
|
||||||
{:halt, {:error, reason}}
|
|> Enum.map(fn
|
||||||
end
|
{:ok, result} -> result
|
||||||
|
{:exit, reason} -> {:error, reason}
|
||||||
end)
|
end)
|
||||||
after
|
|
||||||
File.close(file)
|
case Enum.find(results, &match?({:error, _}, &1)) do
|
||||||
|
{:error, reason} ->
|
||||||
|
{:error, reason}
|
||||||
|
|
||||||
|
nil ->
|
||||||
|
file = File.open!(dest_path, [:write, :binary])
|
||||||
|
|
||||||
|
try do
|
||||||
|
results
|
||||||
|
|> Enum.sort_by(fn {:ok, offset, _} -> offset end)
|
||||||
|
|> Enum.each(fn {:ok, _offset, body} -> IO.binwrite(file, body) end)
|
||||||
|
after
|
||||||
|
File.close(file)
|
||||||
|
end
|
||||||
|
|
||||||
|
:ok
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -176,8 +176,18 @@ defmodule Microwaveprop.Workers.PropagationGridWorker do
|
||||||
# Per-contact HRRR enrichment still uses HrrrFetchWorker, which
|
# Per-contact HRRR enrichment still uses HrrrFetchWorker, which
|
||||||
# writes its own `is_grid_point: false` rows.
|
# writes its own `is_grid_point: false` rows.
|
||||||
|
|
||||||
# Fetch native duct metrics and merge into grid_data for scoring
|
# Native-level duct fetch + wgrib2 pass costs ~7-11 min/hour and
|
||||||
grid_data = merge_native_duct_data(grid_data, run_time, forecast_hour)
|
# is only run on f00. Forecast hours fall back to
|
||||||
|
# derived[:min_refractivity_gradient] from the pressure-level
|
||||||
|
# profile — coarser (~250 m vs ~10-50 m) but good enough for
|
||||||
|
# forecast-hour ducting, which is inherently lower-confidence.
|
||||||
|
grid_data =
|
||||||
|
if forecast_hour == 0 do
|
||||||
|
merge_native_duct_data(grid_data, run_time, forecast_hour)
|
||||||
|
else
|
||||||
|
grid_data
|
||||||
|
end
|
||||||
|
|
||||||
:erlang.garbage_collect()
|
:erlang.garbage_collect()
|
||||||
|
|
||||||
# NEXRAD current-hour composite reflectivity catches fast-moving
|
# NEXRAD current-hour composite reflectivity catches fast-moving
|
||||||
|
|
|
||||||
|
|
@ -402,6 +402,39 @@ defmodule Microwaveprop.Weather.HrrrClientTest do
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
describe "download_grib_ranges_to_file/3" do
|
||||||
|
test "writes bytes to file in offset-ascending order even when fetched out of order" do
|
||||||
|
# Two non-adjacent ranges (so they don't get merged). Responses
|
||||||
|
# deliberately return with the second range "arriving" first to
|
||||||
|
# simulate out-of-order completion under parallel fetches; the
|
||||||
|
# final file must still be ordered by offset.
|
||||||
|
ranges = [{0, 9}, {100, 119}]
|
||||||
|
range_a = String.duplicate("A", 10)
|
||||||
|
range_b = String.duplicate("B", 20)
|
||||||
|
|
||||||
|
Req.Test.stub(HrrrClient, fn conn ->
|
||||||
|
[range] = Plug.Conn.get_req_header(conn, "range")
|
||||||
|
|
||||||
|
case range do
|
||||||
|
"bytes=0-9" ->
|
||||||
|
Plug.Conn.send_resp(conn, 206, range_a)
|
||||||
|
|
||||||
|
"bytes=100-119" ->
|
||||||
|
Plug.Conn.send_resp(conn, 206, range_b)
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
|
||||||
|
tmp = Path.join(System.tmp_dir!(), "hrrr_download_test_#{System.unique_integer([:positive])}.grib2")
|
||||||
|
|
||||||
|
try do
|
||||||
|
assert :ok = HrrrClient.download_grib_ranges_to_file("http://test/grib", ranges, tmp)
|
||||||
|
assert File.read!(tmp) == range_a <> range_b
|
||||||
|
after
|
||||||
|
File.rm(tmp)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
describe "merge_grid_data/2" do
|
describe "merge_grid_data/2" do
|
||||||
test "merges surface and pressure grids by point key" do
|
test "merges surface and pressure grids by point key" do
|
||||||
point_a = {32.90, -97.04}
|
point_a = {32.90, -97.04}
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue