Telemetry puts nexrad_decode_png at 1.76 s/call × 37,847 calls/h — ~18.5 h of CPU across the cluster every real hour, the single biggest CPU consumer in the system. The map-click path (fetch_rain_cells) already cached decoded frames by 5-min bucket, but the per-contact CommonVolumeRadarWorker path (fetch_decoded_frame) went straight to network + decode on every call. Backfill means many contacts share a 5-min window, so the same 66 MB frame was being decoded dozens of times. Wire fetch_decoded_frame through NexradCache keyed on the rounded timestamp. Add a 20-entry size cap in NexradCache so backfill processing contacts in random timestamp order can't grow the ETS table to hundreds of GB. Each frame is ~66 MB, so 20 = ~1.3 GB worst case, well under typical pod memory. Expected impact: cuts sustained decode load by an order of magnitude depending on backfill temporal locality; map-click path is unchanged.
79 lines
2.4 KiB
Elixir
79 lines
2.4 KiB
Elixir
defmodule Microwaveprop.Weather.NexradCache do
|
||
@moduledoc """
|
||
Node-local ETS cache of decoded NEXRAD n0q composite reflectivity frames.
|
||
Keyed by a 5-minute rounded timestamp. Stores the raw pixel buffer + image
|
||
width so per-point rain-cell extraction can skip the HTTP fetch + PNG decode
|
||
(which take 1-5 seconds for a ~5 MB CONUS-wide image).
|
||
|
||
The cache is populated by `Microwaveprop.Weather.NexradClient.fetch_rain_cells/4`
|
||
on its first call per 5-min window, then reused by every concurrent click
|
||
until the window rolls over.
|
||
"""
|
||
use GenServer
|
||
|
||
@table :nexrad_frame_cache
|
||
|
||
# Each frame is ~66 MB (12,200 × 5,400 palette-index bytes). 20 frames
|
||
# caps at ~1.3 GB — comfortable headroom under a 2 GB pod limit, and
|
||
# enough to hit on the common backfill pattern where a worker chews
|
||
# through contacts in one timestamp window before moving on. Adjust
|
||
# if pod memory limits change or the n0q geometry changes.
|
||
@max_entries 20
|
||
|
||
@type pixels :: binary()
|
||
@type width :: pos_integer()
|
||
|
||
@spec start_link(keyword()) :: GenServer.on_start()
|
||
def start_link(opts) do
|
||
GenServer.start_link(__MODULE__, opts, name: __MODULE__)
|
||
end
|
||
|
||
@spec fetch(DateTime.t()) :: {:ok, pixels(), width()} | :miss
|
||
def fetch(rounded_ts) do
|
||
case :ets.lookup(@table, rounded_ts) do
|
||
[{_, pixels, width}] -> {:ok, pixels, width}
|
||
[] -> :miss
|
||
end
|
||
end
|
||
|
||
@spec put(DateTime.t(), pixels(), width()) :: :ok
|
||
def put(rounded_ts, pixels, width) do
|
||
:ets.insert(@table, {rounded_ts, pixels, width})
|
||
enforce_size_cap()
|
||
:ok
|
||
end
|
||
|
||
defp enforce_size_cap do
|
||
size = :ets.info(@table, :size)
|
||
|
||
if size > @max_entries do
|
||
# Drop the oldest-by-key entries until we're back under the cap.
|
||
# O(n log n) but n is ~20, so this runs in microseconds.
|
||
excess = size - @max_entries
|
||
|
||
@table
|
||
|> :ets.tab2list()
|
||
|> Enum.sort_by(fn {ts, _, _} -> DateTime.to_unix(ts) end)
|
||
|> Enum.take(excess)
|
||
|> Enum.each(fn {ts, _, _} -> :ets.delete(@table, ts) end)
|
||
end
|
||
end
|
||
|
||
@spec prune_older_than(DateTime.t()) :: non_neg_integer()
|
||
def prune_older_than(cutoff_ts) do
|
||
match_spec = [{{:"$1", :_, :_}, [{:<, :"$1", {:const, cutoff_ts}}], [true]}]
|
||
:ets.select_delete(@table, match_spec)
|
||
end
|
||
|
||
@spec clear() :: :ok
|
||
def clear do
|
||
:ets.delete_all_objects(@table)
|
||
:ok
|
||
end
|
||
|
||
@impl true
|
||
def init(_opts) do
|
||
:ets.new(@table, [:set, :named_table, :public, read_concurrency: true])
|
||
{:ok, %{}}
|
||
end
|
||
end
|