prop/lib/microwaveprop/workers/admin_task_worker.ex
Graham McIntire 7b78a2574c
fix(workers): 3 bug/perf fixes from codebase review
1. GridCache: auto-release fill lock when the claimer process crashes.
   claim_fill/1 + release_fill/1 go through the GenServer so the
   server can Process.monitor the caller and clean up the ETS entry
   on :DOWN. Clear/0 now resets both the data table and the lock
   table. Fixes a latent bug where a crashed fill leaked the lock
   indefinitely, preventing every subsequent /weather mount for that
   valid_time from claiming and leaving cache cold.

2. RadarFrameWorker: distinguish permanent vs transient fetch errors.
   404 from the IEM n0q archive is permanent (file will never exist)
   and marks contacts :unavailable as before. Any other error shape
   (5xx, timeout, transport failure) now returns {:error, reason}
   so Oban retries — previously those also pinned contacts at
   :unavailable after a transient outage.

3. AdminTaskWorker.native_derive: replace per-row Repo.update_all
   (N round-trips + N fsyncs) with one UPDATE ... FROM unnest(...)
   per 2000-row batch. For the 10k-profile budget this is one
   network round trip per chunk instead of 10k, and one fsync per
   chunk instead of 10k. Restructured the clause to separate
   derivation (pure) from persistence (I/O).

All three changes are test-covered (grid_cache_test auto-release
test, radar_frame_worker_test 5xx + transport tests, existing
admin_task_worker_test native_derive coverage exercises the new
bulk path). Also drops the scorer_diff no-op test that was
verifying the clause removed in 61da51c.
2026-04-21 09:53:13 -05:00

294 lines
10 KiB
Elixir

defmodule Microwaveprop.Workers.AdminTaskWorker do
@moduledoc """
Oban worker for long-running admin tasks (backtests, climatology, etc).
Enqueued via `Microwaveprop.Release` so that `eval` returns immediately
and the work runs in the normal Oban pipeline without blocking.
Each task type is dispatched by the `"task"` arg:
Microwaveprop.Workers.AdminTaskWorker.new(%{task: "backtest_all"})
Microwaveprop.Workers.AdminTaskWorker.new(%{task: "backtest", feature: "naive_gradient"})
Microwaveprop.Workers.AdminTaskWorker.new(%{task: "climatology", min_samples: 3})
Microwaveprop.Workers.AdminTaskWorker.new(%{task: "native_derive", limit: 10000})
"""
use Oban.Worker, queue: :admin, max_attempts: 1, unique: [period: 60]
import Ecto.Query
alias Microwaveprop.Backtest
alias Microwaveprop.Backtest.Features
alias Microwaveprop.Propagation.Duct
alias Microwaveprop.Propagation.Inversion
alias Microwaveprop.Propagation.Recalibrator
alias Microwaveprop.Repo
alias Microwaveprop.Weather.HrrrNativeProfile
alias Microwaveprop.Weather.ThetaE
require Logger
@impl Oban.Worker
def perform(%Oban.Job{args: %{"task" => "backtest_all"} = args}) do
sample_size = Map.get(args, "sample_size", 5000)
features = Features.all_features()
Logger.info("AdminTask: running consolidated backtest for #{map_size(features)} features")
results = Backtest.consolidated_report(features, sample_size: sample_size)
markdown = Backtest.to_consolidated_markdown(results)
path = "priv/backtest_reports/consolidated.md"
File.mkdir_p!(Path.dirname(path))
File.write!(path, markdown)
Logger.info("AdminTask: consolidated backtest complete, wrote #{path}")
:ok
end
def perform(%Oban.Job{args: %{"task" => "backtest", "feature" => feature_name} = args}) do
sample_size = Map.get(args, "sample_size", 5000)
# `function_exported?/3` returns false for any module that hasn't
# been loaded yet in the BEAM — force-load `Features` first so valid
# feature names aren't rejected on a cold VM. Using
# `String.to_existing_atom/1` inside a rescue guards against
# unbounded atom creation (the admin UI is the only dispatcher, but
# cheap insurance).
with true <- Code.ensure_loaded?(Features),
{:ok, fun_atom} <- safe_to_existing_atom(feature_name),
true <- function_exported?(Features, fun_atom, 3) do
run_backtest(feature_name, fun_atom, sample_size)
else
_ ->
Logger.error("AdminTask: unknown feature #{feature_name}")
{:error, "unknown feature: #{feature_name}"}
end
end
def perform(%Oban.Job{args: %{"task" => "climatology"} = args}) do
min_samples = Map.get(args, "min_samples", 3)
%{rows: combos} =
Repo.query!(
"""
SELECT EXTRACT(MONTH FROM valid_time)::int AS month,
EXTRACT(HOUR FROM valid_time)::int AS hour
FROM hrrr_profiles
WHERE surface_temp_c IS NOT NULL AND is_grid_point = true
GROUP BY 1, 2
ORDER BY 1, 2
""",
[],
timeout: 120_000
)
Logger.info("AdminTask: building climatology (min_samples=#{min_samples}, #{length(combos)} batches)")
total =
combos
|> Enum.with_index(1)
|> Enum.reduce(0, fn {[month, hour], idx}, acc ->
%{num_rows: count} =
Repo.query!(
"""
INSERT INTO hrrr_climatology (id, lat, lon, month, hour,
mean_surface_temp_c, stddev_surface_temp_c, sample_count)
SELECT gen_random_uuid(), lat, lon,
$2 AS month,
$3 AS hour,
AVG(surface_temp_c),
STDDEV_SAMP(surface_temp_c),
COUNT(*)
FROM hrrr_profiles
WHERE surface_temp_c IS NOT NULL
AND is_grid_point = true
AND EXTRACT(MONTH FROM valid_time)::int = $2
AND EXTRACT(HOUR FROM valid_time)::int = $3
GROUP BY lat, lon
HAVING COUNT(*) >= $1
ON CONFLICT (lat, lon, month, hour)
DO UPDATE SET
mean_surface_temp_c = EXCLUDED.mean_surface_temp_c,
stddev_surface_temp_c = EXCLUDED.stddev_surface_temp_c,
sample_count = EXCLUDED.sample_count
""",
[min_samples, month, hour],
timeout: 300_000
)
if rem(idx, 10) == 0, do: Logger.info("AdminTask: climatology [#{idx}/#{length(combos)}]")
acc + count
end)
Logger.info("AdminTask: climatology complete, upserted #{total} records")
:ok
end
def perform(%Oban.Job{args: %{"task" => "native_derive"} = args}) do
limit = Map.get(args, "limit", 10_000)
profiles =
HrrrNativeProfile
|> where([p], is_nil(p.bulk_richardson) and p.level_count > 2)
|> limit(^limit)
|> Repo.all()
Logger.info("AdminTask: deriving fields for #{length(profiles)} native profiles")
rows = Enum.map(profiles, &derive_native_row/1)
count = bulk_update_native_derivations(rows)
Logger.info("AdminTask: derived fields for #{count} profiles")
:ok
end
def perform(%Oban.Job{args: %{"task" => "recalibrate"} = args}) do
sample_size = Map.get(args, "sample_size", 5000)
epochs = Map.get(args, "epochs", 2000)
learning_rate = Map.get(args, "learning_rate", 0.01)
Logger.info("AdminTask: running weight recalibration (sample=#{sample_size}, epochs=#{epochs}, lr=#{learning_rate})")
result =
Recalibrator.fit(
sample_size: sample_size,
epochs: epochs,
learning_rate: learning_rate
)
Logger.info("AdminTask: recalibration complete")
Logger.info(" train_loss=#{result.train_loss}, val_loss=#{result.val_loss}, initial_loss=#{result.initial_loss}")
for {factor, weight} <- Enum.sort(result.weights) do
Logger.info(" #{factor}: #{weight}")
end
:ok
end
def perform(%Oban.Job{args: %{"task" => task}}) do
Logger.error("AdminTask: unknown task #{task}")
{:error, "unknown task: #{task}"}
end
defp safe_to_existing_atom(name) do
{:ok, String.to_existing_atom(name)}
rescue
ArgumentError -> :error
end
defp run_backtest(feature_name, fun_atom, sample_size) do
feature_fun = &apply(Features, fun_atom, [&1, &2, &3])
name = "Microwaveprop.Backtest.Features.#{feature_name}"
Logger.info("AdminTask: running backtest for #{feature_name}")
report = Backtest.evaluate(feature_fun, sample_size: sample_size, feature_name: name)
distance_bins = Backtest.lift_by_distance(feature_fun, sample_size: sample_size)
band_stats = Backtest.lift_by_band(feature_fun, sample_size: sample_size)
markdown = Backtest.to_markdown(report, distance_bins: distance_bins, band_stats: band_stats)
path = "priv/backtest_reports/#{feature_name}.md"
File.mkdir_p!(Path.dirname(path))
File.write!(path, markdown)
Logger.info("AdminTask: backtest for #{feature_name} complete, wrote #{path}")
:ok
end
# Bulk-update via one `UPDATE ... FROM unnest(...)` statement per batch:
# one Postgres round-trip and one fsync instead of N, which was the
# dominant cost for runs of 10k profiles against the remote DB host.
@bulk_update_chunk 2000
defp derive_native_row(profile) do
p = %{
heights_m: profile.heights_m,
temp_k: profile.temp_k,
spfh: profile.spfh,
pressure_pa: profile.pressure_pa,
u_wind_ms: profile.u_wind_ms,
v_wind_ms: profile.v_wind_ms,
tke_m2s2: profile.tke_m2s2,
level_count: profile.level_count
}
{top_m, bulk, shear, theta_jump} =
case Inversion.find_inversion_top(p) do
{:ok, %{height_m: top_h, level_idx: top_idx, base_idx: base_idx}} ->
{
top_h,
Inversion.bulk_richardson(p, base_idx, top_idx),
Inversion.shear_magnitude(p, base_idx, top_idx),
ThetaE.theta_e_jump(p, base_idx, top_idx)
}
:none ->
{nil, nil, nil, nil}
end
duct_result = Duct.analyze(p)
%{
id: profile.id,
inversion_top_m: top_m,
bulk_richardson: bulk,
shear_at_top_ms: shear,
theta_e_jump_k: theta_jump,
ducts: duct_result.ducts,
best_duct_band_ghz: duct_result.best_duct_band_ghz
}
end
defp bulk_update_native_derivations([]), do: 0
defp bulk_update_native_derivations(rows) do
rows
|> Enum.chunk_every(@bulk_update_chunk)
|> Enum.reduce(0, fn chunk, acc -> acc + execute_bulk_update(chunk) end)
end
defp execute_bulk_update(chunk) do
# Ecto loads binary_id as the dashed-string form; Postgrex expects raw
# 16-byte binaries for `uuid[]` params. Dump once per id before send.
ids = Enum.map(chunk, fn %{id: id} -> Ecto.UUID.dump!(id) end)
tops = Enum.map(chunk, & &1.inversion_top_m)
bulks = Enum.map(chunk, & &1.bulk_richardson)
shears = Enum.map(chunk, & &1.shear_at_top_ms)
theta_jumps = Enum.map(chunk, & &1.theta_e_jump_k)
# Ducts are sent as a `text[]` of encoded JSON; each element is cast
# to `jsonb` per-row via the v.ducts::jsonb reference in the UPDATE
# so Postgres parses the JSON payload on ingest rather than storing
# it as a literal string.
ducts_json = Enum.map(chunk, fn row -> Jason.encode!(row.ducts || []) end)
bands = Enum.map(chunk, & &1.best_duct_band_ghz)
sql = """
UPDATE hrrr_native_profiles p
SET inversion_top_m = v.inversion_top_m,
bulk_richardson = v.bulk_richardson,
shear_at_top_ms = v.shear_at_top_ms,
theta_e_jump_k = v.theta_e_jump_k,
ducts = v.ducts::jsonb,
best_duct_band_ghz = v.best_duct_band_ghz,
updated_at = now()
FROM (
SELECT
unnest($1::uuid[]) AS id,
unnest($2::float8[]) AS inversion_top_m,
unnest($3::float8[]) AS bulk_richardson,
unnest($4::float8[]) AS shear_at_top_ms,
unnest($5::float8[]) AS theta_e_jump_k,
unnest($6::text[]) AS ducts,
unnest($7::float8[]) AS best_duct_band_ghz
) v
WHERE p.id = v.id
"""
%{num_rows: n} =
Repo.query!(sql, [ids, tops, bulks, shears, theta_jumps, ducts_json, bands])
n
end
end