prop/config/runtime.exs
Graham McIntire cfc5f7583f
feat(pskr): ingest PSK Reporter MQTT firehose for weather correlation
Subscribes to mqtt.pskreporter.info:1883 over plain TCP and folds
every reception report into hourly aggregates per
(band, sender_grid_4, receiver_grid_4). Each spot is the empirical
"propagation actually occurred on this path right now" signal we'll
correlate against HRRR atmospheric state to recalibrate the scoring
weights. Aggregating at the path-hour bucket collapses 50-reporter
QSOs to one row instead of 50.

Topic filter anchors on USA (DXCC 291) on either sender or receiver
side so the broker pre-filters out everything outside our HRRR
domain. Bands subscribed: VHF and up (6m, 2m, 70cm, 23cm, +
microwave) — HF is dominated by ionospheric propagation, which this
project doesn't model.

Pskr.Client cluster-elects a singleton via :global.register_name —
all replicas start the GenServer but only one connects to MQTT;
the rest stay in :standby and watch for the leader's nodedown to
re-run election. Off in dev/test, on by default in prod
(PSKR_MQTT_ENABLED=false as kill-switch).

Pskr.Aggregator buffers in memory keyed by Pskr.path_key/1 and
flushes every 60s via Repo.insert_all/3 with an additive
ON CONFLICT (count summed, SNR envelope by GREATEST/LEAST,
modes unioned via unnest+DISTINCT). Idempotent across overlapping
flushes and across leader handover.

Dockerfile sets BUILD_WITHOUT_QUIC=1 — emqtt's transitive quicer
dep wants CMake + OpenSSL headers we'd otherwise have to add to
the builder image just to never use QUIC over plain MQTT. Base
image is unchanged; the new dep compiles cleanly into the existing
prop-base runtime.
2026-05-04 09:24:20 -05:00

440 lines
19 KiB
Elixir
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import Config
alias Microwaveprop.Workers.CanadianSoundingFetchWorker
alias Oban.Plugins.Cron
require Logger
# config/runtime.exs is executed for all environments, including
# during releases. It is executed after compilation and before the
# system starts, so it is typically used to load production configuration
# and secrets from environment variables or elsewhere. Do not define
# any compile-time configuration in here, as it won't be applied.
# The block below contains prod specific runtime configuration.
# ## Using releases
#
# If you use `mix release`, you need to explicitly enable the server
# by passing the PHX_SERVER=true when you start it:
#
# PHX_SERVER=true bin/microwaveprop start
#
# Alternatively, you can use `mix phx.gen.release` to generate a `bin/server`
# script that automatically sets the env var above.
if System.get_env("PHX_SERVER") do
config :microwaveprop, MicrowavepropWeb.Endpoint, server: true
end
# Log format: "text" (default) keeps Elixir's built-in line formatter for
# local/dev use; "json" swaps the :default :logger handler to emit a single
# JSON object per line via `LoggerJSON.Formatters.Basic`. The ingress +
# Grafana pipeline is already JSON-aware, so prod defaults to JSON.
log_format =
System.get_env("LOG_FORMAT") ||
if(config_env() == :prod, do: "json", else: "text")
if log_format == "json" do
formatter =
LoggerJSON.Formatters.Basic.new(metadata: [:request_id, :trace_id, :worker, :queue, :job_id])
:logger.update_handler_config(:default, :formatter, formatter)
end
config :microwaveprop, MicrowavepropWeb.Endpoint, http: [port: String.to_integer(System.get_env("PORT", "4000"))]
# MicrowavepropWeb.Plugs.RemoteIp will only honour `Cf-Connecting-Ip` /
# `X-Forwarded-For` when the immediate TCP peer (conn.remote_ip) sits inside
# one of these CIDRs. Set TRUSTED_PROXY_CIDRS to a comma-separated list to
# override; leave unset to keep the module default (loopback + RFC1918 +
# ULA + link-local), which matches the k8s cluster pod/service network.
if trusted_proxy_cidrs = System.get_env("TRUSTED_PROXY_CIDRS") do
cidrs =
trusted_proxy_cidrs
|> String.split(",", trim: true)
|> Enum.map(&String.trim/1)
|> Enum.reject(&(&1 == ""))
config :microwaveprop, :trusted_proxies, cidrs
end
if prometheus_token = System.get_env("PROMETHEUS_AUTH_TOKEN") do
config :microwaveprop, :prometheus_auth_token, prometheus_token
end
if qrz_username = System.get_env("QRZ_USERNAME") do
config :microwaveprop, Microwaveprop.Qrz.Client,
username: qrz_username,
password: System.get_env("QRZ_PASSWORD") || raise("QRZ_PASSWORD missing"),
agent: System.get_env("QRZ_AGENT", "microwaveprop/1.0")
end
if google_api_key = System.get_env("GOOGLE_API_KEY") do
config :microwaveprop, Microwaveprop.Geocoder, api_key: google_api_key
end
if config_env() == :prod do
database_url =
System.get_env("DATABASE_URL") ||
raise """
environment variable DATABASE_URL is missing.
For example: ecto://USER:PASS@HOST/DATABASE
"""
maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: []
# The secret key base is used to sign/encrypt cookies and other secrets.
# A default value is used in config/dev.exs and config/test.exs but you
# want to use a different value for prod and you most likely don't want
# to check this value into version control, so we use an environment
# variable instead.
secret_key_base =
System.get_env("SECRET_KEY_BASE") ||
raise """
environment variable SECRET_KEY_BASE is missing.
You can generate one by calling: mix phx.gen.secret
"""
host = System.get_env("PHX_HOST") || "example.com"
# ## SSL Support
#
# To get SSL working, you will need to add the `https` key
# to your endpoint configuration:
#
# config :microwaveprop, MicrowavepropWeb.Endpoint,
# https: [
# ...,
# port: 443,
# cipher_suite: :strong,
# keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"),
# certfile: System.get_env("SOME_APP_SSL_CERT_PATH")
# ]
#
# The `cipher_suite` is set to `:strong` to support only the
# latest and more secure SSL ciphers. This means old browsers
# and clients may not be supported. You can set it to
# `:compatible` for wider support.
#
# `:keyfile` and `:certfile` expect an absolute path to the key
# and cert in disk or a relative path inside priv, for example
# "priv/ssl/server.key". For all supported SSL configuration
# options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1
#
# We also recommend setting `force_ssl` in your config/prod.exs,
# ensuring no data is ever sent via http, always redirecting to https:
#
# config :microwaveprop, MicrowavepropWeb.Endpoint,
# force_ssl: [hsts: true]
#
# Check `Plug.SSL` for all available options in `force_ssl`.
# ## Configuring the mailer
#
# In production you need to configure the mailer to use a different adapter.
# Here is an example configuration for Mailgun:
#
# config :microwaveprop, Microwaveprop.Mailer,
# adapter: Swoosh.Adapters.Mailgun,
# tls: :always + explicit tls_options is required — SMTP2GO's endpoint sits
# behind a wildcard cert (*.smtp2go.com) so SNI is mandatory, and gen_smtp
# won't set it without explicit server_name_indication. Without this, the
# handshake fails with an "unexpected_message" TLS alert.
email_server = System.get_env("EMAIL_SERVER")
# Production Oban: live scoring, polling, and on-demand QSO enrichment.
# Runs on the Oban Pro Smart engine so we can use global_limit / rate_limit
# to protect rate-limited upstreams (Copernicus CDS in particular).
#
# PROP_ROLE controls which workloads this pod picks up:
# * "hot" (default) — full hot-path: web, crons, all queues.
# * "backfill" — no web server (unset PHX_SERVER), no crons, and only
# backfill-style queues. Intended for smaller nodes dedicated to
# catching up historical enrichment without competing with the
# hourly propagation grid chain.
prop_role = System.get_env("PROP_ROLE", "hot")
hot_only_queues = [
# 1 slot per pod. Two concurrent forecast-hour steps in a single
# pod stack HRRR grid + native duct grid + scored band map into
# ~5-6 GiB RSS, hitting the 6 GiB limit and OOM-killing the pod
# mid-chain. With 3 pods we still get 3-way parallel fan-out
# cluster-wide, which is enough to finish the f00f18 chain
# comfortably inside the hourly interval. PropagationPruneWorker
# shares this queue at a lower priority; if a prune job blocks
# the grid chain briefly, the chain's unique window protects
# the next hourly fire.
propagation: 1,
commercial: 1,
solar: 1
]
shared_queues = [
# :hrrr queue removed — per-QSO HRRR fetches flow through
# hrrr_fetch_tasks and the Rust hrrr-point-worker (Phase 3 Stream C).
terrain: 3,
# 1 slot per pod, same reasoning as :propagation. Each
# RoverPathProfileWorker job allocates a full path-compute heap
# (terrain points + 9 HRRR profiles + sounding + ionosphere +
# scoring + loss + forecast); 3 slots × 4 hot pods on the
# :terrain queue OOM-killed pods on 2026-05-03 when a backfill
# flood landed. Cluster-wide capacity is now (hot replicas + 1
# backfill) jobs at a time, which still drains a 200-job
# backlog inside ~30 min.
rover_path: 1,
iemre: 3,
backfill_enqueue: 1,
admin: 1,
radar: 2,
ionosphere: 1,
space_weather: 1,
mechanism: 4,
exports: 1,
gefs: 1
]
# Queues that are only allowed to run on the dedicated backfill pod.
# Historical backfill (weather/narr/rtma/nexrad/contact_import) spins
# against rate-limited upstreams (IEM especially) and stacks Req
# exponential-backoff sleeps under its workers; running those slots
# on the hot pods was thrashing the DB pool and tripping the /health
# readiness probe. Keeping them on prop-backfill alone lets the hot
# pods serve LiveView + cron reliably while still processing
# enrichment cluster-wide.
backfill_only_queues = [
# 1 slot — see earlier note: any higher and the ASOS backfill
# hammers IEM hard enough to get HTTP 429s, which fill the
# retryable queue and thrash pod CPU without forward progress.
weather: 1,
# Historical backfill for pre-2014 contacts (pre-HRRR archive).
# NARR is fetched anonymously from NCEI with no quota, no job
# queue. Replaced the ERA5/CDS pipeline which never produced a
# single row in prod. era5_cds_jobs table is kept for inspection
# but its workers/client are gone. See
# docs/plans/2026-04-15-merra2-historical-backfill.md.
narr: 6,
rtma: 2,
nexrad: 2,
contact_import: 4
]
# config/config.exs sets a base Oban config (queues + cron) that Config
# deep-merges with runtime.exs. To actually drop the hot-path queues on
# backfill pods we can't use `limit: 0` — Oban Pro's Smart engine
# rejects it with "local_limit must be greater than 0". Instead override
# each queue with `paused: true` (queue exists, processes nothing).
paused_queue = [local_limit: 1, paused: true]
# Pause the backfill-only queues on hot pods so the queue exists
# (base config in config/config.exs declares it) but processes
# nothing here. Same `:paused` trick as disabled_hot_queues below.
paused_backfill_queues =
Enum.map(backfill_only_queues, fn {queue, _} -> {queue, paused_queue} end)
disabled_hot_queues = [
propagation: paused_queue,
commercial: paused_queue,
solar: paused_queue,
enqueue: paused_queue
]
oban_queues =
case prop_role do
"backfill" -> disabled_hot_queues ++ shared_queues ++ backfill_only_queues
_ -> hot_only_queues ++ shared_queues ++ paused_backfill_queues
end
cron_plugin =
{Cron,
crontab: [
{"0 8 * * *", Microwaveprop.Workers.SolarIndexWorker},
{"*/5 * * * *", Microwaveprop.Commercial.PollWorker},
# Every 15 min. NOAA's HRRR pressure-level f18 publish latency
# straddles ~50-65 min after cycle hour, so an HH:05 fire
# frequently has to fall back from (now-1h) to (now-2h). A
# 4×/hour cadence lets pick_run_time/1 upgrade from fallback
# to the freshest cycle within 15 min of it becoming available
# instead of waiting for the next HH:05. Re-seeds are cheap:
# GridTaskEnqueuer.seed_with_analysis uses on_conflict: :nothing
# so duplicate fires for an already-seeded run_time are a single
# idempotent INSERT batch.
{"5,20,35,50 * * * *", Microwaveprop.Workers.PropagationGridWorker},
# HRDPS Canadian propagation chain. ECCC publishes HRDPS f000
# ~3-4h after each 00/06/12/18Z cycle. Worker seeds exactly one
# forecast row per fire (current hour). Reseeds are idempotent
# through the (run_time, forecast_hour, kind, source) unique
# index — duplicate fires for an already-seeded hour are no-ops
# at the DB level, so running 3×/hour costs nothing on the happy
# path and gives us 2 retry windows when a rolling deploy kills
# the pod mid-cron-tick. The single-replica `prop` deployment
# combined with frequent ArgoCD deploys was reliably missing the
# hourly :35 fire — multiple shots per hour fixes that.
{"5,25,45 * * * *", Microwaveprop.Workers.HrdpsGridWorker},
# GEFS runs publish ~3-4h after the 00/06/12/18Z cycle; seed the
# Day 2-7 outlook 5h after each run to stay safely past NOMADS'
# publication lag. Seeder enqueues f024-f168 at 6-hour cadence
# (25 jobs per run, 100/day) into the :gefs queue.
{"30 5,11,17,23 * * *", Microwaveprop.Workers.GefsFetchWorker},
{"*/15 * * * *", Microwaveprop.Workers.PropagationPruneWorker},
{"15 * * * *", Microwaveprop.Workers.GridCachePruneWorker},
# Re-enqueues any rover-planning Path stuck in :pending/:failed.
# Idempotent — RoverPathProfileWorker no-ops on :complete.
{"30 * * * *", Microwaveprop.Workers.RoverMissionBackfillWorker},
# The :narr type is virtual: it targets contacts with hrrr_status =
# :unavailable (pre-2014, missing from the HRRR archive) and dispatches
# NarrFetchWorker against NCEI. See narr_jobs_for_contact/1.
# limit=2000 keeps each 30-min tick under the Oban job-execution
# timeout that discards unlimited runs after ~17 min. At 4000
# contacts/hour the backlog of missing-data contacts drains
# steadily without any single job getting killed mid-way.
{"*/30 * * * *", Microwaveprop.Workers.BackfillEnqueueWorker,
args: %{
"limit" => 2000,
"types" => ["hrrr", "weather", "terrain", "iemre", "narr", "radar", "mechanism"]
}},
# Hourly safety net for pos1/pos2/distance_km. Normally every contact
# gets positions at insert time via Radio.resolve_grids_and_insert/1,
# but direct DB writes (manual fixes, bulk imports) can bypass that.
{"0 * * * *", Microwaveprop.Workers.ContactPositionBackfillWorker, args: %{"limit" => 500}},
# UWYO publishes the 00Z/12Z Canadian radiosondes ~90 minutes after launch
{"30 1 * * *", CanadianSoundingFetchWorker},
{"30 13 * * *", CanadianSoundingFetchWorker},
# GIRO DIDBase publishes foF2/foEs/hmF2 at ~7.5 min cadence; poll
# every 10 min per station (Millstone Hill, Alpena) for the 144
# MHz sporadic-E and HF MUF scoring inputs.
{"*/10 * * * *", Microwaveprop.Workers.IonosphereFetchWorker},
# NOAA SWPC: Kp / GOES X-ray at 1-min cadence, F10.7 hourly.
# 5-min poll balances freshness vs request volume.
{"*/5 * * * *", Microwaveprop.Workers.SpaceWeatherFetchWorker}
]}
base_plugins = [
{Oban.Plugins.Pruner, max_age: 3600 * 24},
# DynamicLifeline uses producer records (heartbeats from each
# live node) to rescue orphans, not a timer. When a pod dies
# mid-deploy its Producer row disappears and any in-flight job
# with that producer in `attempted_by` gets moved back to
# `available` on the next rescue cycle — within ~30s instead of
# waiting for a timer-based `rescue_after` to expire. Critical
# for PropagationGridWorker chain steps to recover from rolling
# deploys before the next hourly cron fire.
{Oban.Pro.Plugins.DynamicLifeline, rescue_interval: to_timeout(second: 30)}
]
# Backfill pods must not fire the hourly cron. Config deep-merge by
# module key means we can't just omit the Cron plugin — config.exs
# registers it and that registration survives. Override it with an
# empty crontab on backfill pods so the plugin exists but schedules
# nothing.
empty_cron_plugin = {Cron, crontab: []}
oban_plugins =
case prop_role do
"backfill" -> base_plugins ++ [empty_cron_plugin]
_ -> base_plugins ++ [cron_plugin]
end
config :libcluster,
topologies: [
k8s: [
strategy: Cluster.Strategy.Kubernetes,
config: [
mode: :ip,
kubernetes_ip_lookup_mode: :pods,
kubernetes_node_basename: "microwaveprop",
kubernetes_selector: "app=prop",
kubernetes_namespace: "prop"
]
]
]
config :microwaveprop, Microwaveprop.Mailer,
adapter: Swoosh.Adapters.SMTP,
relay: email_server,
port: 587,
username: System.get_env("EMAIL_USERNAME"),
password: System.get_env("EMAIL_PASSWORD"),
tls: :always,
tls_options: [
server_name_indication: String.to_charlist(email_server || ""),
versions: [:"tlsv1.2", :"tlsv1.3"],
verify: :verify_none
],
auth: :always
config :microwaveprop, Microwaveprop.Repo,
# ssl: true,
url: database_url,
pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"),
# For machines with several cores, consider starting multiple pools of `pool_size`
# pool_count: 4,
socket_options: maybe_ipv6
# Cross-pod cache backend (Valkey). Optional — when unset, GridCache
# falls back to the legacy per-pod ETS implementation. Set to a Redis
# URL like `redis://:password@10.0.15.21:6379/0` in prop-secrets.
case System.get_env("VALKEY_URL") do
nil -> :ok
"" -> :ok
url -> config :microwaveprop, :valkey_url, url
end
# PSK Reporter MQTT firehose ingestion. On by default in prod —
# all replicas start the client GenServer but only the
# cluster-elected leader actually connects (`:global` registration
# in `Microwaveprop.Pskr.Client`). Set `PSKR_MQTT_ENABLED=false`
# to kill the entire feature without code change.
case System.get_env("PSKR_MQTT_ENABLED") do
disabled when disabled in ["0", "false", "FALSE"] ->
config :microwaveprop, :pskr_mqtt_enabled, false
_ ->
config :microwaveprop, :pskr_mqtt_enabled, true
end
# Optional secondary repo for read-only access to aprs.me's database.
# Used by APRS-based 144 MHz calibration tooling. When the env var is
# unset OR an empty string (k8s secret resolves to "") we leave the
# AprsRepo unconfigured; Microwaveprop.Application is the single site
# that logs the skip and decides not to start it.
case System.get_env("APRS_DATABASE_URL") do
url when is_binary(url) and url != "" ->
config :microwaveprop, Microwaveprop.AprsRepo,
url: url,
pool_size: String.to_integer(System.get_env("APRS_POOL_SIZE") || "2"),
socket_options: maybe_ipv6
_ ->
:ok
end
config :microwaveprop, MicrowavepropWeb.Endpoint,
# SSL terminated by Cloudflare tunnel; generated URLs still use https
url: [host: host, port: 443, scheme: "https"],
http: [
# Enable IPv6 and bind on all interfaces.
# Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access.
# See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0
# for details about using IPv6 vs IPv4 and loopback vs public addresses.
ip: {0, 0, 0, 0, 0, 0, 0, 0}
],
secret_key_base: secret_key_base
config :microwaveprop, Oban,
engine: Oban.Pro.Engines.Smart,
queues: oban_queues
config :microwaveprop, Oban, plugins: oban_plugins
config :microwaveprop, :email_from, {"NTMS Propagation", "prop@w5isp.com"}
# Disable the PromEx Oban queue-depth poller in prod. Its 5 s
# telemetry_poller runs an aggregate scan against the oban_jobs
# table; with our current row count that query regularly holds a
# Postgrex connection for > 15 s and starves the Ecto pool enough
# that the /health probe times out and the kubelet SIGKILLs every
# replica. Queue depth is better observed from the oban_web
# dashboard anyway.
config :microwaveprop, :prom_ex_include_oban_plugin, false
config :microwaveprop, hrrr_base_url: System.get_env("HRRR_BASE_URL", "http://skippy.w5isp.com:8080")
config :microwaveprop, srtm_tiles_dir: "/data/srtm"
config :microwaveprop, start_freshness_monitor: false
end