prop/config/runtime.exs
Graham McIntire 316fb2fbc7
Fix low-severity bugs and re-enable Credo checks
- Bug #12: Lower rate limit to 15/min
- Bug #13: Wrap model loading in Task.start
- Bug #14: Fix classify_time_period guard gap at -3.0
- Bug #15: Not applicable (Elixir has no ?? operator)
- Bug #16: Remove fallback Repo.get for station preload
- Bug #17: Extract cache_key() in ContactMapController
- Bug #18: Add has_many :contacts and :beacons to User schema
- A&D #4: Move serve_markdown_if_requested after secure headers
- Config #1: Move signing_salt to runtime.exs env var
- Config #2: Use --check-unused instead of --unused
- Config #3: Re-enable UnsafeToAtom Credo check
- Config #5: Re-enable LeakyEnvironment Credo check
- Add @spec annotations to fix re-enabled Specs violations
- Replace String.to_atom with to_existing_atom where guarded
2026-05-29 17:29:22 -05:00

494 lines
22 KiB
Elixir
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import Config
alias Microwaveprop.Workers.CanadianSoundingFetchWorker
alias Oban.Plugins.Cron
require Logger
# config/runtime.exs is executed for all environments, including
# during releases. It is executed after compilation and before the
# system starts, so it is typically used to load production configuration
# and secrets from environment variables or elsewhere. Do not define
# any compile-time configuration in here, as it won't be applied.
# The block below contains prod specific runtime configuration.
# ## Using releases
#
# If you use `mix release`, you need to explicitly enable the server
# by passing the PHX_SERVER=true when you start it:
#
# PHX_SERVER=true bin/microwaveprop start
#
# Alternatively, you can use `mix phx.gen.release` to generate a `bin/server`
# script that automatically sets the env var above.
if System.get_env("PHX_SERVER") do
config :microwaveprop, MicrowavepropWeb.Endpoint, server: true
end
# Log format: "text" (default) keeps Elixir's built-in line formatter for
# local/dev use; "json" swaps the :default :logger handler to emit a single
# JSON object per line via `LoggerJSON.Formatters.Basic`. The ingress +
# Grafana pipeline is already JSON-aware, so prod defaults to JSON.
log_format =
System.get_env("LOG_FORMAT") ||
if(config_env() == :prod, do: "json", else: "text")
if log_format == "json" do
formatter =
LoggerJSON.Formatters.Basic.new(metadata: [:request_id, :trace_id, :worker, :queue, :job_id])
:logger.update_handler_config(:default, :formatter, formatter)
end
config :microwaveprop, MicrowavepropWeb.Endpoint,
http: [port: String.to_integer(System.get_env("PORT", "4000"))],
live_view: [signing_salt: System.get_env("LIVE_VIEW_SIGNING_SALT", "Gdq36xze")]
# MicrowavepropWeb.Plugs.RemoteIp will only honour `Cf-Connecting-Ip` /
# `X-Forwarded-For` when the immediate TCP peer (conn.remote_ip) sits inside
# one of these CIDRs. Set TRUSTED_PROXY_CIDRS to a comma-separated list to
# override; leave unset to keep the module default (loopback + RFC1918 +
# ULA + link-local), which matches the k8s cluster pod/service network.
if trusted_proxy_cidrs = System.get_env("TRUSTED_PROXY_CIDRS") do
cidrs =
trusted_proxy_cidrs
|> String.split(",", trim: true)
|> Enum.map(&String.trim/1)
|> Enum.reject(&(&1 == ""))
config :microwaveprop, :trusted_proxies, cidrs
end
if prometheus_token = System.get_env("PROMETHEUS_AUTH_TOKEN") do
config :microwaveprop, :prometheus_auth_token, prometheus_token
end
if qrz_username = System.get_env("QRZ_USERNAME") do
config :microwaveprop, Microwaveprop.Qrz.Client,
username: qrz_username,
password: System.get_env("QRZ_PASSWORD") || raise("QRZ_PASSWORD missing"),
agent: System.get_env("QRZ_AGENT", "microwaveprop/1.0")
end
if google_api_key = System.get_env("GOOGLE_API_KEY") do
config :microwaveprop, Microwaveprop.Geocoder, api_key: google_api_key
end
if config_env() == :prod do
database_url =
System.get_env("DATABASE_URL") ||
raise """
environment variable DATABASE_URL is missing.
For example: ecto://USER:PASS@HOST/DATABASE
"""
maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: []
# The secret key base is used to sign/encrypt cookies and other secrets.
# A default value is used in config/dev.exs and config/test.exs but you
# want to use a different value for prod and you most likely don't want
# to check this value into version control, so we use an environment
# variable instead.
secret_key_base =
System.get_env("SECRET_KEY_BASE") ||
raise """
environment variable SECRET_KEY_BASE is missing.
You can generate one by calling: mix phx.gen.secret
"""
host = System.get_env("PHX_HOST") || "example.com"
# ## SSL Support
#
# To get SSL working, you will need to add the `https` key
# to your endpoint configuration:
#
# config :microwaveprop, MicrowavepropWeb.Endpoint,
# https: [
# ...,
# port: 443,
# cipher_suite: :strong,
# keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"),
# certfile: System.get_env("SOME_APP_SSL_CERT_PATH")
# ]
#
# The `cipher_suite` is set to `:strong` to support only the
# latest and more secure SSL ciphers. This means old browsers
# and clients may not be supported. You can set it to
# `:compatible` for wider support.
#
# `:keyfile` and `:certfile` expect an absolute path to the key
# and cert in disk or a relative path inside priv, for example
# "priv/ssl/server.key". For all supported SSL configuration
# options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1
#
# We also recommend setting `force_ssl` in your config/prod.exs,
# ensuring no data is ever sent via http, always redirecting to https:
#
# config :microwaveprop, MicrowavepropWeb.Endpoint,
# force_ssl: [hsts: true]
#
# Check `Plug.SSL` for all available options in `force_ssl`.
# ## Configuring the mailer
#
# In production you need to configure the mailer to use a different adapter.
# Here is an example configuration for Mailgun:
#
# config :microwaveprop, Microwaveprop.Mailer,
# adapter: Swoosh.Adapters.Mailgun,
# tls: :always + explicit tls_options is required — SMTP2GO's endpoint sits
# behind a wildcard cert (*.smtp2go.com) so SNI is mandatory, and gen_smtp
# won't set it without explicit server_name_indication. Without this, the
# handshake fails with an "unexpected_message" TLS alert.
email_server =
System.get_env("EMAIL_SERVER") || raise("EMAIL_SERVER must be set (SMTP relay hostname)")
# Production Oban: live scoring, polling, and on-demand QSO enrichment.
# Runs on the Oban Pro Smart engine so we can use global_limit / rate_limit
# to protect rate-limited upstreams (Copernicus CDS in particular).
#
# PROP_ROLE controls which workloads this pod picks up:
# * "hot" (default) — full hot-path: web, crons, all queues.
# * "backfill" — no web server (unset PHX_SERVER), no crons, and only
# backfill-style queues. Intended for smaller nodes dedicated to
# catching up historical enrichment without competing with the
# hourly propagation grid chain.
prop_role = System.get_env("PROP_ROLE", "hot")
hot_only_queues = [
# 1 slot per pod. Two concurrent forecast-hour steps in a single
# pod stack HRRR grid + native duct grid + scored band map into
# ~5-6 GiB RSS, hitting the 6 GiB limit and OOM-killing the pod
# mid-chain. With 3 pods we still get 3-way parallel fan-out
# cluster-wide, which is enough to finish the f00f18 chain
# comfortably inside the hourly interval. PropagationPruneWorker
# shares this queue at a lower priority; if a prune job blocks
# the grid chain briefly, the chain's unique window protects
# the next hourly fire.
propagation: 1,
commercial: 1,
solar: 1
]
shared_queues = [
# :hrrr queue removed — per-QSO HRRR fetches flow through
# hrrr_fetch_tasks and the Rust hrrr-point-worker (Phase 3 Stream C).
terrain: 3,
# 1 slot per pod, same reasoning as :propagation. Each
# RoverPathProfileWorker job allocates a full path-compute heap
# (terrain points + 9 HRRR profiles + sounding + ionosphere +
# scoring + loss + forecast); 3 slots × 4 hot pods on the
# :terrain queue OOM-killed pods on 2026-05-03 when a backfill
# flood landed. Cluster-wide capacity is now (hot replicas + 1
# backfill) jobs at a time, which still drains a 200-job
# backlog inside ~30 min.
rover_path: 1,
iemre: 3,
backfill_enqueue: 1,
admin: 1,
radar: 2,
ionosphere: 1,
space_weather: 1,
mechanism: 4,
exports: 1,
gefs: 1
]
# Queues that are only allowed to run on the dedicated backfill pod.
# Historical backfill (weather/narr/nexrad/contact_import) spins
# against rate-limited upstreams (IEM especially) and stacks Req
# exponential-backoff sleeps under its workers; running those slots
# on the hot pods was thrashing the DB pool and tripping the /health
# readiness probe. Keeping them on prop-backfill alone lets the hot
# pods serve LiveView + cron reliably while still processing
# enrichment cluster-wide.
backfill_only_queues = [
# 1 slot — see earlier note: any higher and the ASOS backfill
# hammers IEM hard enough to get HTTP 429s, which fill the
# retryable queue and thrash pod CPU without forward progress.
weather: 1,
# Historical backfill for pre-2014 contacts (pre-HRRR archive).
# NARR is fetched anonymously from NCEI with no quota, no job
# queue. Replaced the ERA5/CDS pipeline which never produced a
# single row in prod. era5_cds_jobs table is kept for inspection
# but its workers/client are gone. See
# docs/plans/2026-04-15-merra2-historical-backfill.md.
narr: 6,
nexrad: 2,
contact_import: 4
]
# config/config.exs sets a base Oban config (queues + cron) that Config
# deep-merges with runtime.exs. To actually drop the hot-path queues on
# backfill pods we can't use `limit: 0` — Oban Pro's Smart engine
# rejects it with "local_limit must be greater than 0". Instead override
# each queue with `paused: true` (queue exists, processes nothing).
paused_queue = [local_limit: 1, paused: true]
# Pause the backfill-only queues on hot pods so the queue exists
# (base config in config/config.exs declares it) but processes
# nothing here. Same `:paused` trick as disabled_hot_queues below.
paused_backfill_queues =
Enum.map(backfill_only_queues, fn {queue, _} -> {queue, paused_queue} end)
disabled_hot_queues = [
propagation: paused_queue,
commercial: paused_queue,
solar: paused_queue,
enqueue: paused_queue
]
oban_queues =
case prop_role do
"backfill" -> disabled_hot_queues ++ shared_queues ++ backfill_only_queues
_ -> hot_only_queues ++ shared_queues ++ paused_backfill_queues
end
cron_plugin =
{Cron,
crontab: [
{"0 8 * * *", Microwaveprop.Workers.SolarIndexWorker},
{"*/5 * * * *", Microwaveprop.Commercial.PollWorker},
# Every 15 min. NOAA's HRRR pressure-level f18 publish latency
# straddles ~50-65 min after cycle hour, so an HH:05 fire
# frequently has to fall back from (now-1h) to (now-2h). A
# 4×/hour cadence lets pick_run_time/1 upgrade from fallback
# to the freshest cycle within 15 min of it becoming available
# instead of waiting for the next HH:05. Re-seeds are cheap:
# GridTaskEnqueuer.seed_with_analysis uses on_conflict: :nothing
# so duplicate fires for an already-seeded run_time are a single
# idempotent INSERT batch.
{"5,20,35,50 * * * *", Microwaveprop.Workers.PropagationGridWorker},
# HRDPS Canadian propagation chain. ECCC publishes HRDPS f000
# ~3-4h after each 00/06/12/18Z cycle. Worker seeds exactly one
# forecast row per fire (current hour). Reseeds are idempotent
# through the (run_time, forecast_hour, kind, source) unique
# index — duplicate fires for an already-seeded hour are no-ops
# at the DB level, so running 3×/hour costs nothing on the happy
# path and gives us 2 retry windows when a rolling deploy kills
# the pod mid-cron-tick. The single-replica `prop` deployment
# combined with frequent ArgoCD deploys was reliably missing the
# hourly :35 fire — multiple shots per hour fixes that.
{"5,25,45 * * * *", Microwaveprop.Workers.HrdpsGridWorker},
# GEFS runs publish ~3-4h after the 00/06/12/18Z cycle; seed the
# Day 2-7 outlook 5h after each run to stay safely past NOMADS'
# publication lag. Seeder enqueues f024-f168 at 6-hour cadence
# (25 jobs per run, 100/day) into the :gefs queue.
{"30 5,11,17,23 * * *", Microwaveprop.Workers.GefsFetchWorker},
{"*/15 * * * *", Microwaveprop.Workers.PropagationPruneWorker},
{"15 * * * *", Microwaveprop.Workers.GridCachePruneWorker},
# Re-enqueues any rover-planning Path stuck in :pending/:failed.
# Idempotent — RoverPathProfileWorker no-ops on :complete.
{"30 * * * *", Microwaveprop.Workers.RoverMissionBackfillWorker},
# The :narr type is virtual: it targets contacts with hrrr_status =
# :unavailable (pre-2014, missing from the HRRR archive) and dispatches
# NarrFetchWorker against NCEI. See narr_jobs_for_contact/1.
# limit=2000 keeps each 30-min tick under the Oban job-execution
# timeout that discards unlimited runs after ~17 min. At 4000
# contacts/hour the backlog of missing-data contacts drains
# steadily without any single job getting killed mid-way.
{"*/30 * * * *", Microwaveprop.Workers.BackfillEnqueueWorker,
args: %{
"limit" => 2000,
"types" => ["hrrr", "weather", "terrain", "iemre", "narr", "radar", "mechanism"]
}},
# Hourly safety net for pos1/pos2/distance_km. Normally every contact
# gets positions at insert time via Radio.resolve_grids_and_insert/1,
# but direct DB writes (manual fixes, bulk imports) can bypass that.
{"0 * * * *", Microwaveprop.Workers.ContactPositionBackfillWorker, args: %{"limit" => 500}},
# UWYO publishes the 00Z/12Z Canadian radiosondes ~90 minutes after launch
{"30 1 * * *", CanadianSoundingFetchWorker},
{"30 13 * * *", CanadianSoundingFetchWorker},
# GIRO DIDBase publishes foF2/foEs/hmF2 at ~7.5 min cadence; poll
# every 10 min per station (Millstone Hill, Alpena) for the 144
# MHz sporadic-E and HF MUF scoring inputs.
{"*/10 * * * *", Microwaveprop.Workers.IonosphereFetchWorker},
# NOAA SWPC: Kp / GOES X-ray at 1-min cadence, F10.7 hourly.
# 5-min poll balances freshness vs request volume.
{"*/5 * * * *", Microwaveprop.Workers.SpaceWeatherFetchWorker},
# Recompute hrrr_climatology nightly from the hrrr_profiles
# archive. Idempotent — UPSERT on (lat, lon, month, hour) — and
# cheap enough to run daily so the table self-heals if a
# deploy clears it or new profiles need to fold into the means.
# 02:30 UTC sits inside the quiet pre-NARR/HRRR ingestion gap.
{"30 2 * * *", Microwaveprop.Workers.AdminTaskWorker, args: %{"task" => "climatology"}},
# Build the PSKR calibration corpus *early and often*: every
# 5 min, process the current hour + the prior hour. Each pass
# spreads the work across the hour as spots accumulate so we
# never bind a giant batch at the boundary, and the prior-
# hour pass guarantees the boundary is covered without
# waiting for the next fire. UPSERT is idempotent so the
# ~12 redundant passes per hour cost only the read+merge.
# Longer outages can be backfilled with `lookback_hours`.
{"*/5 * * * *", Microwaveprop.Workers.PskrCalibrationWorker},
# Weekly recalibration analysis pass. The recalibrator
# checks corpus density and bins each scoring feature ×
# band when the corpus is large enough; otherwise records
# a `skipped_insufficient_data` run. Sundays 04:00 UTC sits
# off-peak and after the nightly climatology rebuild.
# Auto-applies nothing — weight changes still go through
# human review of `BandConfig`.
{"0 4 * * 0", Microwaveprop.Workers.PskrRecalibrationWorker},
# Daily partition extension for the time-partitioned weather
# tables (`hrrr_profiles`, `hrdps_profiles`). Idempotent — uses
# CREATE TABLE IF NOT EXISTS so a no-op fire is cheap. Runs at
# 03:15 UTC after the climatology rebuild and well clear of the
# top-of-hour propagation grid pulse.
{"15 3 * * *", Microwaveprop.Workers.PartitionMaintenanceWorker}
]}
base_plugins = [
{Oban.Plugins.Pruner, max_age: 3600 * 24},
# DynamicLifeline uses producer records (heartbeats from each
# live node) to rescue orphans, not a timer. When a pod dies
# mid-deploy its Producer row disappears and any in-flight job
# with that producer in `attempted_by` gets moved back to
# `available` on the next rescue cycle — within ~30s instead of
# waiting for a timer-based `rescue_after` to expire. Critical
# for PropagationGridWorker chain steps to recover from rolling
# deploys before the next hourly cron fire.
{Oban.Pro.Plugins.DynamicLifeline, rescue_interval: to_timeout(second: 30)}
]
# Backfill pods must not fire the hourly cron. Config deep-merge by
# module key means we can't just omit the Cron plugin — config.exs
# registers it and that registration survives. Override it with an
# empty crontab on backfill pods so the plugin exists but schedules
# nothing.
empty_cron_plugin = {Cron, crontab: []}
oban_plugins =
case prop_role do
"backfill" -> base_plugins ++ [empty_cron_plugin]
_ -> base_plugins ++ [cron_plugin]
end
config :libcluster,
topologies: [
k8s: [
strategy: Cluster.Strategy.Kubernetes,
config: [
mode: :ip,
kubernetes_ip_lookup_mode: :pods,
kubernetes_node_basename: "microwaveprop",
kubernetes_selector: "app=prop",
kubernetes_namespace: "prop"
]
]
]
config :microwaveprop, Microwaveprop.Mailer,
adapter: Swoosh.Adapters.SMTP,
relay: email_server,
port: 587,
username: System.get_env("EMAIL_USERNAME"),
password: System.get_env("EMAIL_PASSWORD"),
tls: :always,
tls_options: [
server_name_indication: String.to_charlist(email_server || ""),
versions: [:"tlsv1.2", :"tlsv1.3"],
verify: :verify_none
],
auth: :always
config :microwaveprop, Microwaveprop.Repo,
# ssl: true,
url: database_url,
pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"),
# For machines with several cores, consider starting multiple pools of `pool_size`
# pool_count: 4,
socket_options: maybe_ipv6
# Cross-pod cache backend (Valkey). Optional — when unset, GridCache
# falls back to the legacy per-pod ETS implementation. Set to a Redis
# URL like `redis://:password@10.0.15.21:6379/0` in prop-secrets.
case System.get_env("VALKEY_URL") do
nil -> :ok
"" -> :ok
url -> config :microwaveprop, :valkey_url, url
end
# PSK Reporter MQTT firehose ingestion. On by default in prod —
# all replicas start the client GenServer but only the
# cluster-elected leader actually connects (`:global` registration
# in `Microwaveprop.Pskr.Client`). Set `PSKR_MQTT_ENABLED=false`
# to kill the entire feature without code change.
#
# Backfill pods opt out unconditionally — they share the libcluster
# `app=prop` selector with the hot pods, so without this gate the
# backfill pod can win :global election (FCFS during a rollout)
# and pin the MQTT client onto a node that's also under heavy
# batch-job load. MQTT is real-time; backfill is bursty CPU; the
# two don't mix on the same pod.
pskr_default = prop_role != "backfill"
case System.get_env("PSKR_MQTT_ENABLED") do
disabled when disabled in ["0", "false", "FALSE"] ->
config :microwaveprop, :pskr_mqtt_enabled, false
enabled when enabled in ["1", "true", "TRUE"] ->
config :microwaveprop, :pskr_mqtt_enabled, true
_ ->
config :microwaveprop, :pskr_mqtt_enabled, pskr_default
end
# Optional secondary repo for read-only access to aprs.me's database.
# Used by APRS-based 144 MHz calibration tooling. When the env var is
# unset OR an empty string (k8s secret resolves to "") we leave the
# AprsRepo unconfigured; Microwaveprop.Application is the single site
# that logs the skip and decides not to start it.
case System.get_env("APRS_DATABASE_URL") do
url when is_binary(url) and url != "" ->
config :microwaveprop, Microwaveprop.AprsRepo,
url: url,
pool_size: String.to_integer(System.get_env("APRS_POOL_SIZE") || "2"),
socket_options: maybe_ipv6
_ ->
:ok
end
config :microwaveprop, MicrowavepropWeb.Endpoint,
# SSL terminated by Cloudflare tunnel; generated URLs still use https
url: [host: host, port: 443, scheme: "https"],
http: [
# Enable IPv6 and bind on all interfaces.
# Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access.
# See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0
# for details about using IPv6 vs IPv4 and loopback vs public addresses.
ip: {0, 0, 0, 0, 0, 0, 0, 0}
],
secret_key_base: secret_key_base
config :microwaveprop, Oban,
engine: Oban.Pro.Engines.Smart,
queues: oban_queues
config :microwaveprop, Oban, plugins: oban_plugins
# Oban's leader peer runs all leader-only plugins (Cron, Pruner's
# rescue, etc.) on whichever node wins election. All `app=prop` pods
# share the libcluster selector, so without intervention a backfill
# pod can win — and its empty crontab means *no* crons fire until
# the next leader rotation. Disabling peer participation on backfill
# pods makes only hot pods eligible, which is what we actually want.
if prop_role == "backfill" do
config :microwaveprop, Oban, peer: false
end
config :microwaveprop, :email_from, {"NTMS Propagation", "prop@w5isp.com"}
# Disable the PromEx Oban queue-depth poller in prod. Its 5 s
# telemetry_poller runs an aggregate scan against the oban_jobs
# table; with our current row count that query regularly holds a
# Postgrex connection for > 15 s and starves the Ecto pool enough
# that the /health probe times out and the kubelet SIGKILLs every
# replica. Queue depth is better observed from the oban_web
# dashboard anyway.
config :microwaveprop, :prom_ex_include_oban_plugin, false
config :microwaveprop, hrrr_base_url: System.get_env("HRRR_BASE_URL", "http://skippy.w5isp.com:8080")
config :microwaveprop, srtm_tiles_dir: "/data/srtm"
config :microwaveprop, start_freshness_monitor: false
end