PROP_ROLE=backfill drops the hot-path propagation/commercial/solar queues and the Cron plugin. Paired with a dedicated Deployment (nodeSelector: prop-backfill=true, 4Gi memory, no web) this lets 8GB T620 workers drain the hrrr/weather/terrain/iemre/narr backfill queues without competing with the hourly grid chain.
305 lines
13 KiB
Elixir
305 lines
13 KiB
Elixir
import Config
|
||
|
||
# config/runtime.exs is executed for all environments, including
|
||
# during releases. It is executed after compilation and before the
|
||
# system starts, so it is typically used to load production configuration
|
||
# and secrets from environment variables or elsewhere. Do not define
|
||
# any compile-time configuration in here, as it won't be applied.
|
||
# The block below contains prod specific runtime configuration.
|
||
|
||
# ## Using releases
|
||
#
|
||
# If you use `mix release`, you need to explicitly enable the server
|
||
# by passing the PHX_SERVER=true when you start it:
|
||
#
|
||
# PHX_SERVER=true bin/microwaveprop start
|
||
#
|
||
# Alternatively, you can use `mix phx.gen.release` to generate a `bin/server`
|
||
# script that automatically sets the env var above.
|
||
alias Microwaveprop.Workers.CanadianSoundingFetchWorker
|
||
|
||
if System.get_env("PHX_SERVER") do
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint, server: true
|
||
end
|
||
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint, http: [port: String.to_integer(System.get_env("PORT", "4000"))]
|
||
|
||
if prometheus_token = System.get_env("PROMETHEUS_AUTH_TOKEN") do
|
||
config :microwaveprop, :prometheus_auth_token, prometheus_token
|
||
end
|
||
|
||
if qrz_username = System.get_env("QRZ_USERNAME") do
|
||
config :microwaveprop, Microwaveprop.Qrz.Client,
|
||
username: qrz_username,
|
||
password: System.get_env("QRZ_PASSWORD") || raise("QRZ_PASSWORD missing"),
|
||
agent: System.get_env("QRZ_AGENT", "microwaveprop/1.0")
|
||
end
|
||
|
||
if google_api_key = System.get_env("GOOGLE_API_KEY") do
|
||
config :microwaveprop, Microwaveprop.Geocoder, api_key: google_api_key
|
||
end
|
||
|
||
if config_env() == :prod do
|
||
database_url =
|
||
System.get_env("DATABASE_URL") ||
|
||
raise """
|
||
environment variable DATABASE_URL is missing.
|
||
For example: ecto://USER:PASS@HOST/DATABASE
|
||
"""
|
||
|
||
maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: []
|
||
|
||
# The secret key base is used to sign/encrypt cookies and other secrets.
|
||
# A default value is used in config/dev.exs and config/test.exs but you
|
||
# want to use a different value for prod and you most likely don't want
|
||
# to check this value into version control, so we use an environment
|
||
# variable instead.
|
||
secret_key_base =
|
||
System.get_env("SECRET_KEY_BASE") ||
|
||
raise """
|
||
environment variable SECRET_KEY_BASE is missing.
|
||
You can generate one by calling: mix phx.gen.secret
|
||
"""
|
||
|
||
host = System.get_env("PHX_HOST") || "example.com"
|
||
|
||
# ## SSL Support
|
||
#
|
||
# To get SSL working, you will need to add the `https` key
|
||
# to your endpoint configuration:
|
||
#
|
||
# config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# https: [
|
||
# ...,
|
||
# port: 443,
|
||
# cipher_suite: :strong,
|
||
# keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"),
|
||
# certfile: System.get_env("SOME_APP_SSL_CERT_PATH")
|
||
# ]
|
||
#
|
||
# The `cipher_suite` is set to `:strong` to support only the
|
||
# latest and more secure SSL ciphers. This means old browsers
|
||
# and clients may not be supported. You can set it to
|
||
# `:compatible` for wider support.
|
||
#
|
||
# `:keyfile` and `:certfile` expect an absolute path to the key
|
||
# and cert in disk or a relative path inside priv, for example
|
||
# "priv/ssl/server.key". For all supported SSL configuration
|
||
# options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1
|
||
#
|
||
# We also recommend setting `force_ssl` in your config/prod.exs,
|
||
# ensuring no data is ever sent via http, always redirecting to https:
|
||
#
|
||
# config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# force_ssl: [hsts: true]
|
||
#
|
||
# Check `Plug.SSL` for all available options in `force_ssl`.
|
||
|
||
# ## Configuring the mailer
|
||
#
|
||
# In production you need to configure the mailer to use a different adapter.
|
||
# Here is an example configuration for Mailgun:
|
||
#
|
||
# config :microwaveprop, Microwaveprop.Mailer,
|
||
# adapter: Swoosh.Adapters.Mailgun,
|
||
# tls: :always + explicit tls_options is required — SMTP2GO's endpoint sits
|
||
# behind a wildcard cert (*.smtp2go.com) so SNI is mandatory, and gen_smtp
|
||
# won't set it without explicit server_name_indication. Without this, the
|
||
# handshake fails with an "unexpected_message" TLS alert.
|
||
email_server = System.get_env("EMAIL_SERVER")
|
||
|
||
# Production Oban: live scoring, polling, and on-demand QSO enrichment.
|
||
# Runs on the Oban Pro Smart engine so we can use global_limit / rate_limit
|
||
# to protect rate-limited upstreams (Copernicus CDS in particular).
|
||
#
|
||
# PROP_ROLE controls which workloads this pod picks up:
|
||
# * "hot" (default) — full hot-path: web, crons, all queues.
|
||
# * "backfill" — no web server (unset PHX_SERVER), no crons, and only
|
||
# backfill-style queues. Intended for smaller nodes dedicated to
|
||
# catching up historical enrichment without competing with the
|
||
# hourly propagation grid chain.
|
||
prop_role = System.get_env("PROP_ROLE", "hot")
|
||
|
||
hot_only_queues = [
|
||
# 1 slot per pod. Two concurrent forecast-hour steps in a single
|
||
# pod stack HRRR grid + native duct grid + scored band map into
|
||
# ~5-6 GiB RSS, hitting the 6 GiB limit and OOM-killing the pod
|
||
# mid-chain. With 3 pods we still get 3-way parallel fan-out
|
||
# cluster-wide, which is enough to finish the f00–f18 chain
|
||
# comfortably inside the hourly interval. PropagationPruneWorker
|
||
# shares this queue at a lower priority; if a prune job blocks
|
||
# the grid chain briefly, the chain's unique window protects
|
||
# the next hourly fire.
|
||
propagation: 1,
|
||
commercial: 1,
|
||
solar: 1
|
||
]
|
||
|
||
shared_queues = [
|
||
# 1 slot per pod (3 cluster-wide). Any higher than this and the
|
||
# ASOS backfill hammers IEM hard enough to get HTTP 429s, which
|
||
# fill the retryable queue and thrash pod CPU without making
|
||
# forward progress.
|
||
weather: 1,
|
||
gefs: 1,
|
||
hrrr: 2,
|
||
terrain: 3,
|
||
iemre: 3,
|
||
# Historical backfill for pre-2014 contacts (pre-HRRR archive).
|
||
# NARR is fetched anonymously from NCEI with no quota, no job
|
||
# queue. Replaced the ERA5/CDS pipeline which never produced a
|
||
# single row in prod. era5_cds_jobs table is kept for inspection
|
||
# but its workers/client are gone. See
|
||
# docs/plans/2026-04-15-merra2-historical-backfill.md.
|
||
narr: 6,
|
||
rtma: 2,
|
||
backfill_enqueue: 1,
|
||
admin: 1,
|
||
nexrad: 2,
|
||
radar: 2,
|
||
ionosphere: 1,
|
||
space_weather: 1,
|
||
mechanism: 4,
|
||
exports: 1,
|
||
contact_import: 4
|
||
]
|
||
|
||
oban_queues =
|
||
case prop_role do
|
||
"backfill" -> shared_queues
|
||
_ -> hot_only_queues ++ shared_queues
|
||
end
|
||
|
||
cron_plugin =
|
||
{Oban.Plugins.Cron,
|
||
crontab: [
|
||
{"0 8 * * *", Microwaveprop.Workers.SolarIndexWorker},
|
||
{"*/5 * * * *", Microwaveprop.Commercial.PollWorker},
|
||
# Hourly at :05. With scores written as binary files (not
|
||
# Postgres) and HRRR grid profile persistence removed, a
|
||
# full f00-f18 chain runs in ~45-60 min, so hourly fits.
|
||
# If one chain slips past 60 min the :propagation queue's
|
||
# concurrency-of-2 lets the next chain interleave; files
|
||
# are keyed by (band, valid_time) so last-writer-wins gives
|
||
# newer analysis data naturally.
|
||
{"5 * * * *", Microwaveprop.Workers.PropagationGridWorker},
|
||
# GEFS runs publish ~3-4h after the 00/06/12/18Z cycle; seed the
|
||
# Day 2-7 outlook 5h after each run to stay safely past NOMADS'
|
||
# publication lag. Seeder enqueues f024-f168 at 6-hour cadence
|
||
# (25 jobs per run, 100/day) into the :gefs queue.
|
||
{"30 5,11,17,23 * * *", Microwaveprop.Workers.GefsFetchWorker},
|
||
# MRMS refreshes the PrecipRate cache that AsosAdjustmentWorker
|
||
# overlays onto the score grid. Both must run together — an
|
||
# ASOS nudge without a fresh MRMS frame reverts to HRRR-only
|
||
# rain. Requires wgrib2 built with PNG decoder support (DRT
|
||
# 5.41), which the Dockerfile now enables via NCEPLIBS-g2c.
|
||
{"*/2 * * * *", Microwaveprop.Workers.MrmsFetchWorker},
|
||
# AsosAdjustmentWorker disabled: depends on hrrr_profiles rows
|
||
# that PropagationGridWorker no longer persists. The 10-min
|
||
# ASOS nudge is a nice-to-have refinement we're not paying the
|
||
# full grid-side JSONB write for.
|
||
# {"*/10 * * * *", Microwaveprop.Workers.AsosAdjustmentWorker},
|
||
{"*/15 * * * *", Microwaveprop.Workers.PropagationPruneWorker},
|
||
# The :narr type is virtual: it targets contacts with hrrr_status =
|
||
# :unavailable (pre-2014, missing from the HRRR archive) and dispatches
|
||
# NarrFetchWorker against NCEI. See narr_jobs_for_contact/1.
|
||
{"*/30 * * * *", Microwaveprop.Workers.BackfillEnqueueWorker,
|
||
args: %{"types" => ["hrrr", "weather", "terrain", "iemre", "narr", "radar", "mechanism"]}},
|
||
# Hourly safety net for pos1/pos2/distance_km. Normally every contact
|
||
# gets positions at insert time via Radio.resolve_grids_and_insert/1,
|
||
# but direct DB writes (manual fixes, bulk imports) can bypass that.
|
||
{"0 * * * *", Microwaveprop.Workers.ContactPositionBackfillWorker, args: %{"limit" => 500}},
|
||
# UWYO publishes the 00Z/12Z Canadian radiosondes ~90 minutes after launch
|
||
{"30 1 * * *", CanadianSoundingFetchWorker},
|
||
{"30 13 * * *", CanadianSoundingFetchWorker},
|
||
# GIRO DIDBase publishes foF2/foEs/hmF2 at ~7.5 min cadence; poll
|
||
# every 10 min per station (Millstone Hill, Alpena) for the 144
|
||
# MHz sporadic-E and HF MUF scoring inputs.
|
||
{"*/10 * * * *", Microwaveprop.Workers.IonosphereFetchWorker},
|
||
# NOAA SWPC: Kp / GOES X-ray at 1-min cadence, F10.7 hourly.
|
||
# 5-min poll balances freshness vs request volume.
|
||
{"*/5 * * * *", Microwaveprop.Workers.SpaceWeatherFetchWorker}
|
||
]}
|
||
|
||
base_plugins = [
|
||
{Oban.Plugins.Pruner, max_age: 3600 * 24},
|
||
# DynamicLifeline uses producer records (heartbeats from each
|
||
# live node) to rescue orphans, not a timer. When a pod dies
|
||
# mid-deploy its Producer row disappears and any in-flight job
|
||
# with that producer in `attempted_by` gets moved back to
|
||
# `available` on the next rescue cycle — within ~30s instead of
|
||
# waiting for a timer-based `rescue_after` to expire. Critical
|
||
# for PropagationGridWorker chain steps to recover from rolling
|
||
# deploys before the next hourly cron fire.
|
||
{Oban.Pro.Plugins.DynamicLifeline, rescue_interval: to_timeout(second: 30)}
|
||
]
|
||
|
||
# Backfill pods skip the cron plugin entirely — no risk of a
|
||
# backfill node becoming cron leader and enqueueing work targeted
|
||
# at queues it doesn't even run (propagation, commercial). Hot pods
|
||
# own scheduling; backfill pods just drain the queues.
|
||
oban_plugins =
|
||
case prop_role do
|
||
"backfill" -> base_plugins
|
||
_ -> base_plugins ++ [cron_plugin]
|
||
end
|
||
|
||
config :libcluster,
|
||
topologies: [
|
||
k8s: [
|
||
strategy: Cluster.Strategy.Kubernetes,
|
||
config: [
|
||
mode: :ip,
|
||
kubernetes_ip_lookup_mode: :pods,
|
||
kubernetes_node_basename: "microwaveprop",
|
||
kubernetes_selector: "app=prop",
|
||
kubernetes_namespace: "prop"
|
||
]
|
||
]
|
||
]
|
||
|
||
config :microwaveprop, Microwaveprop.Mailer,
|
||
adapter: Swoosh.Adapters.SMTP,
|
||
relay: email_server,
|
||
port: 587,
|
||
username: System.get_env("EMAIL_USERNAME"),
|
||
password: System.get_env("EMAIL_PASSWORD"),
|
||
tls: :always,
|
||
tls_options: [
|
||
server_name_indication: String.to_charlist(email_server || ""),
|
||
versions: [:"tlsv1.2", :"tlsv1.3"],
|
||
verify: :verify_none
|
||
],
|
||
auth: :always
|
||
|
||
config :microwaveprop, Microwaveprop.Repo,
|
||
# ssl: true,
|
||
url: database_url,
|
||
pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"),
|
||
# For machines with several cores, consider starting multiple pools of `pool_size`
|
||
# pool_count: 4,
|
||
socket_options: maybe_ipv6
|
||
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# SSL terminated by Cloudflare tunnel; generated URLs still use https
|
||
url: [host: host, port: 443, scheme: "https"],
|
||
http: [
|
||
# Enable IPv6 and bind on all interfaces.
|
||
# Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access.
|
||
# See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0
|
||
# for details about using IPv6 vs IPv4 and loopback vs public addresses.
|
||
ip: {0, 0, 0, 0, 0, 0, 0, 0}
|
||
],
|
||
secret_key_base: secret_key_base
|
||
|
||
config :microwaveprop, Oban,
|
||
engine: Oban.Pro.Engines.Smart,
|
||
queues: oban_queues
|
||
|
||
config :microwaveprop, Oban, plugins: oban_plugins
|
||
config :microwaveprop, :email_from, {"NTMS Propagation", "prop@w5isp.com"}
|
||
config :microwaveprop, hrrr_base_url: System.get_env("HRRR_BASE_URL", "http://skippy.w5isp.com:8080")
|
||
config :microwaveprop, srtm_tiles_dir: "/data/srtm"
|
||
# config :microwaveprop, start_freshness_monitor: true
|
||
config :microwaveprop, start_freshness_monitor: false
|
||
end
|