A full f00-f18 sweep takes ~170 min of wall time, longer than the ~90 min gap between deploys on this cluster. Over the last 7 days every run was killed mid-sweep and zero runs completed. The propagation_scores table only ever held the scraps from partial runs that landed before the pod died, which is why the map looked like it was showing "the current hour" (or nothing). The worker now processes exactly one forecast hour per perform/1 (~8-10 min) and enqueues the next hour as a fresh Oban job. A cron fire with empty args seeds the chain at f00; subsequent runs carry run_time + forecast_hour args. At f18 the chain stops and the pruner runs. Each step broadcasts propagation:updated immediately so new hours appear on the map as they land. Wall time per perform drops from ~170 min to ~10 min, so Lifeline's 2h rescue window is no longer a factor and a pod restart loses at most one forecast hour instead of the whole sweep. Retries and max_attempts now describe a single fh, not the whole chain. Also: bump the :propagation queue from 1 to 2 slots so PropagationPruneWorker can run alongside the chain job, drop the worker timeout/1 from 90 min to 20 min to match single-fh runs, and pull Lifeline rescue_after from 120 min to 45 min to keep the safety net above the step timeout. Adds a "Data from HH:MM UTC · Nh ago/now/+Nh" indicator above the pipeline chip in both the mobile and desktop sidebars so the user can tell which valid_time the map is showing when the bottom timeline isn't visible.
239 lines
9.7 KiB
Elixir
239 lines
9.7 KiB
Elixir
import Config
|
||
|
||
# config/runtime.exs is executed for all environments, including
|
||
# during releases. It is executed after compilation and before the
|
||
# system starts, so it is typically used to load production configuration
|
||
# and secrets from environment variables or elsewhere. Do not define
|
||
# any compile-time configuration in here, as it won't be applied.
|
||
# The block below contains prod specific runtime configuration.
|
||
|
||
# ## Using releases
|
||
#
|
||
# If you use `mix release`, you need to explicitly enable the server
|
||
# by passing the PHX_SERVER=true when you start it:
|
||
#
|
||
# PHX_SERVER=true bin/microwaveprop start
|
||
#
|
||
# Alternatively, you can use `mix phx.gen.release` to generate a `bin/server`
|
||
# script that automatically sets the env var above.
|
||
alias Microwaveprop.Workers.CanadianSoundingFetchWorker
|
||
|
||
if System.get_env("PHX_SERVER") do
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint, server: true
|
||
end
|
||
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint, http: [port: String.to_integer(System.get_env("PORT", "4000"))]
|
||
|
||
if qrz_username = System.get_env("QRZ_USERNAME") do
|
||
config :microwaveprop, Microwaveprop.Qrz.Client,
|
||
username: qrz_username,
|
||
password: System.get_env("QRZ_PASSWORD") || raise("QRZ_PASSWORD missing"),
|
||
agent: System.get_env("QRZ_AGENT", "microwaveprop/1.0")
|
||
end
|
||
|
||
if google_api_key = System.get_env("GOOGLE_API_KEY") do
|
||
config :microwaveprop, Microwaveprop.Geocoder, api_key: google_api_key
|
||
end
|
||
|
||
if config_env() == :prod do
|
||
database_url =
|
||
System.get_env("DATABASE_URL") ||
|
||
raise """
|
||
environment variable DATABASE_URL is missing.
|
||
For example: ecto://USER:PASS@HOST/DATABASE
|
||
"""
|
||
|
||
maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: []
|
||
|
||
# The secret key base is used to sign/encrypt cookies and other secrets.
|
||
# A default value is used in config/dev.exs and config/test.exs but you
|
||
# want to use a different value for prod and you most likely don't want
|
||
# to check this value into version control, so we use an environment
|
||
# variable instead.
|
||
secret_key_base =
|
||
System.get_env("SECRET_KEY_BASE") ||
|
||
raise """
|
||
environment variable SECRET_KEY_BASE is missing.
|
||
You can generate one by calling: mix phx.gen.secret
|
||
"""
|
||
|
||
host = System.get_env("PHX_HOST") || "example.com"
|
||
|
||
# ## SSL Support
|
||
#
|
||
# To get SSL working, you will need to add the `https` key
|
||
# to your endpoint configuration:
|
||
#
|
||
# config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# https: [
|
||
# ...,
|
||
# port: 443,
|
||
# cipher_suite: :strong,
|
||
# keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"),
|
||
# certfile: System.get_env("SOME_APP_SSL_CERT_PATH")
|
||
# ]
|
||
#
|
||
# The `cipher_suite` is set to `:strong` to support only the
|
||
# latest and more secure SSL ciphers. This means old browsers
|
||
# and clients may not be supported. You can set it to
|
||
# `:compatible` for wider support.
|
||
#
|
||
# `:keyfile` and `:certfile` expect an absolute path to the key
|
||
# and cert in disk or a relative path inside priv, for example
|
||
# "priv/ssl/server.key". For all supported SSL configuration
|
||
# options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1
|
||
#
|
||
# We also recommend setting `force_ssl` in your config/prod.exs,
|
||
# ensuring no data is ever sent via http, always redirecting to https:
|
||
#
|
||
# config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# force_ssl: [hsts: true]
|
||
#
|
||
# Check `Plug.SSL` for all available options in `force_ssl`.
|
||
|
||
# ## Configuring the mailer
|
||
#
|
||
# In production you need to configure the mailer to use a different adapter.
|
||
# Here is an example configuration for Mailgun:
|
||
#
|
||
# config :microwaveprop, Microwaveprop.Mailer,
|
||
# adapter: Swoosh.Adapters.Mailgun,
|
||
# tls: :always + explicit tls_options is required — SMTP2GO's endpoint sits
|
||
# behind a wildcard cert (*.smtp2go.com) so SNI is mandatory, and gen_smtp
|
||
# won't set it without explicit server_name_indication. Without this, the
|
||
# handshake fails with an "unexpected_message" TLS alert.
|
||
email_server = System.get_env("EMAIL_SERVER")
|
||
|
||
config :libcluster,
|
||
topologies: [
|
||
k8s: [
|
||
strategy: Cluster.Strategy.Kubernetes,
|
||
config: [
|
||
mode: :ip,
|
||
kubernetes_ip_lookup_mode: :pods,
|
||
kubernetes_node_basename: "microwaveprop",
|
||
kubernetes_selector: "app=prop",
|
||
kubernetes_namespace: "prop"
|
||
]
|
||
]
|
||
]
|
||
|
||
config :microwaveprop, Microwaveprop.Mailer,
|
||
adapter: Swoosh.Adapters.SMTP,
|
||
relay: email_server,
|
||
port: 587,
|
||
username: System.get_env("EMAIL_USERNAME"),
|
||
password: System.get_env("EMAIL_PASSWORD"),
|
||
tls: :always,
|
||
tls_options: [
|
||
server_name_indication: String.to_charlist(email_server || ""),
|
||
versions: [:"tlsv1.2", :"tlsv1.3"],
|
||
verify: :verify_none
|
||
],
|
||
auth: :always
|
||
|
||
config :microwaveprop, Microwaveprop.Repo,
|
||
# ssl: true,
|
||
url: database_url,
|
||
pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"),
|
||
# For machines with several cores, consider starting multiple pools of `pool_size`
|
||
# pool_count: 4,
|
||
socket_options: maybe_ipv6
|
||
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# SSL terminated by Cloudflare tunnel; generated URLs still use https
|
||
url: [host: host, port: 443, scheme: "https"],
|
||
http: [
|
||
# Enable IPv6 and bind on all interfaces.
|
||
# Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access.
|
||
# See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0
|
||
# for details about using IPv6 vs IPv4 and loopback vs public addresses.
|
||
ip: {0, 0, 0, 0, 0, 0, 0, 0}
|
||
],
|
||
secret_key_base: secret_key_base
|
||
|
||
# Production Oban: live scoring, polling, and on-demand QSO enrichment.
|
||
# Runs on the Oban Pro Smart engine so we can use global_limit / rate_limit
|
||
# to protect rate-limited upstreams (Copernicus CDS in particular).
|
||
config :microwaveprop, Oban,
|
||
engine: Oban.Pro.Engines.Smart,
|
||
# Per-pod concurrency (×3 pods = effective cluster total)
|
||
queues: [
|
||
# 2 slots so PropagationPruneWorker can run alongside the
|
||
# forecast-hour chain job without blocking.
|
||
propagation: 2,
|
||
commercial: 1,
|
||
solar: 1,
|
||
weather: 3,
|
||
hrrr: 1,
|
||
terrain: 3,
|
||
iemre: 3,
|
||
era5: 2,
|
||
# Split-worker ERA5 pipeline:
|
||
#
|
||
# `era5_submit` runs Era5SubmitWorker, which POSTs two CDS requests
|
||
# and writes an era5_cds_jobs row. Each job is ~1s of real work, so
|
||
# local_limit is the gate on how fast we burn CDS submission quota.
|
||
# rate_limit protects CDS from submit bursts; the per-pod ceiling
|
||
# (local_limit × replicas) determines peak concurrent in-flight.
|
||
era5_submit: [
|
||
local_limit: 4,
|
||
rate_limit: [allowed: 30, period: {1, :hour}]
|
||
],
|
||
# `era5_poll` runs Era5PollWorker, which does a tiny GET per run and
|
||
# either snoozes (releasing the slot) or streams the completed GRIB
|
||
# and inserts. Needs high local_limit because in-flight CDS jobs
|
||
# scale independently of concurrent poll work — a pod may have 50
|
||
# tile-months in flight but be doing ~1 actual poll GET at a time.
|
||
era5_poll: [
|
||
local_limit: 20
|
||
],
|
||
# Legacy queue name retained briefly so any in-flight Era5MonthBatchWorker
|
||
# jobs from the pre-split deploy drain cleanly (the worker now just
|
||
# forwards to :era5_submit). Safe to delete after one rolling deploy.
|
||
era5_batch: 1,
|
||
rtma: 2,
|
||
backfill_enqueue: 1,
|
||
admin: 1,
|
||
nexrad: 2
|
||
],
|
||
plugins: [
|
||
{Oban.Plugins.Pruner, max_age: 3600 * 24},
|
||
# Lifeline rescues jobs stuck in `executing` after this window.
|
||
# MUST be larger than the longest `timeout/1` callback on any worker
|
||
# or Lifeline races the job's own deadline. The chain-style
|
||
# PropagationGridWorker caps a single forecast-hour step at 20
|
||
# min and real steps run ~8-10 min, so 45 min gives comfortable
|
||
# headroom for a slow step without letting a truly stuck job
|
||
# linger for hours before the safety net trips.
|
||
{Oban.Plugins.Lifeline, rescue_after: to_timeout(minute: 45)},
|
||
{Oban.Plugins.Cron,
|
||
crontab: [
|
||
{"0 8 * * *", Microwaveprop.Workers.SolarIndexWorker},
|
||
{"*/5 * * * *", Microwaveprop.Commercial.PollWorker},
|
||
# Every 3 hours: one full run does f00–f18 (~5 min/hour × 19 ≈ 95 min).
|
||
# Hourly cron caused overlapping/failed runs because one full sweep
|
||
# exceeds a 60-minute window.
|
||
{"5 */3 * * *", Microwaveprop.Workers.PropagationGridWorker},
|
||
# MRMS refreshes the PrecipRate cache that AsosAdjustmentWorker
|
||
# overlays onto the score grid. Both must run together — an
|
||
# ASOS nudge without a fresh MRMS frame reverts to HRRR-only
|
||
# rain. Requires wgrib2 built with PNG decoder support (DRT
|
||
# 5.41), which the Dockerfile now enables via NCEPLIBS-g2c.
|
||
{"*/2 * * * *", Microwaveprop.Workers.MrmsFetchWorker},
|
||
{"*/10 * * * *", Microwaveprop.Workers.AsosAdjustmentWorker},
|
||
{"*/15 * * * *", Microwaveprop.Workers.PropagationPruneWorker},
|
||
{"*/30 * * * *", Microwaveprop.Workers.BackfillEnqueueWorker,
|
||
args: %{"limit" => 500, "types" => ["hrrr", "weather", "terrain", "iemre"]}},
|
||
# UWYO publishes the 00Z/12Z Canadian radiosondes ~90 minutes after launch
|
||
{"30 1 * * *", CanadianSoundingFetchWorker},
|
||
{"30 13 * * *", CanadianSoundingFetchWorker}
|
||
]}
|
||
]
|
||
|
||
config :microwaveprop, :email_from, {"NTMS Propagation", "graham@w5isp.com"}
|
||
config :microwaveprop, hrrr_base_url: System.get_env("HRRR_BASE_URL", "http://skippy.w5isp.com:8080")
|
||
config :microwaveprop, srtm_tiles_dir: "/srtm"
|
||
# config :microwaveprop, start_freshness_monitor: true
|
||
config :microwaveprop, start_freshness_monitor: false
|
||
end
|