Prod has 0 era5_profiles rows after ~500 submit cycles: every CDS job either gets marked "rejected" (likely cap mismatch) or runs for 13+h without transitioning to successful, so the handle_both_done branch has literally never fired and the ERA5Batch: stored log line has never been emitted once. Tighten Era5PollWorker's @stuck_after_seconds from 18h to 8h so dead rows recycle within a work shift instead of a full day, and pause the era5 / era5_submit / era5_poll / era5_batch queues in runtime.exs so existing rows stay put for inspection while we figure out why CDS is rejecting everything. Flip paused: true back once the root cause is identified.
248 lines
10 KiB
Elixir
248 lines
10 KiB
Elixir
import Config
|
||
|
||
# config/runtime.exs is executed for all environments, including
|
||
# during releases. It is executed after compilation and before the
|
||
# system starts, so it is typically used to load production configuration
|
||
# and secrets from environment variables or elsewhere. Do not define
|
||
# any compile-time configuration in here, as it won't be applied.
|
||
# The block below contains prod specific runtime configuration.
|
||
|
||
# ## Using releases
|
||
#
|
||
# If you use `mix release`, you need to explicitly enable the server
|
||
# by passing the PHX_SERVER=true when you start it:
|
||
#
|
||
# PHX_SERVER=true bin/microwaveprop start
|
||
#
|
||
# Alternatively, you can use `mix phx.gen.release` to generate a `bin/server`
|
||
# script that automatically sets the env var above.
|
||
alias Microwaveprop.Workers.CanadianSoundingFetchWorker
|
||
|
||
if System.get_env("PHX_SERVER") do
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint, server: true
|
||
end
|
||
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint, http: [port: String.to_integer(System.get_env("PORT", "4000"))]
|
||
|
||
if qrz_username = System.get_env("QRZ_USERNAME") do
|
||
config :microwaveprop, Microwaveprop.Qrz.Client,
|
||
username: qrz_username,
|
||
password: System.get_env("QRZ_PASSWORD") || raise("QRZ_PASSWORD missing"),
|
||
agent: System.get_env("QRZ_AGENT", "microwaveprop/1.0")
|
||
end
|
||
|
||
if google_api_key = System.get_env("GOOGLE_API_KEY") do
|
||
config :microwaveprop, Microwaveprop.Geocoder, api_key: google_api_key
|
||
end
|
||
|
||
if config_env() == :prod do
|
||
database_url =
|
||
System.get_env("DATABASE_URL") ||
|
||
raise """
|
||
environment variable DATABASE_URL is missing.
|
||
For example: ecto://USER:PASS@HOST/DATABASE
|
||
"""
|
||
|
||
maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: []
|
||
|
||
# The secret key base is used to sign/encrypt cookies and other secrets.
|
||
# A default value is used in config/dev.exs and config/test.exs but you
|
||
# want to use a different value for prod and you most likely don't want
|
||
# to check this value into version control, so we use an environment
|
||
# variable instead.
|
||
secret_key_base =
|
||
System.get_env("SECRET_KEY_BASE") ||
|
||
raise """
|
||
environment variable SECRET_KEY_BASE is missing.
|
||
You can generate one by calling: mix phx.gen.secret
|
||
"""
|
||
|
||
host = System.get_env("PHX_HOST") || "example.com"
|
||
|
||
# ## SSL Support
|
||
#
|
||
# To get SSL working, you will need to add the `https` key
|
||
# to your endpoint configuration:
|
||
#
|
||
# config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# https: [
|
||
# ...,
|
||
# port: 443,
|
||
# cipher_suite: :strong,
|
||
# keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"),
|
||
# certfile: System.get_env("SOME_APP_SSL_CERT_PATH")
|
||
# ]
|
||
#
|
||
# The `cipher_suite` is set to `:strong` to support only the
|
||
# latest and more secure SSL ciphers. This means old browsers
|
||
# and clients may not be supported. You can set it to
|
||
# `:compatible` for wider support.
|
||
#
|
||
# `:keyfile` and `:certfile` expect an absolute path to the key
|
||
# and cert in disk or a relative path inside priv, for example
|
||
# "priv/ssl/server.key". For all supported SSL configuration
|
||
# options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1
|
||
#
|
||
# We also recommend setting `force_ssl` in your config/prod.exs,
|
||
# ensuring no data is ever sent via http, always redirecting to https:
|
||
#
|
||
# config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# force_ssl: [hsts: true]
|
||
#
|
||
# Check `Plug.SSL` for all available options in `force_ssl`.
|
||
|
||
# ## Configuring the mailer
|
||
#
|
||
# In production you need to configure the mailer to use a different adapter.
|
||
# Here is an example configuration for Mailgun:
|
||
#
|
||
# config :microwaveprop, Microwaveprop.Mailer,
|
||
# adapter: Swoosh.Adapters.Mailgun,
|
||
# tls: :always + explicit tls_options is required — SMTP2GO's endpoint sits
|
||
# behind a wildcard cert (*.smtp2go.com) so SNI is mandatory, and gen_smtp
|
||
# won't set it without explicit server_name_indication. Without this, the
|
||
# handshake fails with an "unexpected_message" TLS alert.
|
||
email_server = System.get_env("EMAIL_SERVER")
|
||
|
||
config :libcluster,
|
||
topologies: [
|
||
k8s: [
|
||
strategy: Cluster.Strategy.Kubernetes,
|
||
config: [
|
||
mode: :ip,
|
||
kubernetes_ip_lookup_mode: :pods,
|
||
kubernetes_node_basename: "microwaveprop",
|
||
kubernetes_selector: "app=prop",
|
||
kubernetes_namespace: "prop"
|
||
]
|
||
]
|
||
]
|
||
|
||
config :microwaveprop, Microwaveprop.Mailer,
|
||
adapter: Swoosh.Adapters.SMTP,
|
||
relay: email_server,
|
||
port: 587,
|
||
username: System.get_env("EMAIL_USERNAME"),
|
||
password: System.get_env("EMAIL_PASSWORD"),
|
||
tls: :always,
|
||
tls_options: [
|
||
server_name_indication: String.to_charlist(email_server || ""),
|
||
versions: [:"tlsv1.2", :"tlsv1.3"],
|
||
verify: :verify_none
|
||
],
|
||
auth: :always
|
||
|
||
config :microwaveprop, Microwaveprop.Repo,
|
||
# ssl: true,
|
||
url: database_url,
|
||
pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"),
|
||
# For machines with several cores, consider starting multiple pools of `pool_size`
|
||
# pool_count: 4,
|
||
socket_options: maybe_ipv6
|
||
|
||
config :microwaveprop, MicrowavepropWeb.Endpoint,
|
||
# SSL terminated by Cloudflare tunnel; generated URLs still use https
|
||
url: [host: host, port: 443, scheme: "https"],
|
||
http: [
|
||
# Enable IPv6 and bind on all interfaces.
|
||
# Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access.
|
||
# See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0
|
||
# for details about using IPv6 vs IPv4 and loopback vs public addresses.
|
||
ip: {0, 0, 0, 0, 0, 0, 0, 0}
|
||
],
|
||
secret_key_base: secret_key_base
|
||
|
||
# Production Oban: live scoring, polling, and on-demand QSO enrichment.
|
||
# Runs on the Oban Pro Smart engine so we can use global_limit / rate_limit
|
||
# to protect rate-limited upstreams (Copernicus CDS in particular).
|
||
config :microwaveprop, Oban,
|
||
engine: Oban.Pro.Engines.Smart,
|
||
# Per-pod concurrency (×3 pods = effective cluster total)
|
||
queues: [
|
||
# 2 slots so PropagationPruneWorker can run alongside the
|
||
# forecast-hour chain job without blocking.
|
||
propagation: 2,
|
||
commercial: 1,
|
||
solar: 1,
|
||
weather: 3,
|
||
hrrr: 1,
|
||
terrain: 3,
|
||
iemre: 3,
|
||
# ERA5 queues are TEMPORARILY PAUSED while we diagnose the
|
||
# zero-ingestion pipeline. Every poll in prod has been hitting
|
||
# either CDS-rejected or long-running states; no `ERA5Batch:
|
||
# stored` log line has ever fired. Pausing keeps the existing
|
||
# oban_jobs and era5_cds_jobs rows in place for inspection
|
||
# instead of deleting them. Flip `paused: true` to `paused:
|
||
# false` (or remove the key) once the root cause is fixed.
|
||
era5: [local_limit: 2, paused: true],
|
||
era5_submit: [
|
||
local_limit: 4,
|
||
rate_limit: [allowed: 30, period: {1, :hour}],
|
||
paused: true
|
||
],
|
||
era5_poll: [local_limit: 20, paused: true],
|
||
era5_batch: [local_limit: 1, paused: true],
|
||
rtma: 2,
|
||
backfill_enqueue: 1,
|
||
admin: 1,
|
||
nexrad: 2,
|
||
ionosphere: 1,
|
||
space_weather: 1
|
||
],
|
||
plugins: [
|
||
{Oban.Plugins.Pruner, max_age: 3600 * 24},
|
||
# DynamicLifeline uses producer records (heartbeats from each
|
||
# live node) to rescue orphans, not a timer. When a pod dies
|
||
# mid-deploy its Producer row disappears and any in-flight job
|
||
# with that producer in `attempted_by` gets moved back to
|
||
# `available` on the next rescue cycle — within ~30s instead of
|
||
# waiting for a timer-based `rescue_after` to expire. Critical
|
||
# for PropagationGridWorker chain steps to recover from rolling
|
||
# deploys before the next hourly cron fire.
|
||
{Oban.Pro.Plugins.DynamicLifeline, rescue_interval: to_timeout(second: 30)},
|
||
{Oban.Plugins.Cron,
|
||
crontab: [
|
||
{"0 8 * * *", Microwaveprop.Workers.SolarIndexWorker},
|
||
{"*/5 * * * *", Microwaveprop.Commercial.PollWorker},
|
||
# Hourly at :05. With scores written as binary files (not
|
||
# Postgres) and HRRR grid profile persistence removed, a
|
||
# full f00-f18 chain runs in ~45-60 min, so hourly fits.
|
||
# If one chain slips past 60 min the :propagation queue's
|
||
# concurrency-of-2 lets the next chain interleave; files
|
||
# are keyed by (band, valid_time) so last-writer-wins gives
|
||
# newer analysis data naturally.
|
||
{"5 * * * *", Microwaveprop.Workers.PropagationGridWorker},
|
||
# MRMS refreshes the PrecipRate cache that AsosAdjustmentWorker
|
||
# overlays onto the score grid. Both must run together — an
|
||
# ASOS nudge without a fresh MRMS frame reverts to HRRR-only
|
||
# rain. Requires wgrib2 built with PNG decoder support (DRT
|
||
# 5.41), which the Dockerfile now enables via NCEPLIBS-g2c.
|
||
{"*/2 * * * *", Microwaveprop.Workers.MrmsFetchWorker},
|
||
# AsosAdjustmentWorker disabled: depends on hrrr_profiles rows
|
||
# that PropagationGridWorker no longer persists. The 10-min
|
||
# ASOS nudge is a nice-to-have refinement we're not paying the
|
||
# full grid-side JSONB write for.
|
||
# {"*/10 * * * *", Microwaveprop.Workers.AsosAdjustmentWorker},
|
||
{"*/15 * * * *", Microwaveprop.Workers.PropagationPruneWorker},
|
||
{"*/30 * * * *", Microwaveprop.Workers.BackfillEnqueueWorker,
|
||
args: %{"limit" => 500, "types" => ["hrrr", "weather", "terrain", "iemre"]}},
|
||
# UWYO publishes the 00Z/12Z Canadian radiosondes ~90 minutes after launch
|
||
{"30 1 * * *", CanadianSoundingFetchWorker},
|
||
{"30 13 * * *", CanadianSoundingFetchWorker},
|
||
# GIRO DIDBase publishes foF2/foEs/hmF2 at ~7.5 min cadence; poll
|
||
# every 10 min per station (Millstone Hill, Alpena) for the 144
|
||
# MHz sporadic-E and HF MUF scoring inputs.
|
||
{"*/10 * * * *", Microwaveprop.Workers.IonosphereFetchWorker},
|
||
# NOAA SWPC: Kp / GOES X-ray at 1-min cadence, F10.7 hourly.
|
||
# 5-min poll balances freshness vs request volume.
|
||
{"*/5 * * * *", Microwaveprop.Workers.SpaceWeatherFetchWorker}
|
||
]}
|
||
]
|
||
|
||
config :microwaveprop, :email_from, {"NTMS Propagation", "prop@w5isp.com"}
|
||
config :microwaveprop, hrrr_base_url: System.get_env("HRRR_BASE_URL", "http://skippy.w5isp.com:8080")
|
||
config :microwaveprop, srtm_tiles_dir: "/data/srtm"
|
||
# config :microwaveprop, start_freshness_monitor: true
|
||
config :microwaveprop, start_freshness_monitor: false
|
||
end
|