towerops/config/runtime.exs
Graham McIntire a4e08b2aca Major codebase cleanup: DRY schemas, remove dead config, consolidate modules
- Created Towerops.Schema base macro (95+ schemas updated to use it)
- Created Towerops.Snmp.Reading macro (7 reading schemas consolidated)
- Created Towerops.Gaiia.BaseSchema macro (4 Gaiia schemas consolidated)
- Created Towerops.SyncLog shared module (UISP + Preseem merge)
- Created Towerops.LogFilters (3 log filter modules merged into 1)
- Created Towerops.OnCall.ChangesetHelpers (9 on-call schemas simplified)
- Created API v1 ResourceController shared helpers (7 controllers)
- Extracted ConnectionHelpers.format_connection_result (2 LiveViews)
- Removed 5 dead config keys (scopes, mib_dirs, stripe_meter_id, etc.)
- Fixed 2 pre-existing broken tests
- All 13,219 tests pass, Credo clean (0 issues)
2026-06-16 15:29:22 -05:00

525 lines
21 KiB
Elixir

import Config
# config/runtime.exs is executed for all environments, including
# during releases. It is executed after compilation and before the
# system starts, so it is typically used to load production configuration
# and secrets from environment variables or elsewhere. Do not define
# any compile-time configuration in here, as it won't be applied.
# The block below contains prod specific runtime configuration.
# ## Using releases
#
# If you use `mix release`, you need to explicitly enable the server
# by passing the PHX_SERVER=true when you start it:
#
# PHX_SERVER=true bin/towerops start
#
# Alternatively, you can use `mix phx.gen.release` to generate a `bin/server`
# script that automatically sets the env var above.
if System.get_env("PHX_SERVER") do
config :towerops, ToweropsWeb.Endpoint, server: true
end
# DeepSeek LLM (used by Towerops.LLM.DeepSeek for insight enrichment).
# DEEPSEEK_API_KEY is optional — when missing, enrichment is skipped and
# insights still display without an LLM-generated summary.
# Helper: treat empty-string env vars as missing (the DEEPSEEK_MODEL
# secret ships as "" by default, which would otherwise bypass `||`).
deepseek_env = fn name, fallback ->
case System.get_env(name) do
nil -> fallback
"" -> fallback
value -> value
end
end
config :towerops, Towerops.LLM.DeepSeek,
api_key: System.get_env("DEEPSEEK_API_KEY"),
base_url: deepseek_env.("DEEPSEEK_BASE_URL", "https://api.deepseek.com/v1"),
model: deepseek_env.("DEEPSEEK_MODEL", "deepseek-v4-pro")
config :towerops, ToweropsWeb.Endpoint, http: [port: String.to_integer(System.get_env("PORT", "4000"))]
# Configure Google Maps API key for geocoding (available in all environments)
config :towerops, google_maps_api_key: System.get_env("GOOGLE_MAPS_API_KEY")
if config_env() == :prod do
database_url =
System.get_env("DATABASE_URL") ||
raise """
environment variable DATABASE_URL is missing.
For example: ecto://USER:PASS@HOST/DATABASE
"""
maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: []
# The secret key base is used to sign/encrypt cookies and other secrets.
# A default value is used in config/dev.exs and config/test.exs but you
# want to use a different value for prod and you most likely don't want
# to check this value into version control, so we use an environment
# variable instead.
session_signing_salt =
System.get_env("SESSION_SIGNING_SALT") ||
raise """
environment variable SESSION_SIGNING_SALT is missing.
You can generate one by calling: openssl rand -base64 16
"""
session_encryption_salt =
System.get_env("SESSION_ENCRYPTION_SALT") ||
raise """
environment variable SESSION_ENCRYPTION_SALT is missing.
You can generate one by calling: openssl rand -base64 16
"""
live_view_signing_salt =
System.get_env("LIVE_VIEW_SIGNING_SALT") ||
raise """
environment variable LIVE_VIEW_SIGNING_SALT is missing.
You can generate one by calling: openssl rand -base64 16
"""
secret_key_base =
System.get_env("SECRET_KEY_BASE") ||
raise """
environment variable SECRET_KEY_BASE is missing.
You can generate one by calling: mix phx.gen.secret
"""
# Cloak encryption key for MikroTik API passwords and other sensitive data
cloak_key =
System.get_env("CLOAK_KEY") ||
raise """
environment variable CLOAK_KEY is missing.
You can generate one by calling: openssl rand -base64 32
Store the generated key in 1Password and set as environment variable.
"""
host = System.get_env("PHX_HOST") || "example.com"
# Configure Redis/Valkey connection
# Dokku sets REDIS_URL, K8s uses individual env vars
redis_config =
case System.get_env("REDIS_URL") do
nil ->
# Use individual env vars (K8s pattern)
redis_host = System.get_env("REDIS_HOST") || "localhost"
redis_port = String.to_integer(System.get_env("REDIS_PORT") || "6379")
redis_password = System.get_env("REDIS_PASSWORD")
base_config = [
host: redis_host,
port: redis_port
]
if redis_password do
Keyword.put(base_config, :password, redis_password)
else
base_config
end
redis_url ->
# Parse REDIS_URL (Dokku pattern)
uri = URI.parse(redis_url)
base_config = [
host: uri.host || "localhost",
port: uri.port || 6379
]
# Extract password from userinfo if present
case uri.userinfo do
nil ->
base_config
userinfo ->
# userinfo can be "password" or "username:password"
password =
case String.split(userinfo, ":", parts: 2) do
[pass] -> pass
[_user, pass] -> pass
end
Keyword.put(base_config, :password, password)
end
end
# Configure SSL options based on environment variables
ssl_config =
case System.get_env("DATABASE_SSL") do
"true" ->
case System.get_env("DATABASE_SSL_VERIFY") do
"false" ->
[verify: :verify_none]
_ ->
[verify: :verify_peer]
end
_ ->
false
end
# Only enable libcluster in Kubernetes deployments
# Kubernetes automatically sets KUBERNETES_SERVICE_HOST env var
libcluster_topologies =
if System.get_env("KUBERNETES_SERVICE_HOST") do
[
k8s: [
strategy: Cluster.Strategy.Kubernetes.DNS,
config: [
service: "towerops-headless",
application_name: "towerops",
namespace: "towerops",
polling_interval: 10_000
]
]
]
else
# No clustering for non-K8s deployments (Dokku, single-node, etc.)
[]
end
# OBAN_QUEUE_SCALE divides all queue sizes (e.g. "5" = 1/5th capacity for staging)
oban_scale = String.to_integer(System.get_env("OBAN_QUEUE_SCALE") || "1")
# Aggressive TCP keepalive for Kubernetes environments.
# Default OS keepalive is ~2 hours on Linux, far too slow for k8s where
# load balancers and proxies silently drop idle connections within minutes.
# Linux raw socket options (SOL_TCP=6):
# TCP_KEEPIDLE (4) = 5s - seconds idle before first keepalive probe
# TCP_KEEPINTVL (5) = 3s - seconds between subsequent probes
# TCP_KEEPCNT (6) = 3 - failed probes before connection is dead
# Total dead-connection detection: 5 + (3 * 3) = 14 seconds
tcp_keepalive_opts =
if System.get_env("KUBERNETES_SERVICE_HOST") do
[
{:keepalive, true},
{:raw, 6, 4, <<5::native-32>>},
{:raw, 6, 5, <<3::native-32>>},
{:raw, 6, 6, <<3::native-32>>}
]
else
[keepalive: true]
end
config :libcluster, topologies: libcluster_topologies
# Configure Swoosh to use Req for API requests
config :swoosh, :api_client, Swoosh.ApiClient.Req
# Configure Oban for distributed job processing
config :towerops, Oban,
engine: Oban.Pro.Engines.Smart,
repo: Towerops.Repo,
shutdown_grace_period: to_timeout(second: 40),
queues: [
default: max(1, div(10, oban_scale)),
discovery: max(1, div(10, oban_scale)),
# SNMP polling jobs - one per device
pollers: max(1, div(String.to_integer(System.get_env("POLLER_CONCURRENCY") || "50"), oban_scale)),
# Device monitoring jobs - health checks
monitors: max(1, div(50, oban_scale)),
# Service checks - HTTP/TCP/DNS
checks: max(1, div(50, oban_scale)),
check_executors: max(1, div(50, oban_scale)),
# Alert notifications (PagerDuty, email, etc.)
notifications: max(1, div(25, oban_scale)),
maintenance: max(1, div(5, oban_scale)),
weather: max(1, div(2, oban_scale)),
lidar: max(1, div(2, oban_scale)),
# RF coverage prediction (CPU + GDAL/HTTP-range I/O heavy, slow)
coverage: max(1, div(2, oban_scale))
],
plugins: [
# Cron jobs for periodic maintenance tasks
{Oban.Plugins.Cron,
crontab: [
# Run neighbor cleanup every hour
{"0 * * * *", Towerops.Snmp.NeighborCleanupWorker},
# Check for stale agents every minute
{"* * * * *", Towerops.Workers.StaleAgentWorker},
# Cloud latency probes every 8 hours (00:00, 08:00, 16:00)
{"0 */8 * * *", Towerops.Workers.CloudLatencyProbeWorker},
# Evaluate latency-based agent reassignment 30 min after each probe batch
{"30 0,8,16 * * *", Towerops.Workers.AgentLatencyEvaluator},
# Health check for missing jobs every 10 minutes
{"*/10 * * * *", Towerops.Workers.JobHealthCheckWorker},
# Send notification digests for rate-limited alerts every 5 minutes
{"*/5 * * * *", Towerops.Workers.AlertDigestWorker, args: %{"user_id" => "__cron__"}},
# Mark stale backup requests as timeout every 10 minutes
{"*/10 * * * *", Towerops.Workers.BackupTimeoutWorker},
# Clean up old login history daily at 2 AM
{"0 2 * * *", Towerops.Workers.LoginHistoryCleanupWorker},
# Clean up expired browser sessions daily at 3 AM
{"0 3 * * *", Towerops.Workers.SessionCleanupWorker},
# Fetch latest firmware versions daily at 4 AM
{"0 4 * * *", Towerops.Workers.FirmwareVersionFetcherWorker},
# MikroTik configuration backups daily at 7 AM UTC
{"0 7 * * *", Towerops.Workers.MikrotikBackupWorker},
# Send backup summary email daily at 8 AM
{"0 8 * * *", Towerops.Workers.BackupSummaryWorker},
# Delete expired IP bans every hour
{"0 * * * *", Towerops.Workers.ExpiredBanCleanupWorker},
# Delete stale violation records daily at 2 AM
{"0 2 * * *", Towerops.Workers.StaleViolationCleanupWorker},
# Sync Preseem data every 10 minutes
{"*/5 * * * *", Towerops.Workers.PreseemSyncWorker},
# Compute Preseem baselines and fleet profiles nightly at 2:30 AM
{"30 2 * * *", Towerops.Workers.PreseemBaselineWorker},
# Sync Gaiia data every 15 minutes
{"*/5 * * * *", Towerops.Workers.GaiiaSyncWorker},
# Sync NetBox data every 30 minutes
{"*/5 * * * *", Towerops.Workers.NetBoxSyncWorker},
# Sync Sonar billing data every 5 minutes
{"*/5 * * * *", Towerops.Workers.SonarSyncWorker},
# Sync Splynx billing data every 5 minutes
{"*/5 * * * *", Towerops.Workers.SplynxSyncWorker},
# Sync VISP billing data every 5 minutes
{"*/5 * * * *", Towerops.Workers.VispSyncWorker},
# Sync UISP infrastructure data every 5 minutes
{"*/5 * * * *", Towerops.Workers.UispSyncWorker},
# Sync Cambium cnMaestro data every 5 minutes
{"*/5 * * * *", Towerops.Workers.CnMaestroSyncWorker},
# Scheduled reports — check for due reports every hour
{"0 * * * *", Towerops.Workers.ReportWorker},
# Device health insights nightly at 3:30 AM
{"30 3 * * *", Towerops.Workers.DeviceHealthInsightWorker},
# System insights (agent offline detection) every 5 minutes
{"*/5 * * * *", Towerops.Workers.SystemInsightWorker},
# Wireless client health insights every 5 minutes
{"*/5 * * * *", Towerops.Workers.WirelessInsightWorker},
# Run multi-source recommendation rules nightly at 02:00 UTC.
# Fresh insights are ready before the LLM enrichment pass at 03:00.
{"0 2 * * *", Towerops.Workers.RecommendationsRunWorker},
# Enrich active insights with LLM-generated summaries nightly at
# 03:00 UTC. Batched once a day instead of every 5 minutes — the
# API spend isn't justified by minute-by-minute freshness when
# ops review the dashboard in the morning.
{"0 3 * * *", Towerops.Workers.InsightLlmEnrichmentWorker},
# AI network-wide insight pass. Sends a per-org snapshot of
# Preseem/SNMP/Gaiia/system state to the LLM and asks it to
# surface novel patterns the rule-based workers miss. Runs at
# 03:30 UTC after the rule pass and per-insight enrichment so
# the existing_insights field is fresh.
{"30 3 * * *", Towerops.Workers.AiNetworkInsightWorker},
# Auto-resolve insights older than 7 days. Nightly at 04:00 UTC,
# after the rule pass + LLM enrichment have finished. Still-real
# issues will be re-created on the next rule run.
{"0 4 * * *", Towerops.Workers.InsightExpiryWorker},
# Gaiia reconciliation insights nightly at 4:30 AM
{"30 4 * * *", Towerops.Workers.GaiiaInsightWorker},
# Backhaul capacity utilization insights every 15 minutes
{"*/15 * * * *", Towerops.Workers.CapacityInsightWorker},
# Sync device usage to Stripe daily at 3 AM UTC
{"0 3 * * *", Towerops.Workers.BillingSyncWorker},
# Data retention cleanup nightly at 1 AM
{"0 1 * * *", Towerops.Workers.DataRetentionWorker},
# Refresh LIDAR tile catalog from public 3DEP/TNRIS sources monthly
{"0 6 1 * *", Towerops.Workers.LidarCatalogSyncWorker}
]},
# Automatically delete completed jobs after 60 seconds.
# Default limit of 10k/30s (333/s) can't keep up with job throughput (~600+/s),
# causing oban_jobs to bloat to millions of rows and slow all queries.
# 50k/5s (10,000/s) keeps the table small.
{Oban.Plugins.Pruner, max_age: 60, limit: 50_000, interval: to_timeout(second: 5)},
# Rescue orphaned jobs (when node crashes) and prevent index bloat
# High DELETE churn (10k jobs/sec pruned) causes severe index bloat on:
# - Primary key (82.2 bloat, 3.2GB waste)
# - Scheduling index (216.8 bloat, 7.9GB waste)
# REINDEX CONCURRENTLY reclaims space without blocking operations.
{Oban.Plugins.Reindexer,
schedule: "0 2 * * *",
indexes: [
"oban_jobs_pkey",
"oban_jobs_state_queue_priority_scheduled_at_id_index",
"oban_jobs_args_index",
"oban_jobs_meta_index"
]},
# Rescue jobs orphaned in the executing state. DynamicLifeline (not core
# Lifeline) is required with the Pro Smart engine + unique workers: it
# catches the unique_violation that occurs when a rescued orphan collides
# with an already-scheduled successor (same uniq_key) and repairs it,
# whereas core Lifeline crashes on that 23505. It rescues by producer
# records rather than a time threshold, so rescue_after does not apply.
Oban.Pro.Plugins.DynamicLifeline
]
# ## SSL Support
#
# To get SSL working, you will need to add the `https` key
# to your endpoint configuration:
#
# config :towerops, ToweropsWeb.Endpoint,
# https: [
# ...,
# port: 443,
# cipher_suite: :strong,
# keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"),
# certfile: System.get_env("SOME_APP_SSL_CERT_PATH")
# ]
#
# The `cipher_suite` is set to `:strong` to support only the
# latest and more secure SSL ciphers. This means old browsers
# and clients may not be supported. You can set it to
# `:compatible` for wider support.
#
# `:keyfile` and `:certfile` expect an absolute path to the key
# and cert in disk or a relative path inside priv, for example
# "priv/ssl/server.key". For all supported SSL configuration
# options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1
#
# We also recommend setting `force_ssl` in your config/prod.exs,
# ensuring no data is ever sent via http, always redirecting to https:
#
# config :towerops, ToweropsWeb.Endpoint,
# force_ssl: [hsts: true]
#
# Check `Plug.SSL` for all available options in `force_ssl`.
# ## Configuring the mailer
#
# In production you need to configure the mailer to use a different adapter.
# Here is an example configuration for Mailgun:
#
# config :towerops, Towerops.Mailer,
# adapter: Swoosh.Adapters.Mailgun,
# api_key: System.get_env("MAILGUN_API_KEY"),
# domain: System.get_env("MAILGUN_DOMAIN")
#
# Most non-SMTP adapters require an API client. Swoosh supports Req, Hackney,
# and Finch out-of-the-box. This configuration is typically done at
# compile-time in your config/prod.exs:
#
# config :swoosh, :api_client, Swoosh.ApiClient.Req
#
# See https://hexdocs.pm/swoosh/Swoosh.html#module-installation for details.
# Configure Amazon SES for production email
config :towerops, Towerops.Mailer,
adapter: Swoosh.Adapters.AmazonSES,
region: System.get_env("AWS_REGION") || "us-east-1",
access_key: System.get_env("AWS_ACCESS_KEY_ID"),
secret: System.get_env("AWS_SECRET_ACCESS_KEY")
config :towerops, Towerops.Repo,
ssl: ssl_config,
url: database_url,
pool_size: String.to_integer(System.get_env("POOL_SIZE") || "10"),
# Proactively ping idle connections every 1s to detect dead SSL connections
# before they're handed to a real query. Without this, a dead connection
# sits in the pool until a query checks it out and hangs for the full timeout.
idle_interval: 1_000,
# Detect stale connections faster: if a query takes longer than queue_target (2s)
# on average over queue_interval (5s), DBConnection will disconnect idle connections
# to cycle in fresh ones. Helps recover from "ssl recv: closed" errors.
queue_target: 2_000,
queue_interval: 5_000,
# PostgreSQL server-side statement timeout: kill any query running longer than 30s.
# Prevents queries from hanging indefinitely on dead SSL connections.
parameters: [application_name: "towerops", statement_timeout: "30000"],
# Disconnect immediately on fatal PostgreSQL error codes for faster pool recovery:
# 57P01 = admin_shutdown, 57P02 = crash_shutdown, 57P03 = cannot_connect_now
disconnect_on_error_codes: [:"57P01", :"57P02", :"57P03"],
socket_options: maybe_ipv6 ++ tcp_keepalive_opts
config :towerops, Towerops.Vault,
ciphers: [
default: {
Cloak.Ciphers.AES.GCM,
tag: "AES.GCM.V1", key: Base.decode64!(cloak_key)
}
]
config :towerops, ToweropsWeb.Endpoint,
url: [host: host, port: 443, scheme: "https"],
check_origin: [
"//towerops.net",
"//*.towerops.net"
],
live_view: [signing_salt: live_view_signing_salt],
http: [
# Enable IPv6 and bind on all interfaces.
# Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access.
# See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0
# for details about using IPv6 vs IPv4 and loopback vs public addresses.
ip: {0, 0, 0, 0, 0, 0, 0, 0},
# Bandit uses ThousandIsland for connection management
# Timeout for reading client data before closing the connection.
# Set high enough to accommodate long-lived WebSocket connections
# (agents send Phoenix transport heartbeats every 25s).
thousand_island_options: [
read_timeout: 120_000
]
],
secret_key_base: secret_key_base
# Webhook secret for CI-triggered mass agent updates
config :towerops, :agent_webhook_secret, System.get_env("AGENT_WEBHOOK_SECRET")
config :towerops, :dns_cluster_query, System.get_env("DNS_CLUSTER_QUERY")
# Set environment identifier for runtime checks
config :towerops, :env, :prod
# Set default sender for all emails
config :towerops, :mailer_from, {"Towerops", "hi@towerops.net"}
config :towerops, :redis, redis_config
config :towerops, :session_encryption_salt, session_encryption_salt
config :towerops, :session_signing_salt, session_signing_salt
# Cloudflare API credentials for brute force protection
config :towerops,
cloudflare_zone_id: System.get_env("CLOUDFLARE_ZONE_ID"),
cloudflare_api_token: System.get_env("CLOUDFLARE_API_TOKEN")
# Disable automatic timezone database updates in production
# Timezone data is updated through dependency updates instead
# This prevents crashes when network timeouts occur during automatic updates
config :tzdata, :autoupdate, :disabled
# OpenWeatherMap API key for site weather data
if owm_key = System.get_env("OPENWEATHERMAP_API_KEY") do
config :towerops, :openweathermap_api_key, owm_key
end
# Stripe configuration for billing
stripe_secret_key =
System.get_env("STRIPE_SECRET_KEY") ||
raise """
environment variable STRIPE_SECRET_KEY is missing.
Get this from Stripe dashboard and store in K8s secrets.
"""
stripe_webhook_secret =
System.get_env("STRIPE_WEBHOOK_SECRET") ||
raise """
environment variable STRIPE_WEBHOOK_SECRET is missing.
Get this from Stripe webhook settings and store in K8s secrets.
"""
stripe_price_id =
System.get_env("STRIPE_PRICE_ID") ||
raise """
environment variable STRIPE_PRICE_ID is missing.
This is the Stripe price ID for the metered billing plan.
"""
# Configure ErrorTrackerNotifier for email alerts on exceptions
config :error_tracker_notifier,
notification_type: :email,
from_email: "alerts@towerops.net",
to_email: System.get_env("ERROR_ALERT_EMAIL") || "graham@towerops.net",
mailer: Towerops.Mailer
config :towerops,
stripe_secret_key: stripe_secret_key,
stripe_webhook_secret: stripe_webhook_secret,
stripe_price_id: stripe_price_id
end
# Test environment Stripe configuration
if config_env() == :test do
config :towerops,
stripe_secret_key: "sk_test_fake",
stripe_webhook_secret: "whsec_test_fake",
stripe_price_id: "price_test_fake"
end