import Config # config/runtime.exs is executed for all environments, including # during releases. It is executed after compilation and before the # system starts, so it is typically used to load production configuration # and secrets from environment variables or elsewhere. Do not define # any compile-time configuration in here, as it won't be applied. # The block below contains prod specific runtime configuration. # ## Using releases # # If you use `mix release`, you need to explicitly enable the server # by passing the PHX_SERVER=true when you start it: # # PHX_SERVER=true bin/towerops start # # Alternatively, you can use `mix phx.gen.release` to generate a `bin/server` # script that automatically sets the env var above. if System.get_env("PHX_SERVER") do config :towerops, ToweropsWeb.Endpoint, server: true end config :towerops, ToweropsWeb.Endpoint, http: [port: String.to_integer(System.get_env("PORT", "4000"))] # Configure Google Maps API key for geocoding (available in all environments) config :towerops, google_maps_api_key: System.get_env("GOOGLE_MAPS_API_KEY") if config_env() == :prod do database_url = System.get_env("DATABASE_URL") || raise """ environment variable DATABASE_URL is missing. For example: ecto://USER:PASS@HOST/DATABASE """ maybe_ipv6 = if System.get_env("ECTO_IPV6") in ~w(true 1), do: [:inet6], else: [] # The secret key base is used to sign/encrypt cookies and other secrets. # A default value is used in config/dev.exs and config/test.exs but you # want to use a different value for prod and you most likely don't want # to check this value into version control, so we use an environment # variable instead. secret_key_base = System.get_env("SECRET_KEY_BASE") || raise """ environment variable SECRET_KEY_BASE is missing. You can generate one by calling: mix phx.gen.secret """ # Cloak encryption key for MikroTik API passwords and other sensitive data cloak_key = System.get_env("CLOAK_KEY") || raise """ environment variable CLOAK_KEY is missing. You can generate one by calling: openssl rand -base64 32 Store the generated key in 1Password and set as environment variable. """ host = System.get_env("PHX_HOST") || "example.com" # Configure Redis/Valkey connection # Dokku sets REDIS_URL, K8s uses individual env vars redis_config = case System.get_env("REDIS_URL") do nil -> # Use individual env vars (K8s pattern) redis_host = System.get_env("REDIS_HOST") || "localhost" redis_port = String.to_integer(System.get_env("REDIS_PORT") || "6379") redis_password = System.get_env("REDIS_PASSWORD") base_config = [ host: redis_host, port: redis_port ] if redis_password do Keyword.put(base_config, :password, redis_password) else base_config end redis_url -> # Parse REDIS_URL (Dokku pattern) uri = URI.parse(redis_url) base_config = [ host: uri.host || "localhost", port: uri.port || 6379 ] # Extract password from userinfo if present case uri.userinfo do nil -> base_config userinfo -> # userinfo can be "password" or "username:password" password = case String.split(userinfo, ":", parts: 2) do [pass] -> pass [_user, pass] -> pass end Keyword.put(base_config, :password, password) end end # Configure SSL options based on environment variables ssl_config = case System.get_env("DATABASE_SSL") do "true" -> case System.get_env("DATABASE_SSL_VERIFY") do "false" -> [verify: :verify_none] _ -> [verify: :verify_peer] end _ -> false end # Only enable libcluster in Kubernetes deployments # Kubernetes automatically sets KUBERNETES_SERVICE_HOST env var libcluster_topologies = if System.get_env("KUBERNETES_SERVICE_HOST") do [ k8s: [ strategy: Cluster.Strategy.Kubernetes.DNS, config: [ service: "towerops-headless", application_name: "towerops", namespace: "towerops", polling_interval: 10_000 ] ] ] else # No clustering for non-K8s deployments (Dokku, single-node, etc.) [] end # OBAN_QUEUE_SCALE divides all queue sizes (e.g. "5" = 1/5th capacity for staging) oban_scale = String.to_integer(System.get_env("OBAN_QUEUE_SCALE") || "1") # Aggressive TCP keepalive for Kubernetes environments. # Default OS keepalive is ~2 hours on Linux, far too slow for k8s where # load balancers and proxies silently drop idle connections within minutes. # Linux raw socket options (SOL_TCP=6): # TCP_KEEPIDLE (4) = 5s - seconds idle before first keepalive probe # TCP_KEEPINTVL (5) = 3s - seconds between subsequent probes # TCP_KEEPCNT (6) = 3 - failed probes before connection is dead # Total dead-connection detection: 5 + (3 * 3) = 14 seconds tcp_keepalive_opts = if System.get_env("KUBERNETES_SERVICE_HOST") do [ {:keepalive, true}, {:raw, 6, 4, <<5::native-32>>}, {:raw, 6, 5, <<3::native-32>>}, {:raw, 6, 6, <<3::native-32>>} ] else [keepalive: true] end config :libcluster, topologies: libcluster_topologies # Configure Swoosh to use Req for API requests config :swoosh, :api_client, Swoosh.ApiClient.Req # Configure Oban for distributed job processing config :towerops, Oban, engine: Oban.Pro.Engines.Smart, repo: Towerops.Repo, shutdown_grace_period: to_timeout(second: 40), queues: [ default: max(1, div(10, oban_scale)), discovery: max(1, div(10, oban_scale)), # SNMP polling jobs - one per device pollers: max(1, div(String.to_integer(System.get_env("POLLER_CONCURRENCY") || "50"), oban_scale)), # Device monitoring jobs - health checks monitors: max(1, div(50, oban_scale)), # Service checks - HTTP/TCP/DNS checks: max(1, div(50, oban_scale)), check_executors: max(1, div(50, oban_scale)), # Alert notifications (PagerDuty, email, etc.) notifications: max(1, div(25, oban_scale)), maintenance: max(1, div(5, oban_scale)), weather: max(1, div(2, oban_scale)) ], plugins: [ # Cron jobs for periodic maintenance tasks {Oban.Plugins.Cron, crontab: [ # Run neighbor cleanup every hour {"0 * * * *", Towerops.Snmp.NeighborCleanupWorker}, # Check for stale agents every minute {"* * * * *", Towerops.Workers.StaleAgentWorker}, # Cloud latency probes every 8 hours (00:00, 08:00, 16:00) {"0 */8 * * *", Towerops.Workers.CloudLatencyProbeWorker}, # Evaluate latency-based agent reassignment 30 min after each probe batch {"30 0,8,16 * * *", Towerops.Workers.AgentLatencyEvaluator}, # Health check for missing jobs every 10 minutes {"*/10 * * * *", Towerops.Workers.JobHealthCheckWorker}, # Send notification digests for rate-limited alerts every 5 minutes {"*/5 * * * *", Towerops.Workers.AlertDigestWorker, args: %{"user_id" => "__cron__"}}, # Mark stale backup requests as timeout every 10 minutes {"*/10 * * * *", Towerops.Workers.BackupTimeoutWorker}, # Clean up old login history daily at 2 AM {"0 2 * * *", Towerops.Workers.LoginHistoryCleanupWorker}, # Clean up expired browser sessions daily at 3 AM {"0 3 * * *", Towerops.Workers.SessionCleanupWorker}, # Fetch latest firmware versions daily at 4 AM {"0 4 * * *", Towerops.Workers.FirmwareVersionFetcherWorker}, # MikroTik configuration backups daily at 7 AM UTC {"0 7 * * *", Towerops.Workers.MikrotikBackupWorker}, # Send backup summary email daily at 8 AM {"0 8 * * *", Towerops.Workers.BackupSummaryWorker}, # Delete expired IP bans every hour {"0 * * * *", Towerops.Workers.ExpiredBanCleanupWorker}, # Delete stale violation records daily at 2 AM {"0 2 * * *", Towerops.Workers.StaleViolationCleanupWorker}, # Sync Preseem data every 10 minutes {"*/5 * * * *", Towerops.Workers.PreseemSyncWorker}, # Compute Preseem baselines and fleet profiles nightly at 2:30 AM {"30 2 * * *", Towerops.Workers.PreseemBaselineWorker}, # Sync Gaiia data every 15 minutes {"*/5 * * * *", Towerops.Workers.GaiiaSyncWorker}, # Sync NetBox data every 30 minutes {"*/5 * * * *", Towerops.Workers.NetBoxSyncWorker}, # Sync Sonar billing data every 5 minutes {"*/5 * * * *", Towerops.Workers.SonarSyncWorker}, # Sync Splynx billing data every 5 minutes {"*/5 * * * *", Towerops.Workers.SplynxSyncWorker}, # Sync VISP billing data every 5 minutes {"*/5 * * * *", Towerops.Workers.VispSyncWorker}, # Sync UISP infrastructure data every 5 minutes {"*/5 * * * *", Towerops.Workers.UispSyncWorker}, # Sync Cambium cnMaestro data every 5 minutes {"*/5 * * * *", Towerops.Workers.CnMaestroSyncWorker}, # Scheduled reports — check for due reports every hour {"0 * * * *", Towerops.Workers.ReportWorker}, # Device health insights nightly at 3:30 AM {"30 3 * * *", Towerops.Workers.DeviceHealthInsightWorker}, # System insights (agent offline detection) every 5 minutes {"*/5 * * * *", Towerops.Workers.SystemInsightWorker}, # Wireless client health insights every 5 minutes {"*/5 * * * *", Towerops.Workers.WirelessInsightWorker}, # Gaiia reconciliation insights nightly at 4:30 AM {"30 4 * * *", Towerops.Workers.GaiiaInsightWorker}, # Backhaul capacity utilization insights every 15 minutes {"*/15 * * * *", Towerops.Workers.CapacityInsightWorker}, # Sync device usage to Stripe daily at 3 AM UTC {"0 3 * * *", Towerops.Workers.BillingSyncWorker}, # Data retention cleanup nightly at 1 AM {"0 1 * * *", Towerops.Workers.DataRetentionWorker} ]}, # Automatically delete completed jobs after 60 seconds. # Default limit of 10k/30s (333/s) can't keep up with job throughput (~600+/s), # causing oban_jobs to bloat to millions of rows and slow all queries. # 50k/5s (10,000/s) keeps the table small. {Oban.Plugins.Pruner, max_age: 60, limit: 50_000, interval: to_timeout(second: 5)}, # Rescue orphaned jobs (when node crashes) and prevent index bloat # High DELETE churn (10k jobs/sec pruned) causes severe index bloat on: # - Primary key (82.2 bloat, 3.2GB waste) # - Scheduling index (216.8 bloat, 7.9GB waste) # REINDEX CONCURRENTLY reclaims space without blocking operations. {Oban.Plugins.Reindexer, schedule: "0 2 * * *", indexes: [ "oban_jobs_pkey", "oban_jobs_state_queue_priority_scheduled_at_id_index", "oban_jobs_args_index", "oban_jobs_meta_index" ]}, # Rescue jobs stuck in executing state (mark as cancelled after 10 minutes) {Oban.Plugins.Lifeline, rescue_after: to_timeout(minute: 10)} ] # ## SSL Support # # To get SSL working, you will need to add the `https` key # to your endpoint configuration: # # config :towerops, ToweropsWeb.Endpoint, # https: [ # ..., # port: 443, # cipher_suite: :strong, # keyfile: System.get_env("SOME_APP_SSL_KEY_PATH"), # certfile: System.get_env("SOME_APP_SSL_CERT_PATH") # ] # # The `cipher_suite` is set to `:strong` to support only the # latest and more secure SSL ciphers. This means old browsers # and clients may not be supported. You can set it to # `:compatible` for wider support. # # `:keyfile` and `:certfile` expect an absolute path to the key # and cert in disk or a relative path inside priv, for example # "priv/ssl/server.key". For all supported SSL configuration # options, see https://hexdocs.pm/plug/Plug.SSL.html#configure/1 # # We also recommend setting `force_ssl` in your config/prod.exs, # ensuring no data is ever sent via http, always redirecting to https: # # config :towerops, ToweropsWeb.Endpoint, # force_ssl: [hsts: true] # # Check `Plug.SSL` for all available options in `force_ssl`. # ## Configuring the mailer # # In production you need to configure the mailer to use a different adapter. # Here is an example configuration for Mailgun: # # config :towerops, Towerops.Mailer, # adapter: Swoosh.Adapters.Mailgun, # api_key: System.get_env("MAILGUN_API_KEY"), # domain: System.get_env("MAILGUN_DOMAIN") # # Most non-SMTP adapters require an API client. Swoosh supports Req, Hackney, # and Finch out-of-the-box. This configuration is typically done at # compile-time in your config/prod.exs: # # config :swoosh, :api_client, Swoosh.ApiClient.Req # # See https://hexdocs.pm/swoosh/Swoosh.html#module-installation for details. # Configure Amazon SES for production email config :towerops, Towerops.Mailer, adapter: Swoosh.Adapters.AmazonSES, region: System.get_env("AWS_REGION") || "us-east-1", access_key: System.get_env("AWS_ACCESS_KEY_ID"), secret: System.get_env("AWS_SECRET_ACCESS_KEY") config :towerops, Towerops.Repo, ssl: ssl_config, url: database_url, pool_size: String.to_integer(System.get_env("POOL_SIZE") || "10"), # Proactively ping idle connections every 1s to detect dead SSL connections # before they're handed to a real query. Without this, a dead connection # sits in the pool until a query checks it out and hangs for the full timeout. idle_interval: 1_000, # Detect stale connections faster: if a query takes longer than queue_target (2s) # on average over queue_interval (5s), DBConnection will disconnect idle connections # to cycle in fresh ones. Helps recover from "ssl recv: closed" errors. queue_target: 2_000, queue_interval: 5_000, # PostgreSQL server-side statement timeout: kill any query running longer than 30s. # Prevents queries from hanging indefinitely on dead SSL connections. parameters: [application_name: "towerops", statement_timeout: "30000"], # Disconnect immediately on fatal PostgreSQL error codes for faster pool recovery: # 57P01 = admin_shutdown, 57P02 = crash_shutdown, 57P03 = cannot_connect_now disconnect_on_error_codes: [:"57P01", :"57P02", :"57P03"], socket_options: maybe_ipv6 ++ tcp_keepalive_opts config :towerops, Towerops.Vault, ciphers: [ default: { Cloak.Ciphers.AES.GCM, tag: "AES.GCM.V1", key: Base.decode64!(cloak_key) } ] config :towerops, ToweropsWeb.Endpoint, url: [host: host, port: 443, scheme: "https"], check_origin: [ "//towerops.net", "//*.towerops.net" ], http: [ # Enable IPv6 and bind on all interfaces. # Set it to {0, 0, 0, 0, 0, 0, 0, 1} for local network only access. # See the documentation on https://hexdocs.pm/bandit/Bandit.html#t:options/0 # for details about using IPv6 vs IPv4 and loopback vs public addresses. ip: {0, 0, 0, 0, 0, 0, 0, 0}, # Bandit uses ThousandIsland for connection management # Timeout for reading client data before closing the connection. # Set high enough to accommodate long-lived WebSocket connections # (agents send Phoenix transport heartbeats every 25s). thousand_island_options: [ read_timeout: 120_000 ] ], secret_key_base: secret_key_base # Webhook secret for CI-triggered mass agent updates config :towerops, :agent_webhook_secret, System.get_env("AGENT_WEBHOOK_SECRET") config :towerops, :dns_cluster_query, System.get_env("DNS_CLUSTER_QUERY") # Set environment identifier for runtime checks config :towerops, :env, :prod # Set default sender for all emails config :towerops, :mailer_from, {"Towerops", "hi@towerops.net"} config :towerops, :redis, redis_config # Cloudflare API credentials for brute force protection config :towerops, cloudflare_zone_id: System.get_env("CLOUDFLARE_ZONE_ID"), cloudflare_api_token: System.get_env("CLOUDFLARE_API_TOKEN") # Disable automatic timezone database updates in production # Timezone data is updated through dependency updates instead # This prevents crashes when network timeouts occur during automatic updates config :tzdata, :autoupdate, :disabled # OpenWeatherMap API key for site weather data if owm_key = System.get_env("OPENWEATHERMAP_API_KEY") do config :towerops, :openweathermap_api_key, owm_key end # Stripe configuration for billing stripe_secret_key = System.get_env("STRIPE_SECRET_KEY") || raise """ environment variable STRIPE_SECRET_KEY is missing. Get this from Stripe dashboard and store in K8s secrets. """ stripe_webhook_secret = System.get_env("STRIPE_WEBHOOK_SECRET") || raise """ environment variable STRIPE_WEBHOOK_SECRET is missing. Get this from Stripe webhook settings and store in K8s secrets. """ stripe_price_id = System.get_env("STRIPE_PRICE_ID") || raise """ environment variable STRIPE_PRICE_ID is missing. This is the Stripe price ID for the metered billing plan. """ stripe_meter_id = System.get_env("STRIPE_METER_ID") || raise """ environment variable STRIPE_METER_ID is missing. This is the Stripe billing meter ID for usage reporting. """ # Configure ErrorTrackerNotifier for email alerts on exceptions config :error_tracker_notifier, notification_type: :email, from_email: "alerts@towerops.net", to_email: System.get_env("ERROR_ALERT_EMAIL") || "graham@towerops.net", mailer: Towerops.Mailer config :towerops, stripe_secret_key: stripe_secret_key, stripe_webhook_secret: stripe_webhook_secret, stripe_price_id: stripe_price_id, stripe_meter_id: stripe_meter_id end # Test environment Stripe configuration if config_env() == :test do config :towerops, stripe_secret_key: "sk_test_fake", stripe_webhook_secret: "whsec_test_fake", stripe_price_id: "price_test_fake", stripe_meter_id: "meter_test_fake" end