diff --git a/config/runtime.exs b/config/runtime.exs index 1fe02a66..214ba2e9 100644 --- a/config/runtime.exs +++ b/config/runtime.exs @@ -148,6 +148,26 @@ if config_env() == :prod do # OBAN_QUEUE_SCALE divides all queue sizes (e.g. "5" = 1/5th capacity for staging) oban_scale = String.to_integer(System.get_env("OBAN_QUEUE_SCALE") || "1") + # Aggressive TCP keepalive for Kubernetes environments. + # Default OS keepalive is ~2 hours on Linux, far too slow for k8s where + # load balancers and proxies silently drop idle connections within minutes. + # Linux raw socket options (SOL_TCP=6): + # TCP_KEEPIDLE (4) = 15s - seconds idle before first keepalive probe + # TCP_KEEPINTVL (5) = 5s - seconds between subsequent probes + # TCP_KEEPCNT (6) = 3 - failed probes before connection is dead + # Total dead-connection detection: 15 + (5 * 3) = 30 seconds + tcp_keepalive_opts = + if System.get_env("KUBERNETES_SERVICE_HOST") do + [ + {:keepalive, true}, + {:raw, 6, 4, <<15::native-32>>}, + {:raw, 6, 5, <<5::native-32>>}, + {:raw, 6, 6, <<3::native-32>>} + ] + else + [keepalive: true] + end + config :libcluster, topologies: libcluster_topologies # Configure Swoosh to use Req for API requests @@ -306,14 +326,19 @@ if config_env() == :prod do config :towerops, Towerops.Repo, ssl: ssl_config, url: database_url, - pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"), + pool_size: String.to_integer(System.get_env("POOL_SIZE") || "30"), # Detect stale connections faster: if a query takes longer than queue_target (2s) # on average over queue_interval (5s), DBConnection will disconnect idle connections # to cycle in fresh ones. Helps recover from "ssl recv: closed" errors. queue_target: 2_000, queue_interval: 5_000, - # TCP keepalive prevents proxies/firewalls from dropping idle connections - socket_options: maybe_ipv6 ++ [keepalive: true] + # PostgreSQL server-side statement timeout: kill any query running longer than 30s. + # Prevents queries from hanging indefinitely on dead SSL connections. + parameters: [application_name: "towerops", statement_timeout: "30000"], + # Disconnect immediately on fatal PostgreSQL error codes for faster pool recovery: + # 57P01 = admin_shutdown, 57P02 = crash_shutdown, 57P03 = cannot_connect_now + disconnect_on_error_codes: [:"57P01", :"57P02", :"57P03"], + socket_options: maybe_ipv6 ++ tcp_keepalive_opts config :towerops, Towerops.Vault, ciphers: [