From 439c25ba9d3b5e597c4e09738b2327a717779c35 Mon Sep 17 00:00:00 2001 From: Graham McIntire Date: Fri, 20 Mar 2026 10:04:06 -0500 Subject: [PATCH] Fix DB connection timeouts in k8s with aggressive TCP keepalive (#92) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SSL connections between pods and PostgreSQL are silently dying, causing 15s checkout timeouts and cascading Oban failures. - Set Linux TCP keepalive: 15s idle, 5s interval, 3 probes (30s detection) - Add PostgreSQL statement_timeout (30s) to kill hung queries server-side - Add disconnect_on_error_codes for fast recovery from PG shutdowns - Bump default pool_size 20 → 30 to reduce contention under load Reviewed-on: https://git.mcintire.me/graham/towerops-web/pulls/92 --- config/runtime.exs | 31 ++++++++++++++++++++++++++++--- 1 file changed, 28 insertions(+), 3 deletions(-) diff --git a/config/runtime.exs b/config/runtime.exs index 1fe02a66..214ba2e9 100644 --- a/config/runtime.exs +++ b/config/runtime.exs @@ -148,6 +148,26 @@ if config_env() == :prod do # OBAN_QUEUE_SCALE divides all queue sizes (e.g. "5" = 1/5th capacity for staging) oban_scale = String.to_integer(System.get_env("OBAN_QUEUE_SCALE") || "1") + # Aggressive TCP keepalive for Kubernetes environments. + # Default OS keepalive is ~2 hours on Linux, far too slow for k8s where + # load balancers and proxies silently drop idle connections within minutes. + # Linux raw socket options (SOL_TCP=6): + # TCP_KEEPIDLE (4) = 15s - seconds idle before first keepalive probe + # TCP_KEEPINTVL (5) = 5s - seconds between subsequent probes + # TCP_KEEPCNT (6) = 3 - failed probes before connection is dead + # Total dead-connection detection: 15 + (5 * 3) = 30 seconds + tcp_keepalive_opts = + if System.get_env("KUBERNETES_SERVICE_HOST") do + [ + {:keepalive, true}, + {:raw, 6, 4, <<15::native-32>>}, + {:raw, 6, 5, <<5::native-32>>}, + {:raw, 6, 6, <<3::native-32>>} + ] + else + [keepalive: true] + end + config :libcluster, topologies: libcluster_topologies # Configure Swoosh to use Req for API requests @@ -306,14 +326,19 @@ if config_env() == :prod do config :towerops, Towerops.Repo, ssl: ssl_config, url: database_url, - pool_size: String.to_integer(System.get_env("POOL_SIZE") || "20"), + pool_size: String.to_integer(System.get_env("POOL_SIZE") || "30"), # Detect stale connections faster: if a query takes longer than queue_target (2s) # on average over queue_interval (5s), DBConnection will disconnect idle connections # to cycle in fresh ones. Helps recover from "ssl recv: closed" errors. queue_target: 2_000, queue_interval: 5_000, - # TCP keepalive prevents proxies/firewalls from dropping idle connections - socket_options: maybe_ipv6 ++ [keepalive: true] + # PostgreSQL server-side statement timeout: kill any query running longer than 30s. + # Prevents queries from hanging indefinitely on dead SSL connections. + parameters: [application_name: "towerops", statement_timeout: "30000"], + # Disconnect immediately on fatal PostgreSQL error codes for faster pool recovery: + # 57P01 = admin_shutdown, 57P02 = crash_shutdown, 57P03 = cannot_connect_now + disconnect_on_error_codes: [:"57P01", :"57P02", :"57P03"], + socket_options: maybe_ipv6 ++ tcp_keepalive_opts config :towerops, Towerops.Vault, ciphers: [