fix: reduce health-check probe pressure to stop readiness-flapping downtime

Three changes targeting the root cause of production up/down alerting:

1. Skip BruteForceProtection for /health and /health/live paths
   (removes unnecessary Repo.get_by(IpBlock) DB query from every
   k8s probe, matching the existing /socket/agent exemption).

2. Relax k8s readiness probe: periodSeconds 5→10, timeoutSeconds
   2→5, failureThreshold 2→3, successThreshold 3→2 so transient
   DB/Redis blips don't cascade into full service unavailability.

3. Use the persistent Towerops.Redix connection for health checks
   instead of opening a brand-new TCP connection (handshake→AUTH→
   PING→close) on every probe. Towerops.Redix.Fake gains a
   set_ping_response/2 toggle for testing failure paths.
This commit is contained in:
Graham McIntire 2026-06-08 16:08:31 -05:00
parent 26ac82653e
commit b5e3c1f977
8 changed files with 135 additions and 89 deletions

View file

@ -1,3 +1,21 @@
2026-06-08
fix: reduce health-check probe pressure to stop readiness-flapping downtime
Three changes to eliminate the root cause of production up/down alerting:
1. BruteForceProtection plug now skips /health and /health/live paths (like
/socket/agent already does), removing an unnecessary Repo.get_by(IpBlock)
query from every k8s readiness probe.
2. k8s readiness probe relaxed: periodSeconds 5→10, timeoutSeconds 2→5,
failureThreshold 2→3 to tolerate brief DB/Redis blips without marking the
pod unready; successThreshold 3→2 to recover faster after a blip clears.
3. RedisHealthCheck.check_health/2 now sends PING on the persistent
Towerops.Redix connection (started at boot) instead of opening a brand-new
TCP connection + AUTH + PING + close on every probe. Towerops.Redix.Fake
gained a set_ping_response/2 toggle so tests can drive the failure path
without dialing unreachable hosts.
Files: lib/towerops/redis_health_check.ex, lib/towerops/redix/fake.ex,
lib/towerops_web/plugs/brute_force_protection.ex, k8s/deployment.yaml,
test/towerops_web/controllers/health_controller_test.exs
2026-05-20
fix(oban): use Oban Pro DynamicLifeline in prod to stop unique_violation crash loop
CheckExecutorWorker's `unique` constraint covers available/scheduled/retryable

View file

@ -293,10 +293,10 @@ spec:
path: /health
port: 4000
initialDelaySeconds: 5
periodSeconds: 5
timeoutSeconds: 2
successThreshold: 3 # Require 3 consecutive successes (15s) before marking ready
failureThreshold: 2 # Allow 1 failure before marking not ready
periodSeconds: 10 # Check every 10s instead of 5s (halves probe pressure)
timeoutSeconds: 5 # Give DB+Redis checks enough time to respond
successThreshold: 2 # 2 consecutive successes (20s) before marking ready
failureThreshold: 3 # Allow 2 consecutive failures before marking not ready
volumeMounts:
- name: data
mountPath: /data

View file

@ -63,55 +63,29 @@ defmodule Towerops.RedisHealthCheck do
end
@doc """
Checks Redis health by attempting a PING command with a timeout.
Checks Redis health by sending a PING on the existing persistent connection.
Uses the named `Towerops.Redix` connection (started at boot by
`Towerops.Application`) instead of opening a brand-new TCP connection
on every probe. This avoids the TCP handshake + AUTH per probe and
cuts per-check work roughly in half.
## Parameters
- `redis_config` - Keyword list with Redis connection config (host, port, password)
- `timeout_ms` - Maximum time to wait for response in milliseconds
- `redis_config` - Ignored; accepted for backward compat with callers.
- `timeout_ms` - Maximum time to wait for the PONG response.
## Returns
- `:ok` if Redis responds to PONG
- `{:error, reason}` if connection fails or times out
- `{:error, reason}` if the command fails or times out
"""
@spec check_health(keyword(), pos_integer()) :: :ok | {:error, term()}
def check_health(redis_config, timeout_ms \\ 2_000) do
host = Keyword.get(redis_config, :host, "localhost")
port = Keyword.get(redis_config, :port, 6379)
password = Keyword.get(redis_config, :password)
# Build connection options
conn_opts = [
host: host,
port: port,
timeout: timeout_ms,
socket_opts: [keepalive: true]
]
conn_opts =
if password && password != "" do
Keyword.put(conn_opts, :password, password)
else
conn_opts
end
def check_health(_redis_config, timeout_ms \\ 2_000) do
redix = redix_module()
# Start a temporary connection for health check
case redix.start_link(conn_opts) do
{:ok, conn} ->
result =
case redix.command(conn, ["PING"], timeout: timeout_ms) do
case redix.command(Towerops.Redix, ["PING"], timeout: timeout_ms) do
{:ok, "PONG"} -> :ok
{:error, reason} -> {:error, reason}
end
redix.stop(conn)
result
{:error, reason} ->
Logger.debug("Redis health check connection failed: #{inspect(reason)}")
{:error, reason}
end
rescue
error ->
Logger.debug("Redis health check error: #{inspect(error)}")

View file

@ -52,9 +52,19 @@ defmodule Towerops.Redix.Fake do
GenServer.call(conn, :reset)
end
@doc """
Overrides the PING response for testing failure paths.
Pass `:error` to make PING return `{:error, %Redix.ConnectionError{...}}`.
Pass `:ok` (or anything else) to restore the default `{:ok, "PONG"}`.
"""
def set_ping_response(conn \\ __MODULE__, response) do
GenServer.call(conn, {:set_ping_response, response})
end
# ── GenServer ────────────────────────────────────────────────────────
@impl true
def init(_), do: {:ok, %{sets: %{}}}
def init(_), do: {:ok, %{sets: %{}, ping_override: nil}}
@impl true
def handle_call({:command, cmd}, _from, state) do
@ -62,11 +72,26 @@ defmodule Towerops.Redix.Fake do
{:reply, reply, state}
end
def handle_call(:reset, _from, _state), do: {:reply, :ok, %{sets: %{}}}
def handle_call(:reset, _from, _state), do: {:reply, :ok, %{sets: %{}, ping_override: nil}}
def handle_call({:set_ping_response, response}, _from, state) do
{:reply, :ok, %{state | ping_override: response}}
end
# ── Command dispatch ────────────────────────────────────────────────
defp run(["PING"], state), do: {{:ok, "PONG"}, state}
defp run(["PING" | _], state), do: {{:ok, "PONG"}, state}
defp run(["PING"], state) do
case state.ping_override do
:error -> {{:error, %Redix.ConnectionError{reason: :econnrefused}}, state}
_ -> {{:ok, "PONG"}, state}
end
end
defp run(["PING" | _], state) do
case state.ping_override do
:error -> {{:error, %Redix.ConnectionError{reason: :econnrefused}}, state}
_ -> {{:ok, "PONG"}, state}
end
end
defp run(["SADD", key, value | rest], state) do
set = Map.get(state.sets, key, MapSet.new())

View file

@ -47,6 +47,12 @@ defmodule ToweropsWeb.Plugs.BruteForceProtection do
String.starts_with?(conn.request_path, "/socket/agent") ->
conn
# Skip for Kubernetes health probes — these run every 5s and don't
# need brute-force checking; a DB query here adds latency to readiness
# checks and contributes to probe failures during DB contention.
String.starts_with?(conn.request_path, "/health") ->
conn
# Check if IP is whitelisted
BruteForce.check_whitelist(ip_address) ->
conn

View file

@ -1,3 +1,7 @@
2026-06-08 — Availability improvements
* Reduced momentary unavailability during brief database or Redis slowdowns
* Health monitoring overhead cut significantly, improving overall responsiveness
2026-05-20 — Reliability improvements
* The app now rides through brief database hiccups instead of restarting, reducing momentary unavailability
* Background monitoring jobs recover cleanly after a restart instead of getting stuck

View file

@ -4,6 +4,13 @@ defmodule Towerops.RedisHealthCheckTest do
import ExUnit.CaptureLog
alias Towerops.RedisHealthCheck
alias Towerops.Redix.Fake
setup do
# Ensure the Fake is in its default (healthy) state before each test.
Fake.set_ping_response(Towerops.Redix, :ok)
:ok
end
describe "wait_for_redis/2" do
test "returns :ok when Redis is available" do
@ -12,10 +19,13 @@ defmodule Towerops.RedisHealthCheckTest do
assert :ok = RedisHealthCheck.wait_for_redis(redis_config, max_attempts: 2, backoff: 100)
end
test "returns error after max attempts with unreachable Redis" do
redis_config = [host: "127.0.0.1", port: 9999]
test "returns error after max attempts when Redis is unreachable" do
redis_config = [host: "127.0.0.1", port: 6379]
# Make the Fake simulate connection errors so wait_for_redis exhausts
# its retries and returns {:error, :max_attempts_exceeded}.
Fake.set_ping_response(Towerops.Redix, :error)
# Suppress Redis connection warnings during test
_logs =
capture_log(fn ->
assert {:error, :max_attempts_exceeded} =
@ -28,11 +38,13 @@ defmodule Towerops.RedisHealthCheckTest do
end
test "retries with exponential backoff" do
redis_config = [host: "127.0.0.1", port: 9999]
redis_config = [host: "127.0.0.1", port: 6379]
# Simulate failure so the retry loop runs.
Fake.set_ping_response(Towerops.Redix, :error)
start_time = System.monotonic_time(:millisecond)
# Suppress Redis connection warnings during test
_logs =
capture_log(fn ->
RedisHealthCheck.wait_for_redis(redis_config,
@ -48,8 +60,7 @@ defmodule Towerops.RedisHealthCheckTest do
# With backoff of 10ms and 2 attempts, total wait should be at least:
# attempt 1: fail + wait 10ms
# attempt 2: fail
# Total: ~10ms + connection timeouts
# Just verify some time passed (exponential backoff is working)
# Total: ~10ms + processing time
assert elapsed >= 10
end
end
@ -61,20 +72,11 @@ defmodule Towerops.RedisHealthCheckTest do
assert :ok = RedisHealthCheck.check_health(redis_config)
end
test "returns error for unavailable Redis" do
redis_config = [host: "127.0.0.1", port: 9999]
test "returns error when Redis is unreachable" do
redis_config = [host: "127.0.0.1", port: 6379]
# Suppress Redis connection warnings during test
_logs =
capture_log(fn ->
assert {:error, _reason} = RedisHealthCheck.check_health(redis_config, 10)
end)
end
Fake.set_ping_response(Towerops.Redix, :error)
test "handles malformed Redis config" do
redis_config = [host: "not-a-valid-hostname-at-all-12345", port: 6379]
# Suppress Redis connection warnings during test
_logs =
capture_log(fn ->
assert {:error, _reason} = RedisHealthCheck.check_health(redis_config, 10)
@ -82,13 +84,13 @@ defmodule Towerops.RedisHealthCheckTest do
end
test "accepts custom timeout" do
redis_config = [host: "127.0.0.1", port: 9999]
redis_config = [host: "127.0.0.1", port: 6379]
Fake.set_ping_response(Towerops.Redix, :error)
# Suppress Redis connection warnings during test
_logs =
capture_log(fn ->
result = RedisHealthCheck.check_health(redis_config, 10)
# Should return error (either immediate connection failure or timeout)
assert match?({:error, _}, result)
end)
end
@ -110,34 +112,44 @@ defmodule Towerops.RedisHealthCheckTest do
end
describe "check_health/2 with a password" do
test "passes the :password option through and fails auth" do
# Bogus password against a Redis that doesn't require auth → AUTH error.
# Either way the password keyword arm is exercised.
redis_config = [host: "127.0.0.1", port: 9999, password: "wrong"]
test "ignores password and succeeds when the persistent connection is healthy" do
# The health check now uses the persistent Towerops.Redix connection,
# which is already authenticated. The password in redis_config is ignored.
redis_config = [host: "127.0.0.1", port: 6379, password: "bogus"]
_logs =
capture_log(fn ->
assert match?({:error, _}, RedisHealthCheck.check_health(redis_config, 10))
end)
assert :ok = RedisHealthCheck.check_health(redis_config, 10)
end
test "ignores empty-string password" do
redis_config = [host: "127.0.0.1", port: 9999, password: ""]
redis_config = [host: "127.0.0.1", port: 6379, password: ""]
_logs =
capture_log(fn ->
assert match?({:error, _}, RedisHealthCheck.check_health(redis_config, 10))
end)
assert :ok = RedisHealthCheck.check_health(redis_config, 10)
end
end
describe "wait_for_redis/2 — successful retry" do
test "logs 'connection established' when first attempt fails but second succeeds" do
# Hard to deterministically simulate a transient failure; we just confirm
# the success path returns :ok when Redis is up. Together with the
# max-attempts test above this hits both branches around the attempt counter.
# Simulate one failure then recovery.
Fake.set_ping_response(Towerops.Redix, :error)
# Schedule recovery after a short delay — the first attempt will fail,
# then we restore the healthy state before the second attempt.
Task.async(fn ->
Process.sleep(5)
Fake.set_ping_response(Towerops.Redix, :ok)
end)
redis_config = Application.get_env(:towerops, :redis, host: "localhost", port: 6379)
assert :ok = RedisHealthCheck.wait_for_redis(redis_config, max_attempts: 1)
_logs =
capture_log(fn ->
assert :ok =
RedisHealthCheck.wait_for_redis(redis_config,
max_attempts: 3,
backoff: 5,
timeout: 10
)
end)
end
end
end

View file

@ -1,6 +1,8 @@
defmodule ToweropsWeb.HealthControllerTest do
use ToweropsWeb.ConnCase, async: false
alias Towerops.Redix.Fake
describe "GET /health/live" do
test "returns 200 with status ok", %{conn: conn} do
conn = get(conn, "/health/live")
@ -80,9 +82,12 @@ defmodule ToweropsWeb.HealthControllerTest do
test "returns 503 when redis is configured but unreachable", %{conn: conn} do
previous = Application.get_env(:towerops, :redis)
# Point at a closed port on a routable but unreachable address.
# 198.51.100.0/24 is TEST-NET-2 (RFC 5737) — guaranteed not routable on the public internet.
Application.put_env(:towerops, :redis, host: "198.51.100.1", port: 1, timeout: 50)
Application.put_env(:towerops, :redis, host: "localhost", port: 6379)
# Make the Fake return connection errors for PING so we exercise
# the Redis-down branch without trying to open a real socket.
# The Fake is registered as `Towerops.Redix`, not by its module name.
Fake.set_ping_response(Towerops.Redix, :error)
try do
conn = get(conn, "/health")
@ -94,6 +99,8 @@ defmodule ToweropsWeb.HealthControllerTest do
assert result["database"] == "connected"
assert result["redis"] == "disconnected"
after
Fake.set_ping_response(Towerops.Redix, :ok)
if previous do
Application.put_env(:towerops, :redis, previous)
else