aprs.me/lib/aprsme_web/telemetry.ex
Graham McIntire 24796f98d4
Optimize prod: scale deployment, drop expensive metric queries, fix slow callsign and nearby-stations queries
- k8s: 2 replicas, POOL_SIZE 25, cpu 2000m, mem 1Gi (was 1 replica, 5, 500m, 512Mi)
- telemetry poller 10s -> 60s; drop pg_database_size, pg_stat_statements,
  pg_table_size and pg_indexes_size (these were ~7100s of cumulative DB time
  on their own)
- get_latest_packet_for_callsign: bound by :packet_retention_days so partition
  pruning kicks in
- get_nearby_stations_knn: ST_DWithin spatial pre-filter (default 500km) so
  the GiST geography index cuts candidates before DISTINCT ON sort
  (EXPLAIN: 5793ms -> 400ms)
- New migration: partial functional indexes on upper(object_name) and
  upper(item_name), built per-partition CONCURRENTLY then attached to parent.
  Lets get_latest_packet_for_callsign use BitmapOr across all 3 upper()
  indexes instead of falling back to a scan
2026-05-12 17:48:58 -05:00

190 lines
8.3 KiB
Elixir

defmodule AprsmeWeb.Telemetry do
@moduledoc false
use Supervisor
import Telemetry.Metrics
alias Aprsme.Telemetry.DatabaseMetrics
def start_link(arg) do
Supervisor.start_link(__MODULE__, arg, name: __MODULE__)
end
@impl true
def init(_arg) do
children = [
{:telemetry_poller, measurements: periodic_measurements(), period: 60_000}
]
Supervisor.init(children, strategy: :one_for_one)
end
def metrics do
# Default buckets for all distribution metrics
buckets = [0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10_000]
[
# Phoenix Metrics - Use distribution for latency measurements
distribution("phoenix.endpoint.start.system_time",
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.endpoint.stop.duration",
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.router_dispatch.start.system_time",
tags: [:route],
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.router_dispatch.exception.duration",
tags: [:route],
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.router_dispatch.stop.duration",
tags: [:route],
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.socket_connected.duration",
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.channel_join.duration",
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
distribution("phoenix.channel_handled_in.duration",
tags: [:event],
unit: {:native, :millisecond},
reporter_options: [buckets: buckets]
),
# Database Metrics - Use distribution for query times
distribution("aprsme.repo.query.total_time",
unit: {:native, :millisecond},
description: "The sum of the other measurements",
reporter_options: [buckets: buckets]
),
distribution("aprsme.repo.query.decode_time",
unit: {:native, :millisecond},
description: "The time spent decoding the data received from the database",
reporter_options: [buckets: buckets]
),
distribution("aprsme.repo.query.query_time",
unit: {:native, :millisecond},
description: "The time spent executing the query",
reporter_options: [buckets: buckets]
),
distribution("aprsme.repo.query.queue_time",
unit: {:native, :millisecond},
description: "The time spent waiting for a database connection",
reporter_options: [buckets: buckets]
),
distribution("aprsme.repo.query.idle_time",
unit: {:native, :millisecond},
description: "The time the connection spent waiting before being checked out for the query",
reporter_options: [buckets: buckets]
),
# VM Metrics - Use last_value for current state
last_value("vm.memory.total", unit: {:byte, :kilobyte}),
last_value("vm.total_run_queue_lengths.total"),
last_value("vm.total_run_queue_lengths.cpu"),
last_value("vm.total_run_queue_lengths.io"),
# GenStage Packet Pipeline Metrics - Use counter and distribution
counter("aprsme.packet_pipeline.batch.count", unit: :event, description: "Total number of packets processed"),
counter("aprsme.packet_pipeline.batch.success",
unit: :event,
description: "Total number of successful inserts"
),
counter("aprsme.packet_pipeline.batch.error", unit: :event, description: "Total number of errors"),
distribution("aprsme.packet_pipeline.batch.duration_ms",
unit: :millisecond,
description: "Batch insert duration (ms)",
reporter_options: [buckets: buckets]
),
# Backpressure Metrics
counter("aprsme.packet_producer.backpressure.count",
unit: :event,
description: "Backpressure state changes"
),
# Spatial PubSub Metrics
last_value("aprsme.spatial_pubsub.clients.count", description: "Number of connected clients"),
last_value("aprsme.spatial_pubsub.clients.grid_cells", description: "Number of active grid cells"),
last_value("aprsme.spatial_pubsub.clients.avg_clients_per_cell", description: "Average clients per grid cell"),
counter("aprsme.spatial_pubsub.broadcasts.total", description: "Total spatial broadcasts sent"),
counter("aprsme.spatial_pubsub.broadcasts.filtered", description: "Broadcasts filtered by viewport"),
counter("aprsme.spatial_pubsub.broadcasts.packets", description: "Total packets processed"),
last_value("aprsme.spatial_pubsub.efficiency.ratio", unit: :percent, description: "Broadcast efficiency ratio"),
counter("aprsme.spatial_pubsub.efficiency.saved_broadcasts",
description: "Number of broadcasts saved by filtering"
),
# Database Connection Pool Metrics
last_value("aprsme.repo.pool.size", description: "Configured pool size"),
last_value("aprsme.repo.pool.idle", description: "Number of idle connections"),
last_value("aprsme.repo.pool.busy", description: "Number of busy connections"),
last_value("aprsme.repo.pool.available", description: "Number of available connections"),
last_value("aprsme.repo.pool.queue_length", description: "Number of processes waiting for a connection"),
last_value("aprsme.repo.pool.total", description: "Total connections in pool"),
# PostgreSQL Database Metrics
last_value("aprsme.postgres.database.size_bytes", unit: {:byte, :megabyte}, description: "Database size"),
last_value("aprsme.postgres.connections.total", description: "Total database connections"),
last_value("aprsme.postgres.connections.active", description: "Active database connections"),
last_value("aprsme.postgres.connections.idle", description: "Idle database connections"),
last_value("aprsme.postgres.connections.idle_in_transaction", description: "Idle in transaction connections"),
last_value("aprsme.postgres.connections.waiting", description: "Connections waiting on locks"),
# Packets Table Metrics
last_value("aprsme.postgres.packets_table.live_tuples", description: "Number of live rows in packets table"),
last_value("aprsme.postgres.packets_table.dead_tuples", description: "Number of dead rows in packets table"),
counter("aprsme.postgres.packets_table.total_inserts", description: "Total inserts to packets table"),
counter("aprsme.postgres.packets_table.total_updates", description: "Total updates to packets table"),
counter("aprsme.postgres.packets_table.total_deletes", description: "Total deletes from packets table"),
last_value("aprsme.postgres.packets_table.table_size_bytes",
unit: {:byte, :megabyte},
description: "Packets table size"
),
last_value("aprsme.postgres.packets_table.indexes_size_bytes",
unit: {:byte, :megabyte},
description: "Packets indexes size"
),
# Query Performance Metrics
counter("aprsme.postgres.query_stats.total_calls", description: "Total number of queries executed"),
last_value("aprsme.postgres.query_stats.total_time_ms",
unit: :millisecond,
description: "Total query execution time"
),
last_value("aprsme.postgres.query_stats.avg_time_ms",
unit: :millisecond,
description: "Average query execution time"
),
last_value("aprsme.postgres.query_stats.max_time_ms",
unit: :millisecond,
description: "Maximum query execution time"
),
# Replication Metrics (if applicable)
last_value("aprsme.postgres.replication.lag_seconds", unit: :second, description: "Replication lag in seconds")
]
end
defp periodic_measurements do
[
# A module, function and arguments to be invoked periodically.
# This function must call :telemetry.execute/3 and a metric must be added above.
# {AprsmeWeb, :count_users, []}
{DatabaseMetrics, :collect_db_pool_metrics, []},
{DatabaseMetrics, :collect_postgres_metrics, []},
{DatabaseMetrics, :collect_pgbouncer_metrics, []}
]
end
end