fix: implement connection tracking in metrics
Right now metrics observation handler does not track database connections but updates a single Gauge based on HasqlPoolObs events. This is problematic because Hasql pool reports various connection events in multiple phases. The connection state machine is not simple and to precisely report the number of connections in various states, it is necessary to track their lifecycles. This change adds a ConnTrack data structure and logic to track database connections lifecycles. At the moment it supports "connected" and "inUse" connection counts precisely. The "pgrst_db_pool_available" metric is implemented on top of ConnTrack instead of a simple Gauge.
This commit is contained in:
committed by
Steve Chavez
parent
f6e99117ab
commit
ed2dc1fe86
@@ -1811,10 +1811,6 @@ def test_server_timing_transaction_duration(defaultenv, metapostgrest):
|
||||
assert 2000 <= response_dur < 3000
|
||||
|
||||
|
||||
@pytest.mark.xfail(
|
||||
reason="pgrst_db_pool_available should not go negative on pg network failures",
|
||||
strict=True,
|
||||
)
|
||||
def test_positive_pool_metric(defaultenv):
|
||||
"When a network failure is caused on the pg connection, pgrst_db_pool_available stays positive"
|
||||
|
||||
|
||||
@@ -6,17 +6,21 @@
|
||||
|
||||
module Observation.MetricsSpec where
|
||||
|
||||
import Data.List (lookup)
|
||||
import Network.Wai (Application)
|
||||
import Data.List (lookup)
|
||||
import qualified Hasql.Pool.Observation as SQL
|
||||
import Network.Wai (Application)
|
||||
import ObsHelper
|
||||
import qualified PostgREST.AppState as AppState
|
||||
import PostgREST.Config (AppConfig (configDbSchemas))
|
||||
import qualified PostgREST.Metrics as Metrics
|
||||
import qualified PostgREST.AppState as AppState
|
||||
import PostgREST.Config (AppConfig (configDbSchemas))
|
||||
import PostgREST.Metrics (ConnStats (..),
|
||||
MetricsState (..),
|
||||
connectionCounts)
|
||||
import PostgREST.Observation
|
||||
import Prometheus (getCounter, getVectorWith)
|
||||
import Protolude
|
||||
import Test.Hspec (SpecWith, describe, it)
|
||||
import Test.Hspec.Wai (getState)
|
||||
import Prometheus (getCounter, getVectorWith)
|
||||
import Test.Hspec (SpecWith, describe, it)
|
||||
import Test.Hspec.Wai (getState)
|
||||
|
||||
import Protolude
|
||||
|
||||
spec :: SpecWith (SpecState, Application)
|
||||
spec = describe "Server started with metrics enabled" $ do
|
||||
@@ -71,9 +75,40 @@ spec = describe "Server started with metrics enabled" $ do
|
||||
-- (there should be none but we need to verify that)
|
||||
threadDelay $ 1 * sec
|
||||
|
||||
-- The test verifies we properly count in use connections
|
||||
-- The idea is to fork a worker thread that
|
||||
-- borrows connection from the pool and waits for a signal to release it
|
||||
-- Main thread checks that
|
||||
-- in use connections counter is incremented by worker
|
||||
-- then it signals the worker to release the connection
|
||||
-- and finally verifies that in use connection counter is back to original value
|
||||
it "Should track in use connections" $ do
|
||||
SpecState{specAppState = appState, specMetrics = metrics, specObsChan} <- getState
|
||||
let waitFor = waitForObs specObsChan
|
||||
|
||||
liftIO $ checkState' metrics [
|
||||
-- we expect in use connections to be the same once finished
|
||||
inUseConnections (+ 0)
|
||||
] $ do
|
||||
signal <- newEmptyMVar
|
||||
-- make sure waiting thread is signaled
|
||||
(`finally` tryPutMVar signal ()) $
|
||||
-- expecting one more connection in use
|
||||
checkState' metrics [
|
||||
inUseConnections (+ 1)
|
||||
] $ do
|
||||
-- start a thread hanging on a single connection until signaled
|
||||
void $ forkIO $ void $ AppState.usePool appState $ liftIO (readMVar signal)
|
||||
-- main thread waits for ConnectionObservation with InUseConnectionStatus
|
||||
-- after which used connections count should be incremented
|
||||
waitFor (1 * sec) "InUseConnectionStatus" $ \x -> [ o | o@(HasqlPoolObs (SQL.ConnectionObservation _ SQL.InUseConnectionStatus)) <- pure x]
|
||||
|
||||
-- hanging thread was signaled and should return the connection
|
||||
waitFor (1 * sec) "ReadyForUseConnectionStatus" $ \x -> [ o | o@(HasqlPoolObs (SQL.ConnectionObservation _ SQL.ReadyForUseConnectionStatus)) <- pure x]
|
||||
|
||||
where
|
||||
-- prometheus-client api to handle vectors is convoluted
|
||||
schemaCacheLoads label = expectField @"schemaCacheLoads" $
|
||||
fmap (maybe (0::Int) round . lookup label) . (`getVectorWith` getCounter)
|
||||
inUseConnections = expectField @"connTrack" ((inUse <$>) . connectionCounts)
|
||||
sec = 1000000
|
||||
|
||||
Reference in New Issue
Block a user