feat: event driven agent connection metric (#24355)

Moves the `coderd_agents_first_connection_seconds` histogram from the
polling-based `prometheusmetrics.Agents()` loop to the event-driven
`agentConnectionMonitor.init()` path. The metric is now recorded exactly
once when an agent first connects over the RPC websocket, instead of
being retroactively computed each polling tick.

The `username` and `workspace_name` labels are removed to reduce
cardinality; only `template_name` and `agent_name` are retained.

Adds unit tests covering both the happy path (first connection recorded)
and the negative-duration guard (clock skew logs a warning, no sample
emitted).
This commit is contained in:
J. Scott Miller
2026-05-11 14:27:40 -05:00
committed by GitHub
parent e56381eb61
commit 3e46c7986f
7 changed files with 145 additions and 118 deletions
+58
View File
@@ -12,6 +12,7 @@ import (
"github.com/google/uuid"
"github.com/hashicorp/yamux"
"github.com/prometheus/client_golang/prometheus"
"golang.org/x/xerrors"
"cdr.dev/slog/v3"
@@ -258,6 +259,7 @@ func (api *API) startAgentYamuxMonitor(ctx context.Context,
replicaID: api.ID,
updater: api,
disconnectTimeout: api.AgentInactiveDisconnectTimeout,
metrics: api.workspaceAgentRPCMetrics,
logger: api.Logger.With(
slog.F("workspace_id", workspaceBuild.WorkspaceID),
slog.F("agent_id", workspaceAgent.ID),
@@ -291,6 +293,7 @@ type agentConnectionMonitor struct {
updater workspaceUpdater
logger slog.Logger
pingPeriod time.Duration
metrics *WorkspaceAgentRPCMetrics
// state manipulated by both sendPings() and monitor() goroutines: needs to be threadsafe
lastPing atomic.Pointer[time.Time]
@@ -356,6 +359,14 @@ func (m *agentConnectionMonitor) init() {
Time: now,
Valid: true,
}
if m.metrics != nil {
duration := now.Sub(m.workspaceAgent.CreatedAt)
m.metrics.ObserveAgentFirstConnection(
duration,
m.workspace.TemplateName,
m.workspaceAgent.Name,
)
}
}
m.lastConnectedAt = sql.NullTime{
Time: now,
@@ -496,3 +507,50 @@ func checkBuildIsLatest(ctx context.Context, db database.Store, build database.W
}
return nil
}
// WorkspaceAgentRPCMetrics holds Prometheus metrics for the agent
// connection monitor. It is nil when Prometheus is not enabled.
type WorkspaceAgentRPCMetrics struct {
logger slog.Logger
FirstConnectionDuration *prometheus.HistogramVec
}
// NewWorkspaceAgentRPCMetrics creates and registers agent connection
// metrics.
func NewWorkspaceAgentRPCMetrics(reg prometheus.Registerer, logger slog.Logger) *WorkspaceAgentRPCMetrics {
m := &WorkspaceAgentRPCMetrics{
logger: logger,
FirstConnectionDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{
Namespace: "coderd",
Subsystem: "agents",
Name: "first_connection_seconds",
Help: "Duration from agent creation to first connection in seconds.",
Buckets: []float64{1, 10, 30, 60, 120, 300, 600, 1800, 3600},
}, []string{"template_name", "agent_name"}),
}
reg.MustRegister(m.FirstConnectionDuration)
return m
}
// ObserveAgentFirstConnection records the duration from agent creation
// to first connection. Negative durations are logged as warnings and
// not recorded, since they indicate clock skew.
func (m *WorkspaceAgentRPCMetrics) ObserveAgentFirstConnection(
duration time.Duration,
templateName string,
agentName string,
) {
if duration < 0 {
m.logger.Warn(context.Background(),
"negative agent first connection duration, possible clock skew",
slog.F("template_name", templateName),
slog.F("agent_name", agentName),
slog.F("duration", duration),
)
return
}
m.FirstConnectionDuration.WithLabelValues(
templateName,
agentName,
).Observe(duration.Seconds())
}