mirror of
https://github.com/coder/coder.git
synced 2026-09-24 15:04:27 +08:00
feat: add provisionerd prometheus metrics (#4909)
This commit is contained in:
@@ -11,6 +11,8 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/hashicorp/yamux"
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus/promauto"
|
||||
"github.com/spf13/afero"
|
||||
"go.opentelemetry.io/otel/attribute"
|
||||
semconv "go.opentelemetry.io/otel/semconv/v1.11.0"
|
||||
@@ -41,9 +43,10 @@ type Provisioners map[string]sdkproto.DRPCProvisionerClient
|
||||
|
||||
// Options provides customizations to the behavior of a provisioner daemon.
|
||||
type Options struct {
|
||||
Filesystem afero.Fs
|
||||
Logger slog.Logger
|
||||
Tracer trace.TracerProvider
|
||||
Filesystem afero.Fs
|
||||
Logger slog.Logger
|
||||
TracerProvider trace.TracerProvider
|
||||
Metrics *Metrics
|
||||
|
||||
ForceCancelInterval time.Duration
|
||||
UpdateInterval time.Duration
|
||||
@@ -66,14 +69,19 @@ func New(clientDialer Dialer, opts *Options) *Server {
|
||||
if opts.Filesystem == nil {
|
||||
opts.Filesystem = afero.NewOsFs()
|
||||
}
|
||||
if opts.Tracer == nil {
|
||||
opts.Tracer = trace.NewNoopTracerProvider()
|
||||
if opts.TracerProvider == nil {
|
||||
opts.TracerProvider = trace.NewNoopTracerProvider()
|
||||
}
|
||||
if opts.Metrics == nil {
|
||||
reg := prometheus.NewRegistry()
|
||||
mets := NewMetrics(reg)
|
||||
opts.Metrics = &mets
|
||||
}
|
||||
|
||||
ctx, ctxCancel := context.WithCancel(context.Background())
|
||||
daemon := &Server{
|
||||
opts: opts,
|
||||
tracer: opts.Tracer.Tracer(tracing.TracerName),
|
||||
tracer: opts.TracerProvider.Tracer(tracing.TracerName),
|
||||
|
||||
clientDialer: clientDialer,
|
||||
|
||||
@@ -103,6 +111,42 @@ type Server struct {
|
||||
activeJob *runner.Runner
|
||||
}
|
||||
|
||||
type Metrics struct {
|
||||
Runner runner.Metrics
|
||||
}
|
||||
|
||||
func NewMetrics(reg prometheus.Registerer) Metrics {
|
||||
auto := promauto.With(reg)
|
||||
durationToFloatMs := func(d time.Duration) float64 {
|
||||
return float64(d.Milliseconds())
|
||||
}
|
||||
|
||||
return Metrics{
|
||||
Runner: runner.Metrics{
|
||||
ConcurrentJobs: auto.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Namespace: "coderd",
|
||||
Subsystem: "provisionerd",
|
||||
Name: "jobs_current",
|
||||
}, []string{"provisioner"}),
|
||||
JobTimings: auto.NewHistogramVec(prometheus.HistogramOpts{
|
||||
Namespace: "coderd",
|
||||
Subsystem: "provisionerd",
|
||||
Name: "job_timings_ms",
|
||||
Buckets: []float64{
|
||||
durationToFloatMs(1 * time.Second),
|
||||
durationToFloatMs(10 * time.Second),
|
||||
durationToFloatMs(30 * time.Second),
|
||||
durationToFloatMs(1 * time.Minute),
|
||||
durationToFloatMs(5 * time.Minute),
|
||||
durationToFloatMs(10 * time.Minute),
|
||||
durationToFloatMs(30 * time.Minute),
|
||||
durationToFloatMs(1 * time.Hour),
|
||||
},
|
||||
}, []string{"provisioner", "status"}),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Connect establishes a connection to coderd.
|
||||
func (p *Server) connect(ctx context.Context) {
|
||||
// An exponential back-off occurs when the connection is failing to dial.
|
||||
@@ -282,6 +326,7 @@ func (p *Server) acquireJob(ctx context.Context) {
|
||||
p.opts.UpdateInterval,
|
||||
p.opts.ForceCancelInterval,
|
||||
p.tracer,
|
||||
p.opts.Metrics.Runner,
|
||||
)
|
||||
|
||||
go p.activeJob.Run()
|
||||
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/google/uuid"
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/spf13/afero"
|
||||
"go.opentelemetry.io/otel/codes"
|
||||
semconv "go.opentelemetry.io/otel/semconv/v1.11.0"
|
||||
@@ -34,6 +35,7 @@ const (
|
||||
|
||||
type Runner struct {
|
||||
tracer trace.Tracer
|
||||
metrics Metrics
|
||||
job *proto.AcquiredJob
|
||||
sender JobUpdater
|
||||
logger slog.Logger
|
||||
@@ -65,6 +67,12 @@ type Runner struct {
|
||||
okToSend bool
|
||||
}
|
||||
|
||||
type Metrics struct {
|
||||
ConcurrentJobs *prometheus.GaugeVec
|
||||
// JobTimings also counts the total amount of jobs.
|
||||
JobTimings *prometheus.HistogramVec
|
||||
}
|
||||
|
||||
type JobUpdater interface {
|
||||
UpdateJob(ctx context.Context, in *proto.UpdateJobRequest) (*proto.UpdateJobResponse, error)
|
||||
FailJob(ctx context.Context, in *proto.FailedJob) error
|
||||
@@ -82,6 +90,7 @@ func NewRunner(
|
||||
updateInterval time.Duration,
|
||||
forceCancelInterval time.Duration,
|
||||
tracer trace.Tracer,
|
||||
metrics Metrics,
|
||||
) *Runner {
|
||||
m := new(sync.Mutex)
|
||||
|
||||
@@ -91,6 +100,7 @@ func NewRunner(
|
||||
|
||||
return &Runner{
|
||||
tracer: tracer,
|
||||
metrics: metrics,
|
||||
job: job,
|
||||
sender: updater,
|
||||
logger: logger.With(slog.F("job_id", job.JobId)),
|
||||
@@ -120,9 +130,22 @@ func NewRunner(
|
||||
// that goroutine on the context passed into Fail(), and it marks okToSend false to signal us here
|
||||
// that this function should not also send a terminal message.
|
||||
func (r *Runner) Run() {
|
||||
start := time.Now()
|
||||
ctx, span := r.startTrace(r.notStopped, tracing.FuncName())
|
||||
defer span.End()
|
||||
|
||||
concurrentGauge := r.metrics.ConcurrentJobs.WithLabelValues(r.job.Provisioner)
|
||||
concurrentGauge.Inc()
|
||||
defer func() {
|
||||
status := "success"
|
||||
if r.failedJob != nil {
|
||||
status = "failed"
|
||||
}
|
||||
|
||||
concurrentGauge.Dec()
|
||||
r.metrics.JobTimings.WithLabelValues(r.job.Provisioner, status).Observe(float64(time.Since(start).Milliseconds()))
|
||||
}()
|
||||
|
||||
r.mutex.Lock()
|
||||
defer r.mutex.Unlock()
|
||||
defer r.stop()
|
||||
|
||||
Reference in New Issue
Block a user