diff --git a/cli/testdata/coder_server_--help.golden b/cli/testdata/coder_server_--help.golden index 030a18f2bd01b..151ca8285fa6b 100644 --- a/cli/testdata/coder_server_--help.golden +++ b/cli/testdata/coder_server_--help.golden @@ -281,6 +281,14 @@ Configure the background chat processing daemon. Force chat debug logging on for every chat, bypassing the runtime admin and user opt-in settings. + --chat-stage-metrics off|basic|full, $CODER_CHAT_STAGE_METRICS (default: off) + How much of the chat lifecycle stage instrumentation to expose as + Prometheus metrics. "off" exposes none. "basic" records per-occurrence + durations for the wait, connect, and model-call stages and the + per-turn time partition by category. "full" adds every stage and the + per-turn stage distributions at a higher series count. Tracing spans + are unaffected. + CLIENT OPTIONS: These options change the behavior of how clients interact with the Coder. Clients include the Coder CLI, Coder Desktop, IDE extensions, and the web UI. diff --git a/cli/testdata/server-config.yaml.golden b/cli/testdata/server-config.yaml.golden index 20b3e71aba9ce..1ab52f6d9bd61 100644 --- a/cli/testdata/server-config.yaml.golden +++ b/cli/testdata/server-config.yaml.golden @@ -812,6 +812,13 @@ chat: # opt-in settings. # (default: false, type: bool) debugLoggingEnabled: false + # How much of the chat lifecycle stage instrumentation to expose as Prometheus + # metrics. "off" exposes none. "basic" records per-occurrence durations for the + # wait, connect, and model-call stages and the per-turn time partition by + # category. "full" adds every stage and the per-turn stage distributions at a + # higher series count. Tracing spans are unaffected. + # (default: off, type: enum[off\|basic\|full]) + stageMetrics: off # HTTPS URL to receive chat agent lifecycle hook events (plain HTTP requires # --chat-hook-allow-insecure). Hooks are disabled when unset. Requires the # agent-lifecycle-hooks experiment. diff --git a/coderd/apidoc/docs.go b/coderd/apidoc/docs.go index 9676e20e4fa3f..ea9a1e12cf70f 100644 --- a/coderd/apidoc/docs.go +++ b/coderd/apidoc/docs.go @@ -19535,6 +19535,9 @@ const docTemplate = `{ }, "hook_url": { "$ref": "#/definitions/serpent.URL" + }, + "stage_metrics": { + "type": "string" } } }, diff --git a/coderd/apidoc/swagger.json b/coderd/apidoc/swagger.json index 5bd07bf91c4c5..e6dc8cf58315b 100644 --- a/coderd/apidoc/swagger.json +++ b/coderd/apidoc/swagger.json @@ -17567,6 +17567,9 @@ }, "hook_url": { "$ref": "#/definitions/serpent.URL" + }, + "stage_metrics": { + "type": "string" } } }, diff --git a/coderd/coderd.go b/coderd/coderd.go index a6867c793bb8c..7b7db5d73477e 100644 --- a/coderd/coderd.go +++ b/coderd/coderd.go @@ -955,6 +955,7 @@ func New(options *Options) *API { HookDispatcher: hookDispatcher, UsageTracker: options.WorkspaceUsageTracker, PrometheusRegistry: options.PrometheusRegistry, + StageMetrics: codersdk.NewChatStageMetricsLevelFromString(options.DeploymentValues.AI.Chat.StageMetrics), AgentCapacityUnlock: options.ChatAgentCapacityUnlock, OIDCTokenSource: oidcMCPSrc, NotificationsEnqueuer: options.NotificationsEnqueuer, diff --git a/coderd/x/chatd/chatd.go b/coderd/x/chatd/chatd.go index 456031add59af..1254aaf13f8e0 100644 --- a/coderd/x/chatd/chatd.go +++ b/coderd/x/chatd/chatd.go @@ -3061,6 +3061,9 @@ type Config struct { AIBridgeTransportFactory *atomic.Pointer[aibridge.TransportFactory] Experiments codersdk.Experiments PrometheusRegistry prometheus.Registerer + // StageMetrics selects which chat lifecycle stage metric families + // are registered. The zero value means codersdk.ChatStageMetricsLevelOff. + StageMetrics codersdk.ChatStageMetricsLevel AgentCapacityUnlock AgentCapacityUnlock @@ -3179,7 +3182,9 @@ func New(ps pubsub.Pubsub, cfg Config) *Server { } var chatAutoArchiveRecords prometheus.Counter if cfg.PrometheusRegistry != nil { - p.metrics = chatloop.NewMetrics(cfg.PrometheusRegistry) + p.metrics = chatloop.NewMetricsWithOptions(cfg.PrometheusRegistry, chatloop.MetricsOptions{ + StageMetrics: cfg.StageMetrics, + }) chatAutoArchiveRecords = prometheus.NewCounter(prometheus.CounterOpts{ Namespace: "coderd", Subsystem: "chat_auto_archive", diff --git a/coderd/x/chatd/chatloop/metrics.go b/coderd/x/chatd/chatloop/metrics.go index 874a0ca52a0ae..a7db0fa90696f 100644 --- a/coderd/x/chatd/chatloop/metrics.go +++ b/coderd/x/chatd/chatloop/metrics.go @@ -3,12 +3,14 @@ package chatloop import ( "context" "errors" + "time" "charm.land/fantasy" "github.com/prometheus/client_golang/prometheus" "github.com/prometheus/client_golang/prometheus/promauto" "github.com/coder/coder/v2/coderd/x/chatd/chaterror" + "github.com/coder/coder/v2/codersdk" ) const ( @@ -23,8 +25,44 @@ const ( CompactionResultSuccess = "success" CompactionResultError = "error" CompactionResultTimeout = "timeout" + + // Label values for StageAnomaliesTotal. + // StageAnomalyNegativeElapsed is a stage whose measured duration + // was negative and was not observed. + StageAnomalyNegativeElapsed = "negative_elapsed" + // StageAnomalyInvertedWindow is a stage reconstructed from + // timestamps whose end preceded its start, or which lacked one of + // them, and was not observed. + StageAnomalyInvertedWindow = "inverted_window" ) +// basicStages is the set of stages observed into StageDurationSeconds +// at codersdk.ChatStageMetricsLevelBasic. It holds the wait, connect, and +// model-call stages; the stages that only describe chatd's own work +// inside a step (generation_step, prepare, thinking, compaction) are +// left to the full level. +var basicStages = map[string]struct{}{ + StageChatTurn: {}, + StageQueueWait: {}, + StageCapacityWait: {}, + StageAcquisition: {}, + StageMCPConnect: {}, + StageStream: {}, + StageTimeToFirstToken: {}, + StageProviderAttempt: {}, + StageToolCall: {}, + StageCommit: {}, + StageRetryBackoff: {}, +} + +// MetricsOptions configures which optional metric families NewMetrics +// registers. +type MetricsOptions struct { + // StageMetrics selects the chat lifecycle stage families to expose. + // Unrecognized or empty values mean codersdk.ChatStageMetricsLevelOff. + StageMetrics codersdk.ChatStageMetricsLevel +} + // Metrics holds Prometheus metrics for the chatd subsystem. type Metrics struct { Chats *prometheus.GaugeVec @@ -34,6 +72,9 @@ type Metrics struct { ToolResultTruncatedTotal *prometheus.CounterVec ToolErrorsTotal *prometheus.CounterVec TTFTSeconds *prometheus.HistogramVec + StageMetricsLevel *prometheus.GaugeVec + StageDurationSeconds *prometheus.HistogramVec + StageAnomaliesTotal *prometheus.CounterVec CompactionTotal *prometheus.CounterVec StepsTotal *prometheus.CounterVec StreamRetriesTotal *prometheus.CounterVec @@ -42,13 +83,31 @@ type Metrics struct { FindToolsEmptyTotal prometheus.Counter FindToolsMatchCount prometheus.Histogram FindToolsActivationsTotal prometheus.Counter + + // stageMetrics is the level the stage families were built for. + stageMetrics codersdk.ChatStageMetricsLevel } // NewMetrics creates a new Metrics instance registered with the -// given registerer. +// given registerer, with every stage metric family enabled. func NewMetrics(reg prometheus.Registerer) *Metrics { + return NewMetricsWithOptions(reg, MetricsOptions{StageMetrics: codersdk.ChatStageMetricsLevelFull}) +} + +// NewMetricsWithOptions creates a new Metrics instance registered with +// the given registerer. Stage families that opts leaves disabled are +// still constructed, against no registerer, so every recorder can be +// called at any level; they simply never appear in a scrape. +func NewMetricsWithOptions(reg prometheus.Registerer, opts MetricsOptions) *Metrics { + level := codersdk.NewChatStageMetricsLevelFromString(string(opts.StageMetrics)) factory := promauto.With(reg) - return &Metrics{ + // stageFactory registers the families exposed at basic and full. + stageFactory := factory + if level == codersdk.ChatStageMetricsLevelOff { + stageFactory = promauto.With(nil) + } + m := &Metrics{ + stageMetrics: level, Chats: factory.NewGaugeVec(prometheus.GaugeOpts{ Namespace: metricsNamespace, Subsystem: metricsSubsystem, @@ -95,6 +154,25 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Help: "Time-to-first-token: wall time from LLM request to first streamed chunk.", Buckets: []float64{0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 30, 60}, }, []string{"provider", "model"}), + StageMetricsLevel: factory.NewGaugeVec(prometheus.GaugeOpts{ + Namespace: metricsNamespace, + Subsystem: metricsSubsystem, + Name: "stage_metrics_level", + Help: "Always 1, labeled with the configured chat stage metrics level (off, basic, or full). Tells dashboards and alerts which stage and turn families this replica exposes.", + }, []string{"level"}), + StageDurationSeconds: stageFactory.NewHistogramVec(prometheus.HistogramOpts{ + Namespace: metricsNamespace, + Subsystem: metricsSubsystem, + Name: "stage_duration_seconds", + Help: "Wall time spent in each chat lifecycle stage. Stages overlap in wall time; this is a stage-time profile, not a partition of the turn. The scope label separates stages that run inside a chat turn from detached background work. The chat_kind label is empty for stages recorded without a known chat, and the model label is empty for stages that are not tied to a model call. At the basic stage metrics level only the wait, connect, and model-call stages are observed.", + Buckets: stageDurationBuckets(), + }, []string{"stage", "scope", "chat_kind", "model"}), + StageAnomaliesTotal: stageFactory.NewCounterVec(prometheus.CounterOpts{ + Namespace: metricsNamespace, + Subsystem: metricsSubsystem, + Name: "stage_anomalies_total", + Help: "Chat lifecycle stage observations dropped by reason. Reasons: negative_elapsed and inverted_window (clock inconsistencies).", + }, []string{"reason"}), CompactionTotal: factory.NewCounterVec(prometheus.CounterOpts{ Namespace: metricsNamespace, Subsystem: metricsSubsystem, @@ -145,6 +223,17 @@ func NewMetrics(reg prometheus.Registerer) *Metrics { Help: "Number of chat stream buffer events dropped due to the per-chat buffer cap.", }), } + m.StageMetricsLevel.WithLabelValues(string(level)).Set(1) + return m +} + +// stageDurationBuckets returns the duration buckets for the +// per-occurrence stage histogram: round boundaries from 50ms to 1h so +// alert thresholds land on bucket edges, denser between 1s and 10min +// where model calls, tool calls, and turns concentrate. Faster stages +// (a warm mcp_connect, commit) collapse into the first bucket. +func stageDurationBuckets() []float64 { + return []float64{0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 20, 30, 60, 120, 300, 600, 1800, 3600} } // NopMetrics returns a Metrics instance that discards all data. @@ -153,6 +242,37 @@ func NopMetrics() *Metrics { return NewMetrics(prometheus.NewRegistry()) } +// RecordStageDuration observes one chat lifecycle stage duration. +// chatKind is empty when the stage was recorded without a known chat, +// and model is empty when the stage is not tied to a model call. +// Negative durations are dropped and counted as an anomaly at every +// level. At the basic level, stages outside basicStages are dropped +// silently. No-op when m is nil. +func (m *Metrics) RecordStageDuration(stage, scope, chatKind, model string, elapsed time.Duration) { + if m == nil { + return + } + if elapsed < 0 { + m.RecordStageAnomaly(StageAnomalyNegativeElapsed) + return + } + if m.stageMetrics == codersdk.ChatStageMetricsLevelBasic { + if _, ok := basicStages[stage]; !ok { + return + } + } + m.StageDurationSeconds.WithLabelValues(stage, scope, chatKind, model).Observe(elapsed.Seconds()) +} + +// RecordStageAnomaly counts a stage observation that was dropped, by +// reason. No-op when m is nil. +func (m *Metrics) RecordStageAnomaly(reason string) { + if m == nil { + return + } + m.StageAnomaliesTotal.WithLabelValues(reason).Inc() +} + // RecordCompaction classifies and records a compaction attempt. // It is a no-op when m is nil. func (m *Metrics) RecordCompaction(provider, model string, compacted bool, err error) { diff --git a/coderd/x/chatd/chatloop/stage.go b/coderd/x/chatd/chatloop/stage.go new file mode 100644 index 0000000000000..9fb8e41c7fb5b --- /dev/null +++ b/coderd/x/chatd/chatloop/stage.go @@ -0,0 +1,408 @@ +package chatloop + +import ( + "context" + "time" + + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/codes" + "go.opentelemetry.io/otel/trace" + "go.opentelemetry.io/otel/trace/noop" + + "github.com/coder/quartz" +) + +// Stage names. Every value is both a span name and the `stage` label +// value recorded on stage_duration_seconds. +const ( + StageChatTurn = "chat_turn" + StageQueueWait = "queue_wait" + StageCapacityWait = "capacity_wait" + StageAcquisition = "acquisition" + StageGenerationStep = "generation_step" + StagePrepare = "prepare" + StageMCPConnect = "mcp_connect" + StageProviderAttempt = "provider_attempt" + StageStream = "stream" + StageTimeToFirstToken = "time_to_first_token" + StageThinking = "thinking" + StageToolCall = "tool_call" + StageCommit = "commit" + StageCompaction = "compaction" + StageRetryBackoff = "retry_backoff" +) + +// Span attribute keys. Keys are lowercase snake_case and shared by +// every stage that carries the value. +const ( + AttrProvider = "provider" + AttrModel = "model" + AttrReasoningEffort = "reasoning_effort" + AttrChatID = "chat_id" + AttrChatKind = "chat_kind" + AttrGenerationAttempt = "generation_attempt" + AttrGenerationAction = "generation_action" + AttrToolName = "tool_name" + AttrHTTPStatusCode = "http_status_code" + AttrHTTPMethod = "http_method" + AttrHTTPHost = "http_host" + AttrCompactionSource = "compaction_source" + AttrScope = "scope" +) + +// Scope values. A stage is turn scoped when it runs inside a chat +// turn's trace, and background scoped when it runs on work detached +// from the turn. +const ( + ScopeTurn = "turn" + ScopeBackground = "background" +) + +// Chat kind attribute values. +const ( + ChatKindRoot = "root" + ChatKindSubagent = "subagent" +) + +// tracerName is the instrumentation scope reported on chatd spans. +const tracerName = "chatd" + +// StageTracer emits one span and one stage_duration_seconds +// observation per chat lifecycle stage. Both are produced from the +// same call so span and histogram durations cannot diverge. +// +// A nil *StageTracer is usable and discards everything. +type StageTracer struct { + tracer trace.Tracer + metrics *Metrics + // clock is the time source for the stage windows measured here. + // Stages recorded from explicit timestamps do not use it. + clock quartz.Clock +} + +// NewStageTracer builds a stage tracer from a tracer provider and the +// chatd metrics. A nil provider falls back to a no-op tracer and nil +// metrics to a discarding registry, so callers without tracing or +// metrics configured still get a usable tracer. +func NewStageTracer(provider trace.TracerProvider, metrics *Metrics) *StageTracer { + if provider == nil { + provider = noop.NewTracerProvider() + } + if metrics == nil { + metrics = NopMetrics() + } + return &StageTracer{ + tracer: provider.Tracer(tracerName), + metrics: metrics, + clock: quartz.NewReal(), + } +} + +// NopStageTracer returns a stage tracer that discards spans and +// metrics. +func NopStageTracer() *StageTracer { + return NewStageTracer(nil, nil) +} + +func (t *StageTracer) otelTracer() trace.Tracer { + if t == nil || t.tracer == nil { + return noop.NewTracerProvider().Tracer(tracerName) + } + return t.tracer +} + +// Now returns the current time from the tracer's clock, the time +// source for the stage windows it measures. +func (t *StageTracer) Now() time.Time { + if t == nil || t.clock == nil { + return time.Now() + } + return t.clock.Now() +} + +// StageModel identifies the model a stage ran against. Both fields +// are empty for stages that run before a model is resolved, such as +// the queue and capacity waits. Effort is the effective reasoning +// effort sent to the provider, empty when the model config sets none. +type StageModel struct { + Model string + Effort string +} + +// attributes returns the span attributes for the identity, omitting +// the ones that are unknown. +func (m StageModel) attributes() []attribute.KeyValue { + attrs := make([]attribute.KeyValue, 0, 2) + if m.Model != "" { + attrs = append(attrs, attribute.String(AttrModel, m.Model)) + } + if m.Effort != "" { + attrs = append(attrs, attribute.String(AttrReasoningEffort, m.Effort)) + } + return attrs +} + +// StageSpan is an in-flight stage. End must be called exactly once; +// the duration observation happens there. +type StageSpan struct { + tracer *StageTracer + stage string + scope string + chatKind string + model StageModel + span trace.Span + start time.Time + ended bool +} + +// stageScopeKey keys the stage scope carried by a context. It is +// private so the scope can only be set through ContextWithScope. +type stageScopeKey struct{} + +// stageChatKindKey keys the chat kind carried by a context. It is +// private so the chat kind can only be set through +// ContextWithChatKind. +type stageChatKindKey struct{} + +// ContextWithScope returns ctx carrying scope for the stages started +// on it. The scope is a plain context value rather than a property of +// the span in ctx, so it survives configurations where spans are not +// recorded, such as a no-op tracer provider. +func ContextWithScope(ctx context.Context, scope string) context.Context { + return context.WithValue(ctx, stageScopeKey{}, scope) +} + +// ContextWithChatKind returns ctx carrying kind for the stages started +// on it. Callers that hold the chat row set it once, on the context a +// turn runs on or on the context of a single stage recorded outside a +// turn, and every stage derived from that context carries the value. +func ContextWithChatKind(ctx context.Context, kind string) context.Context { + return context.WithValue(ctx, stageChatKindKey{}, kind) +} + +// scopeFromContext reads the scope ContextWithScope put on ctx. +// Contexts with no scope are background scoped, so work detached from +// a turn is kept out of the turn profile. +func scopeFromContext(ctx context.Context) string { + if scope, ok := ctx.Value(stageScopeKey{}).(string); ok && scope != "" { + return scope + } + return ScopeBackground +} + +// chatKindFromContext reads the chat kind ContextWithChatKind put on +// ctx. It is empty when the stage runs without a known chat, which +// keeps the label present but unset rather than guessing a kind. +func chatKindFromContext(ctx context.Context) string { + kind, _ := ctx.Value(stageChatKindKey{}).(string) + return kind +} + +// Start begins a stage span as a child of the span in ctx and returns +// a context carrying it. The stage takes the scope and chat kind on +// ctx, so stages started on a context with no scope are background +// scoped. +func (t *StageTracer) Start( + ctx context.Context, + stage string, + attrs ...attribute.KeyValue, +) (context.Context, *StageSpan) { + return t.startSpan(ctx, stage, scopeFromContext(ctx), time.Time{}, + []trace.SpanStartOption{trace.WithAttributes(attrs...)}) +} + +// StartRoot begins a stage span in its own trace, ignoring any span +// in ctx. links records the relationship to the originating span +// context instead of making that span the parent, so the stage's +// trace stays scoped to the chat turn. The span opens a turn, so it +// is turn scoped regardless of what ctx carries. +func (t *StageTracer) StartRoot( + ctx context.Context, + stage string, + links []trace.Link, + attrs ...attribute.KeyValue, +) (context.Context, *StageSpan) { + return t.StartRootAt(ctx, stage, time.Time{}, links, attrs...) +} + +// StartRootAt begins a root stage span that started at an earlier, +// already known instant. The span timestamp and the recorded duration +// both run from start, so stages reconstructed inside the span still +// fall within it. A zero start means the span begins now. +func (t *StageTracer) StartRootAt( + ctx context.Context, + stage string, + start time.Time, + links []trace.Link, + attrs ...attribute.KeyValue, +) (context.Context, *StageSpan) { + return t.startSpan(ctx, stage, ScopeTurn, start, []trace.SpanStartOption{ + trace.WithNewRoot(), + trace.WithLinks(links...), + trace.WithAttributes(attrs...), + }) +} + +func (t *StageTracer) startSpan( + ctx context.Context, + stage string, + scope string, + start time.Time, + opts []trace.SpanStartOption, +) (context.Context, *StageSpan) { + if t == nil { + return ctx, nil + } + now := t.Now() + if start.IsZero() || start.After(now) { + start = now + } else { + opts = append(opts, trace.WithTimestamp(start)) + } + chatKind := chatKindFromContext(ctx) + opts = append(opts, trace.WithAttributes(stageIdentityAttributes(scope, chatKind)...)) + ctx, span := t.otelTracer().Start(ContextWithScope(ctx, scope), stage, opts...) + return ctx, &StageSpan{ + tracer: t, + stage: stage, + scope: scope, + chatKind: chatKind, + span: span, + start: start, + } +} + +// stageIdentityAttributes returns the attributes every stage span +// carries. An unknown chat kind is omitted from the span, where an +// absent attribute reads better than an empty one. +func stageIdentityAttributes(scope, chatKind string) []attribute.KeyValue { + attrs := []attribute.KeyValue{attribute.String(AttrScope, scope)} + if chatKind != "" { + attrs = append(attrs, attribute.String(AttrChatKind, chatKind)) + } + return attrs +} + +// SetAttributes adds attributes to the stage span. It is a no-op +// after End. +func (s *StageSpan) SetAttributes(attrs ...attribute.KeyValue) { + if s == nil || s.ended { + return + } + s.span.SetAttributes(attrs...) +} + +// SetModel records the model identity on the span and on the +// duration observation End makes, for stages that learn the model +// after they start. +func (s *StageSpan) SetModel(model StageModel) { + if s == nil || s.ended { + return + } + s.model = model + s.span.SetAttributes(model.attributes()...) +} + +// SpanContext returns the span context of the stage span, which is +// invalid when tracing is not configured. +func (s *StageSpan) SpanContext() trace.SpanContext { + if s == nil { + return trace.SpanContext{} + } + return s.span.SpanContext() +} + +// End closes the stage span, records its duration, and marks the span +// as errored when err is non-nil. Calls after the first are ignored so +// a deferred End cannot double-count a stage. +func (s *StageSpan) End(err error) { + if elapsed, ok := s.closeSpan(err); ok { + s.tracer.observe(s.stage, s.scope, s.chatKind, s.model, elapsed) + } +} + +// EndWithoutObservation closes the stage span exactly as End does but +// makes no duration observation. It is for stages whose window is only +// comparable across runs when it completed, so a truncated window +// would skew the histogram while the span still needs to report the +// failure. +func (s *StageSpan) EndWithoutObservation(err error) { + s.closeSpan(err) +} + +// closeSpan ends the span and returns the window it covered. ok is +// false for a nil span and for calls after the first, so a deferred +// end cannot double-count a stage. +func (s *StageSpan) closeSpan(err error) (elapsed time.Duration, ok bool) { + if s == nil || s.ended { + return 0, false + } + s.ended = true + elapsed = s.tracer.Now().Sub(s.start) + if err != nil { + s.span.RecordError(err) + s.span.SetStatus(codes.Error, err.Error()) + } + s.span.End() + return elapsed, true +} + +// Record emits an already-finished stage span with explicit start and +// end timestamps. It is for stages whose boundaries are only known +// after the fact, such as durations reconstructed from persisted +// timestamps. The stage takes the scope and chat kind on ctx. +// Non-positive or unset windows are dropped. +func (t *StageTracer) Record( + ctx context.Context, + stage string, + model StageModel, + start, end time.Time, + err error, + attrs ...attribute.KeyValue, +) { + t.RecordAs(ctx, stage, scopeFromContext(ctx), model, start, end, err, attrs...) +} + +// RecordAs is Record with an explicit scope, for stages recorded on a +// context that does not carry the scope they belong to. +func (t *StageTracer) RecordAs( + ctx context.Context, + stage string, + scope string, + model StageModel, + start, end time.Time, + err error, + attrs ...attribute.KeyValue, +) { + if start.IsZero() || end.IsZero() || end.Before(start) { + t.recordAnomaly(StageAnomalyInvertedWindow) + return + } + chatKind := chatKindFromContext(ctx) + _, span := t.otelTracer().Start(ctx, stage, + trace.WithTimestamp(start), + trace.WithAttributes(attrs...), + trace.WithAttributes(model.attributes()...), + trace.WithAttributes(stageIdentityAttributes(scope, chatKind)...), + ) + if err != nil { + span.RecordError(err) + span.SetStatus(codes.Error, err.Error()) + } + span.End(trace.WithTimestamp(end)) + t.observe(stage, scope, chatKind, model, end.Sub(start)) +} + +func (t *StageTracer) recordAnomaly(reason string) { + if t == nil || t.metrics == nil { + return + } + t.metrics.RecordStageAnomaly(reason) +} + +func (t *StageTracer) observe(stage, scope, chatKind string, model StageModel, elapsed time.Duration) { + if t == nil || t.metrics == nil { + return + } + t.metrics.RecordStageDuration(stage, scope, chatKind, model.Model, elapsed) +} diff --git a/coderd/x/chatd/chatloop/stage_test.go b/coderd/x/chatd/chatloop/stage_test.go new file mode 100644 index 0000000000000..ff626699bb906 --- /dev/null +++ b/coderd/x/chatd/chatloop/stage_test.go @@ -0,0 +1,652 @@ +package chatloop_test + +import ( + "context" + "slices" + "testing" + "time" + + "github.com/prometheus/client_golang/prometheus" + dto "github.com/prometheus/client_model/go" + "github.com/stretchr/testify/require" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/codes" + sdktrace "go.opentelemetry.io/otel/sdk/trace" + "go.opentelemetry.io/otel/sdk/trace/tracetest" + "go.opentelemetry.io/otel/trace" + "golang.org/x/xerrors" + + "github.com/coder/coder/v2/coderd/x/chatd/chatloop" + "github.com/coder/coder/v2/codersdk" +) + +// stageFixture wires a stage tracer to an in-memory span recorder and +// a private metrics registry. +type stageFixture struct { + tracer *chatloop.StageTracer + spans *tracetest.SpanRecorder + registry *prometheus.Registry +} + +func newStageFixture(t *testing.T) stageFixture { + t.Helper() + recorder := tracetest.NewSpanRecorder() + provider := sdktrace.NewTracerProvider(sdktrace.WithSpanProcessor(recorder)) + t.Cleanup(func() { + // The test context is already canceled during cleanup, so the + // flush uses a fresh one. + require.NoError(t, provider.Shutdown(context.Background())) + }) + registry := prometheus.NewRegistry() + return stageFixture{ + tracer: chatloop.NewStageTracer(provider, chatloop.NewMetrics(registry)), + spans: recorder, + registry: registry, + } +} + +// stageKey identifies one stage_duration_seconds series. +type stageKey struct { + stage string + scope string + chatKind string + model string +} + +// stageObservations returns the observation count per stage series +// recorded on coderd_chatd_stage_duration_seconds. +func (f stageFixture) stageObservations(t *testing.T) map[stageKey]uint64 { + t.Helper() + families, err := f.registry.Gather() + require.NoError(t, err) + counts := map[stageKey]uint64{} + for _, family := range families { + if family.GetName() != "coderd_chatd_stage_duration_seconds" { + continue + } + for _, metric := range family.GetMetric() { + key := stageKey{ + stage: labelValue(metric, "stage"), + scope: labelValue(metric, "scope"), + chatKind: labelValue(metric, "chat_kind"), + model: labelValue(metric, "model"), + } + counts[key] = metric.GetHistogram().GetSampleCount() + } + } + return counts +} + +func (f stageFixture) stageSum(t *testing.T, stage string) float64 { + t.Helper() + families, err := f.registry.Gather() + require.NoError(t, err) + for _, family := range families { + if family.GetName() != "coderd_chatd_stage_duration_seconds" { + continue + } + for _, metric := range family.GetMetric() { + if labelValue(metric, "stage") == stage { + return metric.GetHistogram().GetSampleSum() + } + } + } + t.Fatalf("stage %q was not recorded", stage) + return 0 +} + +func labelValue(metric *dto.Metric, name string) string { + for _, label := range metric.GetLabel() { + if label.GetName() == name { + return label.GetValue() + } + } + return "" +} + +func TestStageTracerStart(t *testing.T) { + t.Parallel() + + t.Run("RecordsSpanAndMetricOnce", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + _, span := fixture.tracer.Start(t.Context(), chatloop.StageCommit, + attribute.String(chatloop.AttrProvider, "anthropic"), + ) + span.End(nil) + span.End(nil) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + require.Equal(t, chatloop.StageCommit, ended[0].Name()) + require.Equal(t, codes.Unset, ended[0].Status().Code) + require.Contains(t, ended[0].Attributes(), + attribute.String(chatloop.AttrProvider, "anthropic")) + require.Contains(t, ended[0].Attributes(), + attribute.String(chatloop.AttrScope, chatloop.ScopeBackground)) + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageCommit, scope: chatloop.ScopeBackground}: 1, + }, fixture.stageObservations(t)) + }) + + t.Run("MarksErrorStatus", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + _, span := fixture.tracer.Start(t.Context(), chatloop.StageStream) + span.End(xerrors.New("stream failed")) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + require.Equal(t, codes.Error, ended[0].Status().Code) + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageStream, scope: chatloop.ScopeBackground}: 1, + }, fixture.stageObservations(t)) + }) + + t.Run("NestsUnderParent", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + parentCtx, parent := fixture.tracer.Start(t.Context(), chatloop.StageGenerationStep) + _, child := fixture.tracer.Start(parentCtx, chatloop.StagePrepare) + child.End(nil) + parent.End(nil) + + ended := fixture.spans.Ended() + require.Len(t, ended, 2) + require.Equal(t, chatloop.StagePrepare, ended[0].Name()) + require.Equal(t, ended[1].SpanContext().SpanID(), ended[0].Parent().SpanID()) + require.Equal(t, ended[1].SpanContext().TraceID(), ended[0].SpanContext().TraceID()) + }) +} + +func TestStageTracerStartRoot(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + outerCtx, outer := fixture.tracer.Start(t.Context(), chatloop.StageGenerationStep) + _, root := fixture.tracer.StartRoot(outerCtx, chatloop.StageChatTurn, + []trace.Link{{SpanContext: trace.SpanContextFromContext(outerCtx)}}, + ) + root.End(nil) + outer.End(nil) + + ended := fixture.spans.Ended() + require.Len(t, ended, 2) + turn, step := ended[0], ended[1] + require.Equal(t, chatloop.StageChatTurn, turn.Name()) + require.False(t, turn.Parent().IsValid()) + require.NotEqual(t, step.SpanContext().TraceID(), turn.SpanContext().TraceID()) + require.Len(t, turn.Links(), 1) + require.Equal(t, step.SpanContext().SpanID(), turn.Links()[0].SpanContext.SpanID()) +} + +func TestStageTracerStartRootAt(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + start := time.Now().Add(-45 * time.Second) + turnCtx, turn := fixture.tracer.StartRootAt(t.Context(), chatloop.StageChatTurn, start, nil) + fixture.tracer.Record(turnCtx, chatloop.StageAcquisition, chatloop.StageModel{}, start, start.Add(time.Second), nil) + turn.End(nil) + + var chatTurn, acquisition sdktrace.ReadOnlySpan + for _, span := range fixture.spans.Ended() { + switch span.Name() { + case chatloop.StageChatTurn: + chatTurn = span + case chatloop.StageAcquisition: + acquisition = span + } + } + require.NotNil(t, chatTurn) + require.NotNil(t, acquisition) + require.Equal(t, start.UTC(), chatTurn.StartTime().UTC()) + require.False(t, acquisition.StartTime().Before(chatTurn.StartTime())) + require.Equal(t, chatTurn.SpanContext().SpanID(), acquisition.Parent().SpanID()) + // A root turn span opens a turn, and stages recorded inside it + // inherit the turn scope from its context. + require.Contains(t, chatTurn.Attributes(), + attribute.String(chatloop.AttrScope, chatloop.ScopeTurn)) + require.Contains(t, acquisition.Attributes(), + attribute.String(chatloop.AttrScope, chatloop.ScopeTurn)) + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageChatTurn, scope: chatloop.ScopeTurn}: 1, + {stage: chatloop.StageAcquisition, scope: chatloop.ScopeTurn}: 1, + }, fixture.stageObservations(t)) + // The histogram observation runs from the explicit start, so it + // covers the same window the span reports. + require.GreaterOrEqual(t, fixture.stageSum(t, chatloop.StageChatTurn), 45.0) +} + +func TestStageTracerScope(t *testing.T) { + t.Parallel() + + t.Run("TurnWorkStaysInTurnScope", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + turnCtx, turn := fixture.tracer.StartRoot(t.Context(), chatloop.StageChatTurn, nil) + stepCtx, step := fixture.tracer.Start(turnCtx, chatloop.StageGenerationStep) + _, attempt := fixture.tracer.Start(stepCtx, chatloop.StageProviderAttempt) + attempt.End(nil) + step.End(nil) + turn.End(nil) + + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageChatTurn, scope: chatloop.ScopeTurn}: 1, + {stage: chatloop.StageGenerationStep, scope: chatloop.ScopeTurn}: 1, + {stage: chatloop.StageProviderAttempt, scope: chatloop.ScopeTurn}: 1, + }, fixture.stageObservations(t)) + }) + + t.Run("DetachedWorkIsBackgroundScope", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + turnCtx, turn := fixture.tracer.StartRoot(t.Context(), chatloop.StageChatTurn, nil) + // Background work detaches from the turn by stripping the span + // and marking the context background scoped. + detachedCtx := chatloop.ContextWithScope( + trace.ContextWithSpanContext(turnCtx, trace.SpanContext{}), + chatloop.ScopeBackground, + ) + _, attempt := fixture.tracer.Start(detachedCtx, chatloop.StageProviderAttempt) + attempt.End(nil) + turn.End(nil) + + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageChatTurn, scope: chatloop.ScopeTurn}: 1, + {stage: chatloop.StageProviderAttempt, scope: chatloop.ScopeBackground}: 1, + }, fixture.stageObservations(t)) + + for _, span := range fixture.spans.Ended() { + if span.Name() != chatloop.StageProviderAttempt { + continue + } + require.False(t, span.Parent().IsValid()) + require.Contains(t, span.Attributes(), + attribute.String(chatloop.AttrScope, chatloop.ScopeBackground)) + } + }) + + t.Run("RecordAsOverridesContextScope", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + start := time.Now().Add(-5 * time.Second) + fixture.tracer.RecordAs(t.Context(), chatloop.StageCapacityWait, chatloop.ScopeTurn, + chatloop.StageModel{}, start, start.Add(time.Second), nil) + + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageCapacityWait, scope: chatloop.ScopeTurn}: 1, + }, fixture.stageObservations(t)) + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + require.Contains(t, ended[0].Attributes(), + attribute.String(chatloop.AttrScope, chatloop.ScopeTurn)) + }) +} + +func TestStageTracerChatKind(t *testing.T) { + t.Parallel() + + t.Run("InheritedByDerivedStages", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + ctx := chatloop.ContextWithChatKind(t.Context(), chatloop.ChatKindSubagent) + turnCtx, turn := fixture.tracer.StartRoot(ctx, chatloop.StageChatTurn, nil) + stepCtx, step := fixture.tracer.Start(turnCtx, chatloop.StageGenerationStep) + start := time.Now().Add(-time.Second) + fixture.tracer.Record(stepCtx, chatloop.StageToolCall, chatloop.StageModel{}, start, time.Now(), nil) + step.End(nil) + turn.End(nil) + + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageChatTurn, scope: chatloop.ScopeTurn, chatKind: chatloop.ChatKindSubagent}: 1, + {stage: chatloop.StageGenerationStep, scope: chatloop.ScopeTurn, chatKind: chatloop.ChatKindSubagent}: 1, + {stage: chatloop.StageToolCall, scope: chatloop.ScopeTurn, chatKind: chatloop.ChatKindSubagent}: 1, + }, fixture.stageObservations(t)) + + for _, span := range fixture.spans.Ended() { + require.Contains(t, span.Attributes(), + attribute.String(chatloop.AttrChatKind, chatloop.ChatKindSubagent)) + } + }) + + t.Run("UnknownChatLeavesLabelEmpty", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + start := time.Now().Add(-time.Second) + fixture.tracer.RecordAs(t.Context(), chatloop.StageCapacityWait, chatloop.ScopeTurn, + chatloop.StageModel{}, start, time.Now(), nil) + + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageCapacityWait, scope: chatloop.ScopeTurn}: 1, + }, fixture.stageObservations(t)) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + for _, attr := range ended[0].Attributes() { + require.NotEqual(t, chatloop.AttrChatKind, string(attr.Key)) + } + }) +} + +func TestStageTracerModelLabels(t *testing.T) { + t.Parallel() + + model := chatloop.StageModel{Model: "claude-sonnet-4-5", Effort: "high"} + + t.Run("SetModelLabelsSpanAndDuration", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + turnCtx, turn := fixture.tracer.StartRoot(t.Context(), chatloop.StageChatTurn, nil) + _, step := fixture.tracer.Start(turnCtx, chatloop.StageGenerationStep) + // The step learns its model only after preparation resolves it. + step.SetModel(model) + step.End(nil) + turn.End(nil) + + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageChatTurn, scope: chatloop.ScopeTurn}: 1, + { + stage: chatloop.StageGenerationStep, + scope: chatloop.ScopeTurn, + model: model.Model, + }: 1, + }, fixture.stageObservations(t)) + + for _, span := range fixture.spans.Ended() { + if span.Name() != chatloop.StageGenerationStep { + continue + } + require.Contains(t, span.Attributes(), + attribute.String(chatloop.AttrModel, model.Model)) + require.Contains(t, span.Attributes(), + attribute.String(chatloop.AttrReasoningEffort, model.Effort)) + } + }) + + t.Run("RecordCarriesModelLabels", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + start := time.Now().Add(-time.Second) + fixture.tracer.Record(t.Context(), chatloop.StageThinking, model, start, time.Now(), nil) + + require.Equal(t, map[stageKey]uint64{{ + stage: chatloop.StageThinking, + scope: chatloop.ScopeBackground, + model: model.Model, + }: 1}, fixture.stageObservations(t)) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + require.Contains(t, ended[0].Attributes(), + attribute.String(chatloop.AttrModel, model.Model)) + require.Contains(t, ended[0].Attributes(), + attribute.String(chatloop.AttrReasoningEffort, model.Effort)) + }) + + t.Run("UnknownIdentityUsesEmptyLabels", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + start := time.Now().Add(-time.Second) + fixture.tracer.Record(t.Context(), chatloop.StageQueueWait, chatloop.StageModel{}, start, time.Now(), nil) + + require.Equal(t, map[stageKey]uint64{{ + stage: chatloop.StageQueueWait, + scope: chatloop.ScopeBackground, + }: 1}, fixture.stageObservations(t)) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + for _, attr := range ended[0].Attributes() { + require.NotEqual(t, chatloop.AttrModel, string(attr.Key)) + require.NotEqual(t, chatloop.AttrReasoningEffort, string(attr.Key)) + } + }) + + t.Run("ModelWithoutEffort", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + _, span := fixture.tracer.Start(t.Context(), chatloop.StageStream) + span.SetModel(chatloop.StageModel{Model: "gpt-5"}) + span.End(nil) + + require.Equal(t, map[stageKey]uint64{{ + stage: chatloop.StageStream, + scope: chatloop.ScopeBackground, + model: "gpt-5", + }: 1}, fixture.stageObservations(t)) + }) +} + +func TestStageTracerRecord(t *testing.T) { + t.Parallel() + + t.Run("UsesExplicitTimestamps", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + start := time.Now().Add(-90 * time.Second) + end := start.Add(30 * time.Second) + fixture.tracer.Record(t.Context(), chatloop.StageQueueWait, chatloop.StageModel{}, start, end, nil, + attribute.String(chatloop.AttrChatKind, chatloop.ChatKindRoot), + ) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + require.Equal(t, chatloop.StageQueueWait, ended[0].Name()) + require.Equal(t, start.UTC(), ended[0].StartTime().UTC()) + require.Equal(t, end.UTC(), ended[0].EndTime().UTC()) + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageQueueWait, scope: chatloop.ScopeBackground}: 1, + }, fixture.stageObservations(t)) + require.InDelta(t, 30, fixture.stageSum(t, chatloop.StageQueueWait), 0.001) + }) + + t.Run("DropsUnusableWindows", func(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + now := time.Now() + fixture.tracer.Record(t.Context(), chatloop.StageAcquisition, chatloop.StageModel{}, time.Time{}, now, nil) + fixture.tracer.Record(t.Context(), chatloop.StageAcquisition, chatloop.StageModel{}, now, time.Time{}, nil) + fixture.tracer.Record(t.Context(), chatloop.StageAcquisition, chatloop.StageModel{}, now, now.Add(-time.Second), nil) + + require.Empty(t, fixture.spans.Ended()) + require.Empty(t, fixture.stageObservations(t)) + }) +} + +func TestStageSpanEndWithoutObservation(t *testing.T) { + t.Parallel() + fixture := newStageFixture(t) + + _, span := fixture.tracer.Start(t.Context(), chatloop.StageTimeToFirstToken) + span.EndWithoutObservation(xerrors.New("stream ended before the first token")) + // The first end wins, so a later End cannot revive the + // observation. + span.End(nil) + + ended := fixture.spans.Ended() + require.Len(t, ended, 1) + require.Equal(t, chatloop.StageTimeToFirstToken, ended[0].Name()) + require.Equal(t, codes.Error, ended[0].Status().Code) + require.Empty(t, fixture.stageObservations(t)) +} + +func TestStageDurationBuckets(t *testing.T) { + t.Parallel() + registry := prometheus.NewRegistry() + metrics := chatloop.NewMetrics(registry) + metrics.RecordStageDuration(chatloop.StageChatTurn, chatloop.ScopeTurn, chatloop.ChatKindRoot, "", 45*time.Minute) + + families, err := registry.Gather() + require.NoError(t, err) + var buckets []*dto.Bucket + for _, family := range families { + if family.GetName() != "coderd_chatd_stage_duration_seconds" { + continue + } + require.Len(t, family.GetMetric(), 1) + buckets = family.GetMetric()[0].GetHistogram().GetBucket() + } + require.Len(t, buckets, 16) + // Alert thresholds are written against these edges, so they must + // be round numbers rather than a generated ladder. + for _, want := range []float64{1, 5, 10, 30, 60, 300, 600, 3600} { + require.True(t, slices.ContainsFunc(buckets, func(b *dto.Bucket) bool { + return b.GetUpperBound() == want + }), "bucket edge %v missing", want) + } + // A 45 minute turn lands in the 1800-3600 bucket, not the overflow. + for _, bucket := range buckets { + if bucket.GetUpperBound() < 3600 { + require.Zero(t, bucket.GetCumulativeCount(), "le=%v", bucket.GetUpperBound()) + } else { + require.Equal(t, uint64(1), bucket.GetCumulativeCount(), "le=%v", bucket.GetUpperBound()) + } + } +} + +// TestStageMetricsLevels covers which stage families each +// --chat-stage-metrics level exposes. Every level must accept every +// recorder call, since the tracer does not know the level. +func TestStageMetricsLevels(t *testing.T) { + t.Parallel() + + record := func(m *chatloop.Metrics) { + m.RecordStageDuration(chatloop.StageTimeToFirstToken, chatloop.ScopeTurn, chatloop.ChatKindRoot, "m", time.Second) + m.RecordStageDuration(chatloop.StagePrepare, chatloop.ScopeTurn, chatloop.ChatKindRoot, "m", time.Second) + m.RecordStageDuration(chatloop.StageQueueWait, chatloop.ScopeTurn, chatloop.ChatKindRoot, "", time.Second) + m.RecordStageDuration(chatloop.StageCommit, chatloop.ScopeTurn, chatloop.ChatKindRoot, "", -time.Second) + m.RecordStageAnomaly(chatloop.StageAnomalyInvertedWindow) + } + + tests := []struct { + level codersdk.ChatStageMetricsLevel + wantLevel string + wantStages []string + wantFamily bool + }{ + {level: codersdk.ChatStageMetricsLevelOff, wantLevel: "off"}, + { + level: codersdk.ChatStageMetricsLevelBasic, wantLevel: "basic", wantFamily: true, + wantStages: []string{chatloop.StageQueueWait, chatloop.StageTimeToFirstToken}, + }, + { + level: codersdk.ChatStageMetricsLevelFull, wantLevel: "full", wantFamily: true, + wantStages: []string{chatloop.StagePrepare, chatloop.StageQueueWait, chatloop.StageTimeToFirstToken}, + }, + // Case is ignored; unknown and empty values fall back to off. + { + level: "FULL", wantLevel: "full", wantFamily: true, + wantStages: []string{chatloop.StagePrepare, chatloop.StageQueueWait, chatloop.StageTimeToFirstToken}, + }, + {level: "", wantLevel: "off"}, + {level: "verbose", wantLevel: "off"}, + } + for _, tt := range tests { + t.Run(string(tt.level), func(t *testing.T) { + t.Parallel() + registry := prometheus.NewRegistry() + metrics := chatloop.NewMetricsWithOptions(registry, chatloop.MetricsOptions{StageMetrics: tt.level}) + record(metrics) + + families, err := registry.Gather() + require.NoError(t, err) + byName := map[string]*dto.MetricFamily{} + for _, family := range families { + byName[family.GetName()] = family + } + + level := byName["coderd_chatd_stage_metrics_level"] + require.NotNil(t, level) + require.Len(t, level.GetMetric(), 1) + require.Equal(t, tt.wantLevel, labelValue(level.GetMetric()[0], "level")) + + _, hasDurations := byName["coderd_chatd_stage_duration_seconds"] + _, hasAnomalies := byName["coderd_chatd_stage_anomalies_total"] + require.Equal(t, tt.wantFamily, hasDurations) + require.Equal(t, tt.wantFamily, hasAnomalies) + if !tt.wantFamily { + return + } + + var stages []string + for _, metric := range byName["coderd_chatd_stage_duration_seconds"].GetMetric() { + stages = append(stages, labelValue(metric, "stage")) + } + slices.Sort(stages) + require.Equal(t, tt.wantStages, stages) + + // The negative commit and the inverted window count at + // every level that exposes the family, including basic + // where commit itself would have been observed. + anomalies := map[string]float64{} + for _, metric := range byName["coderd_chatd_stage_anomalies_total"].GetMetric() { + anomalies[labelValue(metric, "reason")] = metric.GetCounter().GetValue() + } + require.Equal(t, map[string]float64{ + chatloop.StageAnomalyNegativeElapsed: 1, + chatloop.StageAnomalyInvertedWindow: 1, + }, anomalies) + }) + } +} + +func TestStageTracerWithoutProvider(t *testing.T) { + t.Parallel() + + registry := prometheus.NewRegistry() + tracer := chatloop.NewStageTracer(nil, chatloop.NewMetrics(registry)) + _, span := tracer.Start(t.Context(), chatloop.StageToolCall) + span.End(nil) + tracer.Record(t.Context(), chatloop.StageThinking, chatloop.StageModel{}, time.Now().Add(-time.Second), time.Now(), nil) + + fixture := stageFixture{registry: registry} + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageToolCall, scope: chatloop.ScopeBackground}: 1, + {stage: chatloop.StageThinking, scope: chatloop.ScopeBackground}: 1, + }, fixture.stageObservations(t)) + + var nilTracer *chatloop.StageTracer + _, nilSpan := nilTracer.Start(t.Context(), chatloop.StageToolCall) + nilSpan.End(nil) + nilTracer.Record(t.Context(), chatloop.StageThinking, chatloop.StageModel{}, time.Now().Add(-time.Second), time.Now(), nil) +} + +// TestStageTracerScopeWithoutProvider covers a metrics-only +// deployment: a no-op tracer produces invalid span contexts, and the +// scope must still follow the turn. +func TestStageTracerScopeWithoutProvider(t *testing.T) { + t.Parallel() + + registry := prometheus.NewRegistry() + tracer := chatloop.NewStageTracer(nil, chatloop.NewMetrics(registry)) + + turnCtx, turn := tracer.StartRoot(t.Context(), chatloop.StageChatTurn, nil) + require.False(t, turn.SpanContext().IsValid()) + stepCtx, step := tracer.Start(turnCtx, chatloop.StageGenerationStep) + start := time.Now().Add(-time.Second) + tracer.Record(stepCtx, chatloop.StageThinking, chatloop.StageModel{}, start, time.Now(), nil) + step.End(nil) + turn.End(nil) + + fixture := stageFixture{registry: registry} + require.Equal(t, map[stageKey]uint64{ + {stage: chatloop.StageChatTurn, scope: chatloop.ScopeTurn}: 1, + {stage: chatloop.StageGenerationStep, scope: chatloop.ScopeTurn}: 1, + {stage: chatloop.StageThinking, scope: chatloop.ScopeTurn}: 1, + }, fixture.stageObservations(t)) +} diff --git a/codersdk/deployment.go b/codersdk/deployment.go index 2782ce982a119..ad3ca234932e7 100644 --- a/codersdk/deployment.go +++ b/codersdk/deployment.go @@ -4366,6 +4366,16 @@ Write out the current server config as YAML to stdout.`, Group: &deploymentGroupChat, YAML: "debugLoggingEnabled", }, + { + Name: "Chat: Stage Metrics", + Description: "How much of the chat lifecycle stage instrumentation to expose as Prometheus metrics. \"off\" exposes none. \"basic\" records per-occurrence durations for the wait, connect, and model-call stages and the per-turn time partition by category. \"full\" adds every stage and the per-turn stage distributions at a higher series count. Tracing spans are unaffected.", + Flag: "chat-stage-metrics", + Env: "CODER_CHAT_STAGE_METRICS", + Value: serpent.EnumOf(&c.AI.Chat.StageMetrics, ChatStageMetricsLevelValues...), + Default: string(ChatStageMetricsLevelOff), + Group: &deploymentGroupChat, + YAML: "stageMetrics", + }, { Name: "Chat: Hook URL", Description: "HTTPS URL to receive chat agent lifecycle hook events (plain HTTP requires --chat-hook-allow-insecure). Hooks are disabled when unset. Requires the agent-lifecycle-hooks experiment.", @@ -5103,9 +5113,45 @@ type AIBridgeProxyConfig struct { APIDumpDir serpent.String `json:"api_dump_dir" typescript:",notnull"` } +// ChatStageMetricsLevel selects how much of the chat lifecycle stage +// instrumentation is exposed as Prometheus metrics. Tracing spans are +// emitted at every level. +type ChatStageMetricsLevel string + +const ( + // ChatStageMetricsLevelOff exposes no stage or turn metrics. + ChatStageMetricsLevelOff ChatStageMetricsLevel = "off" + // ChatStageMetricsLevelBasic exposes per-occurrence durations for the + // wait, connect, and model-call stages and the per-turn time + // partition by category. + ChatStageMetricsLevelBasic ChatStageMetricsLevel = "basic" + // ChatStageMetricsLevelFull exposes every stage and the per-turn stage + // distributions. + ChatStageMetricsLevelFull ChatStageMetricsLevel = "full" +) + +// ChatStageMetricsLevelValues lists the supported ChatStageMetricsLevel values. +var ChatStageMetricsLevelValues = []string{ + string(ChatStageMetricsLevelOff), + string(ChatStageMetricsLevelBasic), + string(ChatStageMetricsLevelFull), +} + +// NewChatStageMetricsLevelFromString converts s to a ChatStageMetricsLevel, +// ignoring case and falling back to ChatStageMetricsLevelOff when s is +// empty or not a recognized level. +func NewChatStageMetricsLevelFromString(s string) ChatStageMetricsLevel { + s = strings.ToLower(s) + if slices.Contains(ChatStageMetricsLevelValues, s) { + return ChatStageMetricsLevel(s) + } + return ChatStageMetricsLevelOff +} + type ChatConfig struct { AcquireBatchSize serpent.Int64 `json:"acquire_batch_size" typescript:",notnull"` DebugLoggingEnabled serpent.Bool `json:"debug_logging_enabled" typescript:",notnull"` + StageMetrics string `json:"stage_metrics" typescript:",notnull"` HookURL serpent.URL `json:"hook_url" typescript:",notnull"` HookSecret serpent.String `json:"hook_secret" typescript:",notnull"` HookTimeout serpent.Duration `json:"hook_timeout" typescript:",notnull"` diff --git a/docs/admin/integrations/prometheus.md b/docs/admin/integrations/prometheus.md index 7ae8cad8801c4..0913aac7dd55e 100644 --- a/docs/admin/integrations/prometheus.md +++ b/docs/admin/integrations/prometheus.md @@ -249,6 +249,9 @@ The `coder_ai_gateway_cost_control_*` metrics are exported only by `coderd`. | `coderd_chatd_hook_input_overrides_total` | counter | Total lifecycle hook input overrides by event. | `event` | | `coderd_chatd_message_count` | histogram | Number of messages in the prompt per LLM request. | `model` `provider` | | `coderd_chatd_prompt_size_bytes` | histogram | Estimated byte size of the prompt per LLM request. | `model` `provider` | +| `coderd_chatd_stage_anomalies_total` | counter | Chat lifecycle stage observations dropped by reason. Reasons: negative_elapsed and inverted_window (clock inconsistencies). | `reason` | +| `coderd_chatd_stage_duration_seconds` | histogram | Wall time spent in each chat lifecycle stage. Stages overlap in wall time; this is a stage-time profile, not a partition of the turn. The scope label separates stages that run inside a chat turn from detached background work. The chat_kind label is empty for stages recorded without a known chat, and the model label is empty for stages that are not tied to a model call. At the basic stage metrics level only the wait, connect, and model-call stages are observed. | `chat_kind` `model` `scope` `stage` | +| `coderd_chatd_stage_metrics_level` | gauge | Always 1, labeled with the configured chat stage metrics level (off, basic, or full). Tells dashboards and alerts which stage and turn families this replica exposes. | `level` | | `coderd_chatd_steps_total` | counter | Total agentic loop steps across all chats. | `model` `provider` | | `coderd_chatd_stream_buffer_dropped_total` | counter | Number of chat stream buffer events dropped due to the per-chat buffer cap. | | | `coderd_chatd_stream_retries_total` | counter | Total LLM stream retries. | `kind` `model` `provider` | diff --git a/docs/admin/setup/configuration-reference.md b/docs/admin/setup/configuration-reference.md index cf997e98e0efa..5417dd4d0a69a 100644 --- a/docs/admin/setup/configuration-reference.md +++ b/docs/admin/setup/configuration-reference.md @@ -490,6 +490,15 @@ Force chat debug logging on for every chat, bypassing the runtime admin and user - YAML key: `chat.debugLoggingEnabled` - Default value: `false` +### Stage metrics + +How much of the chat lifecycle stage instrumentation to expose as Prometheus metrics. "off" exposes none. "basic" records per-occurrence durations for the wait, connect, and model-call stages and the per-turn time partition by category. "full" adds every stage and the per-turn stage distributions at a higher series count. Tracing spans are unaffected. + +- Environment variable: `CODER_CHAT_STAGE_METRICS` +- CLI flag: [`--chat-stage-metrics`](../../reference/cli/server.md#--chat-stage-metrics) +- YAML key: `chat.stageMetrics` +- Default value: `off` + ## Client These options change the behavior of how clients interact with the Coder. Clients include the Coder CLI, Coder Desktop, IDE extensions, and the web UI. diff --git a/docs/reference/api/general.md b/docs/reference/api/general.md index fb91e0385a779..b2f38978d1fe8 100644 --- a/docs/reference/api/general.md +++ b/docs/reference/api/general.md @@ -254,7 +254,8 @@ curl -X GET http://coder-server:8080/api/v2/deployment/config \ "rawQuery": "string", "scheme": "string", "user": {} - } + }, + "stage_metrics": "string" } }, "allow_workspace_renames": true, diff --git a/docs/reference/api/schemas.md b/docs/reference/api/schemas.md index 0182d4f6535bb..d33626a01c836 100644 --- a/docs/reference/api/schemas.md +++ b/docs/reference/api/schemas.md @@ -1118,7 +1118,8 @@ title: Schemas "rawQuery": "string", "scheme": "string", "user": {} - } + }, + "stage_metrics": "string" } } ``` @@ -2627,7 +2628,8 @@ AuthorizationObject can represent a "set" of objects, such as: all workspaces in "rawQuery": "string", "scheme": "string", "user": {} - } + }, + "stage_metrics": "string" } ``` @@ -2642,6 +2644,7 @@ AuthorizationObject can represent a "set" of objects, such as: all workspaces in | `hook_secret` | string | false | | | | `hook_timeout` | integer | false | | | | `hook_url` | [serpent.URL](#serpenturl) | false | | | +| `stage_metrics` | string | false | | | ## codersdk.ChatContext @@ -7444,7 +7447,8 @@ CreateWorkspaceRequest provides options for creating a new workspace. Only one o "rawQuery": "string", "scheme": "string", "user": {} - } + }, + "stage_metrics": "string" } }, "allow_workspace_renames": true, @@ -8073,7 +8077,8 @@ CreateWorkspaceRequest provides options for creating a new workspace. Only one o "rawQuery": "string", "scheme": "string", "user": {} - } + }, + "stage_metrics": "string" } }, "allow_workspace_renames": true, diff --git a/docs/reference/cli/server.md b/docs/reference/cli/server.md index 60f096e779ed2..258e507fd4571 100644 --- a/docs/reference/cli/server.md +++ b/docs/reference/cli/server.md @@ -1748,6 +1748,17 @@ How often to reconcile workspace prebuilds state. Force chat debug logging on for every chat, bypassing the runtime admin and user opt-in settings. +### --chat-stage-metrics + +| | | +|-------------|----------------------------------------| +| Type | off\|basic\|full | +| Environment | $CODER_CHAT_STAGE_METRICS | +| YAML | chat.stageMetrics | +| Default | off | + +How much of the chat lifecycle stage instrumentation to expose as Prometheus metrics. "off" exposes none. "basic" records per-occurrence durations for the wait, connect, and model-call stages and the per-turn time partition by category. "full" adds every stage and the per-turn stage distributions at a higher series count. Tracing spans are unaffected. + ### --ai-gateway-enabled | | | diff --git a/enterprise/cli/testdata/coder_server_--help.golden b/enterprise/cli/testdata/coder_server_--help.golden index 8890db6821dfa..fa9230c8170aa 100644 --- a/enterprise/cli/testdata/coder_server_--help.golden +++ b/enterprise/cli/testdata/coder_server_--help.golden @@ -282,6 +282,14 @@ Configure the background chat processing daemon. Force chat debug logging on for every chat, bypassing the runtime admin and user opt-in settings. + --chat-stage-metrics off|basic|full, $CODER_CHAT_STAGE_METRICS (default: off) + How much of the chat lifecycle stage instrumentation to expose as + Prometheus metrics. "off" exposes none. "basic" records per-occurrence + durations for the wait, connect, and model-call stages and the + per-turn time partition by category. "full" adds every stage and the + per-turn stage distributions at a higher series count. Tracing spans + are unaffected. + CLIENT OPTIONS: These options change the behavior of how clients interact with the Coder. Clients include the Coder CLI, Coder Desktop, IDE extensions, and the web UI. diff --git a/scripts/metricsdocgen/generated_metrics b/scripts/metricsdocgen/generated_metrics index 739a4f84f1f63..46f6f7b624de1 100644 --- a/scripts/metricsdocgen/generated_metrics +++ b/scripts/metricsdocgen/generated_metrics @@ -319,6 +319,15 @@ coderd_chatd_message_count{provider="",model=""} 0 # HELP coderd_chatd_prompt_size_bytes Estimated byte size of the prompt per LLM request. # TYPE coderd_chatd_prompt_size_bytes histogram coderd_chatd_prompt_size_bytes{provider="",model=""} 0 +# HELP coderd_chatd_stage_anomalies_total Chat lifecycle stage observations dropped by reason. Reasons: negative_elapsed and inverted_window (clock inconsistencies). +# TYPE coderd_chatd_stage_anomalies_total counter +coderd_chatd_stage_anomalies_total{reason=""} 0 +# HELP coderd_chatd_stage_duration_seconds Wall time spent in each chat lifecycle stage. Stages overlap in wall time; this is a stage-time profile, not a partition of the turn. The scope label separates stages that run inside a chat turn from detached background work. The chat_kind label is empty for stages recorded without a known chat, and the model label is empty for stages that are not tied to a model call. At the basic stage metrics level only the wait, connect, and model-call stages are observed. +# TYPE coderd_chatd_stage_duration_seconds histogram +coderd_chatd_stage_duration_seconds{stage="",scope="",chat_kind="",model=""} 0 +# HELP coderd_chatd_stage_metrics_level Always 1, labeled with the configured chat stage metrics level (off, basic, or full). Tells dashboards and alerts which stage and turn families this replica exposes. +# TYPE coderd_chatd_stage_metrics_level gauge +coderd_chatd_stage_metrics_level{level=""} 0 # HELP coderd_chatd_steps_total Total agentic loop steps across all chats. # TYPE coderd_chatd_steps_total counter coderd_chatd_steps_total{provider="",model=""} 0 diff --git a/site/src/api/typesGenerated.ts b/site/src/api/typesGenerated.ts index 46532f0689b3e..422b0ea440c44 100644 --- a/site/src/api/typesGenerated.ts +++ b/site/src/api/typesGenerated.ts @@ -2103,6 +2103,7 @@ export const ChatComputerUseProviders: ChatComputerUseProvider[] = [ export interface ChatConfig { readonly acquire_batch_size: number; readonly debug_logging_enabled: boolean; + readonly stage_metrics: string; readonly hook_url: string; readonly hook_secret: string; readonly hook_timeout: number; @@ -3382,6 +3383,15 @@ export interface ChatSourcePart { readonly title?: string; } +// From codersdk/deployment.go +export type ChatStageMetricsLevel = "basic" | "full" | "off"; + +export const ChatStageMetricsLevels: ChatStageMetricsLevel[] = [ + "basic", + "full", + "off", +]; + // From codersdk/chats.go export type ChatStatus = | "error"