Documentation
¶
Overview ¶
Package observability provides OpenTelemetry-based metrics for the mcp-data-platform server.
Phase 1 instruments two chokepoints: the MCP tool-call middleware and the apigateway outbound HTTP path. Metrics are exported in Prometheus format on a separate HTTP listener so scrape traffic is isolated from the main MCP/HTTP listener.
Configuration is environment-only in this phase to keep the surface small. See ConfigFromEnv for the recognized variables.
Index ¶
- Constants
- func ChildSpan(ctx context.Context, name string, opts ...trace.SpanStartOption) (context.Context, trace.Span)
- func ClassifyError(err error) string
- func ClassifyToolCallResult(err error, isToolError bool, errCategory string) string
- func DBOperationAttributes(operation string) []attribute.KeyValue
- func DeploymentIDSet() bool
- func HTTPMethodLabel(method string) string
- func HTTPStatusCategory(status int, transportErr error) string
- func HTTPStatusClass(status int) string
- func PersonaLabel(persona string) string
- func RedactError(err error) error
- func RedactMessage(msg string) string
- func Resource() *resource.Resource
- func SetSpanStatus(span trace.Span, statusCategory string, err error)
- func SourceLabel(source string) string
- func UpstreamStatus(err error) string
- type APIGatewayAttrs
- type APIGatewayInboundAttrs
- type BackgroundQueueSample
- type BackgroundQueueSampler
- type CategorizedError
- type Config
- type ConfigInfo
- type ConnectionStateCount
- type DBConfig
- type DependencySample
- type IndexAgeSample
- type IndexQueueSample
- type IndexQueueSampler
- type Listener
- type LogProvider
- type LogsConfig
- type LogsExporter
- type Metrics
- func (m *Metrics) DBMeterProvider(slow time.Duration) metric.MeterProvider
- func (m *Metrics) DecInflightToolCalls(ctx context.Context)
- func (m *Metrics) Enabled() bool
- func (m *Metrics) Handler() http.Handler
- func (m *Metrics) IncInflightToolCalls(ctx context.Context)
- func (m *Metrics) IndexEmbedCall(ctx context.Context, kind string, texts int, status string, d time.Duration)
- func (m *Metrics) IndexJobEnqueued(ctx context.Context, kind, trigger string, created bool)
- func (m *Metrics) IndexJobFinished(ctx context.Context, kind, trigger, outcome string, d time.Duration)
- func (m *Metrics) IndexJobItems(ctx context.Context, kind string, embedded, reused int)
- func (m *Metrics) IndexJobStarted(ctx context.Context, kind string)
- func (m *Metrics) IndexLeasesReleased(ctx context.Context, released int)
- func (m *Metrics) IndexUnitsDeferred(ctx context.Context, kind string, deferred int)
- func (m *Metrics) RecordAPIGatewayInbound(ctx context.Context, attrs APIGatewayInboundAttrs, duration time.Duration)
- func (m *Metrics) RecordAPIGatewayOutbound(ctx context.Context, attrs APIGatewayAttrs, duration time.Duration)
- func (m *Metrics) RecordArchiveExtraction(ctx context.Context, members int, bytes int64)
- func (m *Metrics) RecordArchiveRefusal(ctx context.Context, reason string)
- func (m *Metrics) RecordAuditEventDropped(ctx context.Context, reason string)
- func (m *Metrics) RecordAuditWrite(ctx context.Context, result string, d time.Duration)
- func (m *Metrics) RecordAuthAttempt(ctx context.Context, method, result, reason string)
- func (m *Metrics) RecordAuthFailOpen(ctx context.Context)
- func (m *Metrics) RecordConfigInfo(ctx context.Context, info ConfigInfo)
- func (m *Metrics) RecordConfigWarning(ctx context.Context, code string, n int64)
- func (m *Metrics) RecordConnectionOAuthCredentials(ctx context.Context, kind, state string, n int)
- func (m *Metrics) RecordConnectionOAuthRefresh(ctx context.Context, kind, result string)
- func (m *Metrics) RecordDataHubRequest(ctx context.Context, operation, status string, duration time.Duration)
- func (m *Metrics) RecordEgressBlocked(ctx context.Context, reason string)
- func (m *Metrics) RecordEmbeddingCall(ctx context.Context, model, status string, d time.Duration)
- func (m *Metrics) RecordEmbeddingFallback(ctx context.Context, model string)
- func (m *Metrics) RecordEnrichmentBytes(ctx context.Context, attrs ToolCallAttrs, bytes int)
- func (m *Metrics) RecordGatewaySessionRedial(ctx context.Context, connection, result string)
- func (m *Metrics) RecordGatewayUpstreamCall(ctx context.Context, connection, outcome string, d time.Duration)
- func (m *Metrics) RecordHTTPClientRequest(ctx context.Context, kind, connection, statusClass string, d time.Duration)
- func (m *Metrics) RecordHTTPRateLimited(ctx context.Context, limiter string)
- func (m *Metrics) RecordHTTPServerRequest(ctx context.Context, route, method, statusClass string, d time.Duration)
- func (m *Metrics) RecordJWKSFetch(ctx context.Context, err error, d time.Duration)
- func (m *Metrics) RecordKnowledgeChange(ctx context.Context, sink, result string)
- func (m *Metrics) RecordListenConnected(ctx context.Context, loop string, up, reconnect bool)
- func (m *Metrics) RecordListenNotification(loop string)
- func (m *Metrics) RecordLoopIteration(ctx context.Context, loop, result string, d time.Duration)
- func (m *Metrics) RecordMCPRequest(ctx context.Context, method, status string, d time.Duration)
- func (m *Metrics) RecordNotificationAttempt(ctx context.Context, kind, result string)
- func (m *Metrics) RecordOAuthIssuance(ctx context.Context, grantType, status string)
- func (m *Metrics) RecordOAuthRefresh(ctx context.Context, status string, duration time.Duration)
- func (m *Metrics) RecordOAuthRegistration(ctx context.Context, err error)
- func (m *Metrics) RecordRateLimitQueued(ctx context.Context)
- func (m *Metrics) RecordRateLimited(ctx context.Context)
- func (m *Metrics) RecordRowsPurged(ctx context.Context, loop string, n int64)
- func (m *Metrics) RecordS3Operation(ctx context.Context, operation, status string, duration time.Duration)
- func (m *Metrics) RecordScriptAdmissionRefused(ctx context.Context, reason string)
- func (m *Metrics) RecordScriptMissedFires(ctx context.Context, scriptName string, missed int)
- func (m *Metrics) RecordScriptQueueWait(ctx context.Context, wait time.Duration)
- func (m *Metrics) RecordScriptRun(ctx context.Context, attrs ScriptRunAttrs, duration time.Duration)
- func (m *Metrics) RecordScriptRunFailure(ctx context.Context, cause string)
- func (m *Metrics) RecordScriptRunReclaim(ctx context.Context, outcome string)
- func (m *Metrics) RecordScriptRunShed(ctx context.Context)
- func (m *Metrics) RecordScriptWorkerLoad(ctx context.Context, reason string, ratio float64)
- func (m *Metrics) RecordSearchResults(ctx context.Context, n int)
- func (m *Metrics) RecordSessionResolution(ctx context.Context, source string)
- func (m *Metrics) RecordStorageOperation(ctx context.Context, o StorageOperation)
- func (m *Metrics) RecordThumbnailFailure(ctx context.Context, kind, reason string)
- func (m *Metrics) RecordThumbnailRender(ctx context.Context, kind string, d time.Duration)
- func (m *Metrics) RecordThumbnailRenderer(ctx context.Context, up bool)
- func (m *Metrics) RecordToolCall(ctx context.Context, attrs ToolCallAttrs, duration time.Duration)
- func (m *Metrics) RecordToolDenial(ctx context.Context, persona, reason string)
- func (m *Metrics) RecordTrinoQuery(ctx context.Context, status, queryKind string, duration time.Duration)
- func (m *Metrics) RecordUpstreamRetry(ctx context.Context, kind string, exhausted bool)
- func (m *Metrics) RegisterAuditQueueDepth(fn func() int)
- func (m *Metrics) RegisterBackgroundQueue(loop string, s BackgroundQueueSampler)
- func (m *Metrics) RegisterDBPool(db *sql.DB, name string)
- func (m *Metrics) RegisterIndexQueue(s IndexQueueSampler)
- func (m *Metrics) RegisterPlatformState(s PlatformStateSampler)
- func (m *Metrics) ScriptRunFinished(ctx context.Context)
- func (m *Metrics) ScriptRunStarted(ctx context.Context)
- func (m *Metrics) Shutdown(ctx context.Context) error
- func (m *Metrics) StartOp(ctx context.Context, name string) (context.Context, *Op)
- func (m *Metrics) WebhookAck(ctx context.Context, source string, d time.Duration)
- func (m *Metrics) WebhookBuffer(ctx context.Context, source string, events int)
- func (m *Metrics) WebhookCompaction(ctx context.Context, source, result string)
- func (m *Metrics) WebhookDuplicatesDropped(ctx context.Context, source string, n int64)
- func (m *Metrics) WebhookEvents(ctx context.Context, source string, n int)
- func (m *Metrics) WebhookRequest(ctx context.Context, source, outcome string)
- func (m *Metrics) WebhookSegmentWritten(ctx context.Context, source string)
- type MetricsExporter
- type OTLPEndpoint
- type Op
- type PlatformStateSample
- type PlatformStateSampler
- type ScriptRunAttrs
- type StorageOperation
- type ToolCallAttrs
- type Tracer
- type TracingConfig
Constants ¶
const ( // DefaultOTLPEndpoint is used when tracing is enabled but // OTEL_EXPORTER_OTLP_ENDPOINT is unset. DefaultOTLPEndpoint = "localhost:4317" // DefaultServiceName is the service.name resource value when neither // OTEL_SERVICE_NAME nor OTEL_RESOURCE_ATTRIBUTES names one. DefaultServiceName = "mcp-data-platform" // DefaultSamplerArg is the head-based sampling ratio when // OTEL_TRACES_SAMPLER_ARG is unset or unparsable. DefaultSamplerArg = 0.1 )
Tracing defaults.
const ( // AuditDropQueueFull is an event refused because the writer's queue was // full or the writer was closing. AuditDropQueueFull = "queue_full" // AuditDropWriteFailed is an event whose store write returned an error. AuditDropWriteFailed = "write_failed" // AuditDropTimeout is an event whose store write was abandoned at the // per-write timeout or at shutdown. AuditDropTimeout = "timeout" )
Audit drop reasons, as RecordAuditEventDropped labels them (#1897).
const ( // ReclaimReexecuted is a run another worker took over and executes again. ReclaimReexecuted = "reexecuted" // ReclaimFailed is a run failed because its reclaims were spent. ReclaimFailed = "failed" )
Reclaim outcomes, as RecordScriptRunReclaim labels them (#1860).
const ( // LoopResultOK is an iteration that did its work. LoopResultOK = "ok" // LoopResultError is an iteration that returned an error. LoopResultError = "error" // LoopResultSkipped is an iteration that found the work held elsewhere // (another replica's advisory lock) and did nothing. LoopResultSkipped = "skipped" )
Loop iteration results, as RecordLoopIteration labels them.
const ( // QueueStatePending is an item due and not yet claimed. QueueStatePending = "pending" // QueueStateRunning is an item claimed and in progress. QueueStateRunning = "running" // QueueStateWaiting is an item not yet due (a retry backoff, an open // compaction window). QueueStateWaiting = "waiting" )
Queue item states, as a BackgroundQueueSample reports them.
const ( StoragePurposePortalAssets = "portal_assets" StoragePurposeResources = "resources" StoragePurposeThumbnails = "thumbnails" StoragePurposeScriptOutputs = "script_outputs" StoragePurposeExports = "exports" StoragePurposeWebhooks = "webhooks" )
The purposes a platform-owned object operation is reported under: what the bucket is being used for, independent of which bucket a deployment named.
const ( StorageReasonAccessDenied = "access_denied" StorageReasonBucketMissing = "bucket_missing" StorageReasonQuotaExceeded = "quota_exceeded" StorageReasonNotFound = "not_found" StorageReasonOther = "other" )
The reason classes a failed object operation carries, so a credential or bucket-policy mistake reads differently from an outage.
const ( GatewayOutcomeOK = "ok" GatewayOutcomeToolError = "tool_error" GatewayOutcomeTransportError = "transport_error" GatewayOutcomeTimeout = "timeout" )
The outcomes of a forwarded MCP gateway call.
const ( RedialOK = "ok" RedialFailed = "failed" )
The results of a re-dial after a dropped upstream session.
const ( StatusOK = "ok" StatusAuthErr = "auth_err" StatusAuthzErr = "authz_err" // StatusGateErr is a call the platform refused before the handler ran for // a reason other than identity: the session gate, the search-first gate, // a missing session handle or purpose, or the per-user rate limit. The // specific gate is the error category on the audit row and on the span; // the metric carries one value so the label set stays closed (#1892). StatusGateErr = "gate_err" // StatusDeclined is a person answering no to an elicitation prompt. It is // not bad input and not a platform fault, so it has its own value rather // than folding into StatusValidationErr (#1892). StatusDeclined = "declined" StatusValidationErr = "validation_err" StatusUpstreamErr = "upstream_err" StatusInternalErr = "internal_err" )
Status category labels for tool calls and outbound HTTP. The set is closed and small so total label cardinality on counters and histograms stays bounded.
const ( StatusClientErr = "client_err" StatusServerErr = "server_err" )
Status labels for the OAuth server's own token endpoint. A grant the client got wrong (an expired code, a mismatched client_id, a bad verifier) and a grant the platform could not complete (its token store or signer failed) are different operational conditions and were one upstream_err label until #1892; the AuthFailureSpike alert reads status!="ok", so both still count.
const ( OutcomeOK = "ok" OutcomeUpstream4xx = "upstream_4xx" OutcomeUpstream5xx = "upstream_5xx" OutcomeTransportErr = "transport_err" OutcomeUpstreamTimeout = "upstream_timeout" // OutcomeUpstreamError is a 2xx whose body reports failure. A GraphQL // endpoint answers almost every failure as an HTTP 200 carrying an // errors array, and the status-line categories above cannot name // that; this one does, so the audit row and the call record of such // a call carry success=false like a 4xx does (#1678). OutcomeUpstreamError = "upstream_error" )
Audit outcome categories for upstream-proxying toolkits (e.g. the apigateway). These are bounded labels that distinguish gateway-level failure (the gateway could not reach the upstream) from upstream-level failure (the upstream responded with an error status). The two are fundamentally different operational concerns and should not share a status code or a success boolean:
- The gateway returning 502 means the gateway broke.
- The upstream returning 502 (proxied through the gateway as wire 200 with the upstream code in the body) means the upstream broke. The gateway did its job.
A toolkit stamps one of these on its result's _meta (MetaAuditOutcome below); the audit middleware records success=false on the row for any value but OutcomeOK, and carries the value as the category of the event it builds.
const ( // MetaAuditOutcome carries one of the Outcome* string constants // above. When present and not OutcomeOK, the audit middleware // sets success=false and uses the value as error_category. MetaAuditOutcome = "audit_outcome" // MetaAuditOutcomeMessage carries an optional human-readable // summary of the outcome (typically the upstream status text or // the scrubbed transport error). Used to populate // audit_logs.error_message when no other source is available. MetaAuditOutcomeMessage = "audit_outcome_message" // MetaAuditResult is the _meta key under which a tool reports facts // about the call's outcome for the audit row: a map the audit // middleware records under parameters.result. The api gateway's page // walk stamps its pages_fetched, items_merged, and stopped_by here, // which is how one call that walked 160 pages stays observable in // audit (issue #1535). MetaAuditResult = "audit_result" )
Well-known CallToolResult Meta keys read by the audit middleware to override its success / error_category derivation. Toolkits that proxy external services populate these so the audit row reflects the real upstream outcome instead of just "the MCP tool ran." Keys are namespaced under "audit_" to keep them out of the way of other _meta consumers.
const ( StatusClass2xx = "2xx" StatusClass3xx = "3xx" StatusClass4xx = "4xx" StatusClass5xx = "5xx" StatusClassOther = "other" )
HTTP status class labels for outbound calls. The "other" bucket covers transport-level failures (status code 0) and the rarely-seen 1xx informational range. Recording the raw status code as a label would explode cardinality.
const ( CategoryAuth = "authentication_failed" CategoryAuthz = "authorization_denied" CategoryDeclined = "user_declined" CategoryClientInput = "client_input" CategoryNotFound = "not_found" CategoryInternal = "internal" // The gate refusals (StatusGateErr). The values are the categories the // gates in pkg/middleware and internal/platform/toolratelimit stamp on // their refusals. CategorySetupRequired = "setup_required" CategorySearchRequired = "search_required" CategorySessionRequired = "session_required" CategoryPurposeRequired = "purpose_required" CategoryRateLimited = "rate_limited" )
Category constants recognized by ClassifyToolCall when a CategorizedError is returned. These match the values pkg/middleware.ErrCategory* uses so the platform's existing error taxonomy maps to bounded metric labels without duplication.
const DefaultDBSlowStatement = time.Second
DefaultDBSlowStatement is the slow-statement threshold when MCP_PLATFORM_DB_SLOW_STATEMENT_THRESHOLD is unset.
const DefaultListenAddr = ":9090"
DefaultListenAddr is the address the /metrics listener binds to when OTEL_METRICS_ADDR is unset. Port 9090 is the conventional Prometheus scrape port and does not collide with the platform's main HTTP port (8080 by default).
const InstrumentationScope = "github.com/txn2/mcp-data-platform"
InstrumentationScope is the tracer name used for spans the platform creates directly. Call sites use otel.Tracer(InstrumentationScope) so every span shares one scope without threading a *Tracer through every constructor.
const MetricLabelUnknown = "unknown"
MetricLabelUnknown is the single value every principal label falls back to when no principal could be resolved. Recording one fixed value keeps an unauthenticated or pre-authorization call countable without minting a series per unresolved caller.
const ToolLabelUnregistered = "unregistered"
ToolLabelUnregistered is the tool label for a tools/call naming a tool no toolkit registers. The name a caller sends is theirs to choose, and a persona allowing "*" admits it as far as the handler lookup, so recording it as sent would let one caller mint a series per invented name (#1892). The name itself is on the audit row.
Variables ¶
This section is empty.
Functions ¶
func ChildSpan ¶ added in v1.87.0
func ChildSpan(ctx context.Context, name string, opts ...trace.SpanStartOption) (context.Context, trace.Span)
ChildSpan starts a span ONLY when ctx already carries an active trace (a valid span context from an upstream root span). When tracing is disabled — or this code runs outside any traced request — it returns a non-recording no-op span (and a context carrying it). This lets adapter call sites (Trino/DataHub/S3/OAuth/enrichment/audit) open a child span unconditionally with a single cheap check and never create orphan spans or pay span-allocation cost when tracing is off.
func ClassifyError ¶
ClassifyError maps an error returned from a tool handler (or from any internal stage of the call) to a bounded status_category label. A nil error yields StatusOK.
The classifier prefers a CategorizedError's ErrorCategory() over string inspection so the platform's error taxonomy stays authoritative. Categories the metrics package does not recognize fall through to StatusInternalErr — a recognized-but-unmapped category is a signal that the taxonomy and the classifier have drifted; the deliberate bucket makes the drift visible in a dashboard.
func ClassifyToolCallResult ¶
ClassifyToolCallResult maps the (err, isToolError, errCategory) triple from an MCP tool call to a bounded status_category. This is the shape pkg/middleware.MCPAuditMiddleware already computes, so the metrics middleware can pass through the same fields without re-deriving them.
Logic:
- err != nil → ClassifyError(err) (protocol-level failure)
- !isToolError → StatusOK
- isToolError with a recognized category → mapped label; a refusal by one of the platform's gates maps to StatusGateErr, a declined elicitation to StatusDeclined
- isToolError without a category → StatusUpstreamErr (most tool-level errors are upstream — Trino query failures, S3 access errors, DataHub fetch errors, etc.)
func DBOperationAttributes ¶ added in v1.141.0
DBOperationAttributes is the one label a statement is measured under: its operation. The driver wrapper hands these to the statement histogram, whose view keeps nothing else (histogramAttributeFilter).
func DeploymentIDSet ¶ added in v1.141.0
func DeploymentIDSet() bool
DeploymentIDSet reports whether the resource names the deployment (mcp_platform.deployment.id), from either variable that can set it. A fleet backend receiving OTLP from several deployments cannot tell them apart without it, which is why startup warns when an OTLP exporter is on and this is false.
func HTTPMethodLabel ¶ added in v1.141.0
HTTPMethodLabel clamps an HTTP method to the standard set, so a request with an invented method cannot mint a series; anything else is unknown. Each case returns the constant rather than the request's own string, so the value a label or a log line carries is never the caller's bytes.
func HTTPStatusCategory ¶
HTTPStatusCategory returns the status_category label for an outbound HTTP call. 2xx and 3xx are treated as OK; 4xx and 5xx as upstream errors. Transport errors (status 0) are upstream errors too — the upstream did not respond.
func HTTPStatusClass ¶
HTTPStatusClass returns the bounded class label for an HTTP status code. Status 0 is reserved for transport-level errors (no response received); it maps to StatusClassOther so it is recordable without inflating the 5xx bucket.
func PersonaLabel ¶ added in v1.130.0
PersonaLabel bounds a resolved persona name into a label value. An empty name -- a call that failed authentication, or one assembled outside the tool-call middleware -- becomes MetricLabelUnknown, so mcp_tool_calls_total and apigateway_outbound_total name an unresolved principal identically (#1615). Applied inside the Record methods rather than at each call site, so the bound holds for every recorder.
func RedactError ¶ added in v1.141.0
RedactError returns an error whose text is RedactMessage of err's. It is what SetSpanStatus records on the span; the original error is for the caller, the audit row and the log line, which stay inside the platform. A nil err yields nil.
func RedactMessage ¶ added in v1.141.0
RedactMessage returns the form of an error message that may leave the platform on telemetry (#1892): quoted literals and email addresses are replaced, control characters are stripped, and the result is cut at maxRedactedBytes. The error's shape survives (which operation failed and why); the values it carried do not.
func Resource ¶ added in v1.141.0
Resource is the OpenTelemetry resource that identifies this process on every signal it emits (#1893): service.name (OTEL_SERVICE_NAME, default DefaultServiceName), service.version and vcs.ref.head.revision from the build, service.instance.id (the hostname, else a UUID minted at boot), deployment.environment.name and mcp_platform.deployment.id from the platform's own variables, merged over OTEL_RESOURCE_ATTRIBUTES so an operator's extra attributes are honored and the specific variables win. Built once and shared by the tracer, the meter provider and the log provider.
func SetSpanStatus ¶ added in v1.87.0
SetSpanStatus records the outcome of an operation on span from the platform's bounded status_category plus the underlying error. It maps every category except StatusOK to codes.Error so error traces stand out in Tempo/Jaeger. The status description is the category itself, and the error is recorded as a span event after RedactError: an upstream error's text can quote the SQL a query ran or the body a response carried, and a trace backend is outside the platform (#1892). Nil-safe span handling is the caller's (trace.Span is never nil from Start).
func SourceLabel ¶ added in v1.141.0
SourceLabel bounds a call's source into a label value: an empty source, which only a call assembled outside the tool-call middleware carries, records MetricLabelUnknown.
func UpstreamStatus ¶ added in v1.70.0
UpstreamStatus maps an error from an external dependency (Trino, DataHub, S3, an IdP) to a bounded status label: nil is StatusOK, anything else is StatusUpstreamErr. It reuses the platform's existing status taxonomy (see status.go) rather than introducing a parallel client_err/server_err set. Call sites with an HTTP status code should use HTTPStatusCategory instead.
Types ¶
type APIGatewayAttrs ¶
type APIGatewayAttrs struct {
Connection string
HTTPStatusClass string
StatusCategory string
Persona string
}
APIGatewayAttrs is the bounded label set for outbound HTTP from the apigateway toolkit. Connection is operator-configured (small set); the URL, path, query string, and raw status code are NOT recorded as labels — they would be cardinality bombs and live on trace spans instead.
Persona is the persona the call was authorized under, the dimension that separates an automated principal's traffic from an analyst's on a connection they share (#1615). It is bounded by the deployment's persona definitions, which an operator authors; the caller's identity -- an OIDC subject, one per person -- is deliberately NOT a label here, since it grows with the organization. An unresolved persona records MetricLabelUnknown, one fixed value rather than a new series. Persona is applied to the call counter only; RecordAPIGatewayOutbound states why.
type APIGatewayInboundAttrs ¶ added in v1.69.0
type APIGatewayInboundAttrs struct {
Connection string
OperationID string
Method string
StatusClass string
Identity string
}
APIGatewayInboundAttrs is the bounded label set for inbound HTTP requests to the apigateway REST shim. OperationID is the OpenAPI operationId resolved from the connection's catalog ("unknown" when unresolved); Identity is the API key name, "oidc" for a signed-in person, or "unknown" when unauthenticated. The raw path, query string, and numeric status code are NOT labels; they are cardinality bombs and belong on trace spans. Identity is applied to the request counter only (see RecordAPIGatewayInbound).
type BackgroundQueueSample ¶ added in v1.141.0
type BackgroundQueueSample struct {
Kind string
Pending int64
Running int64
Waiting int64
OldestAge time.Duration
}
BackgroundQueueSample is one queue's state as its store holds it, read on a scrape. Kind subdivides a queue (a notification's kind, a thumbnail's family) and may be empty.
type BackgroundQueueSampler ¶ added in v1.141.0
type BackgroundQueueSampler func(ctx context.Context) ([]BackgroundQueueSample, error)
BackgroundQueueSampler returns a queue's state. It is called on a scrape with a bounded context; the result is cached for backgroundSampleMaxAge.
type CategorizedError ¶
CategorizedError lets call sites attach a category to an error that the metrics layer can read without a string-match. This mirrors the pattern used by pkg/middleware's PlatformError so the existing auth/authz/declined categories surface in metrics without a second classification scheme.
type Config ¶
type Config struct {
// Enabled gates the entire subsystem. When false, New returns a
// Metrics value whose Record methods are no-ops, the listener is
// not started, and no OTel MeterProvider is constructed.
Enabled bool
// ListenAddr is the bind address for the /metrics HTTP listener,
// e.g. ":9090" or "127.0.0.1:9090". Ignored when Enabled is false or
// Exporter leaves Prometheus out.
ListenAddr string
// Exporter is where metrics go: the Prometheus listener, an OTLP push to
// the collector, or both (#1893).
Exporter MetricsExporter
// OTLP is the collector the OTLP reader pushes to; shared with the
// tracer and the log provider.
OTLP OTLPEndpoint
}
Config holds the operator-configurable knobs for the metrics subsystem.
func ConfigFromEnv ¶
func ConfigFromEnv() Config
ConfigFromEnv reads the observability configuration from environment variables. Unset or unparsable values fall back to the defaults so the platform can boot even with a partial configuration.
Metrics are enabled by default. The /metrics listener binds to DefaultListenAddr on a separate port from the main MCP/HTTP listener so operators can isolate scrape traffic with a NetworkPolicy. Set OTEL_METRICS_ENABLED=false to disable.
type ConfigInfo ¶ added in v1.141.0
type ConfigInfo struct {
ToolkitKinds []string
AuthMethods []string
Tracing bool
SamplerRatio float64
}
ConfigInfo is what mcp_platform_config_info reports. Every field is a bounded, non-secret value; RecordConfigInfo drops any value that reads as an address (a scheme separator or an @), so a hostname or a DSN cannot reach a label even if a caller passes one.
type ConnectionStateCount ¶ added in v1.141.0
ConnectionStateCount is how many connections of a kind are in a state.
type DBConfig ¶ added in v1.141.0
type DBConfig struct {
// SlowThreshold is the statement duration counted as slow; zero turns
// the slow count and its log off.
SlowThreshold time.Duration
// IncludeStatement puts the SQL text on the statement span.
IncludeStatement bool
}
DBConfig is how the platform's SQL driver wrapper observes statements.
func DBConfigFromEnv ¶ added in v1.141.0
func DBConfigFromEnv() DBConfig
DBConfigFromEnv reads the statement observation settings. An unparseable or negative threshold is the default.
type DependencySample ¶ added in v1.141.0
DependencySample is one dependency's last probe.
type IndexAgeSample ¶ added in v1.141.0
IndexAgeSample is how long ago a search source last finished indexing.
type IndexQueueSample ¶ added in v1.134.1
type IndexQueueSample struct {
Kind string
Pending int64
Running int64
Retrying int64
FailedUnits int64
// OldestRunnableWait is how long the longest-waiting runnable job has
// waited for a worker; zero when none is waiting.
OldestRunnableWait time.Duration
// CoverageKnown is false when the kind reported no coverage; the vector
// gauges are then not observed for it. ExpectedKnown is false for a kind
// with no fixed denominator (the tool registry), which observes indexed
// alone.
CoverageKnown bool
Indexed int64
Expected int64
ExpectedKnown bool
}
IndexQueueSample is one kind's state as the database holds it, read on a scrape by the sampler RegisterIndexQueue installs.
type IndexQueueSampler ¶ added in v1.134.1
type IndexQueueSampler func(ctx context.Context) ([]IndexQueueSample, error)
IndexQueueSampler returns every registered kind's state. It is called on each scrape with a bounded context and is expected to cache.
type Listener ¶
type Listener struct {
// contains filtered or unexported fields
}
Listener runs the dedicated HTTP server that exposes /metrics. The server is separate from the platform's main HTTP listener so that:
- scrape traffic does not share the MCP/admin/portal auth path,
- the metrics port can sit behind a NetworkPolicy (or be unreachable from outside the cluster) without affecting client-facing routes,
- a slow or stuck scraper cannot starve the main listener's accept loop.
Listener is a no-op when the underlying Metrics is nil or the listen address is empty; callers can mount it unconditionally.
func NewListener ¶
NewListener constructs a Listener for the supplied Metrics. The listener serves only /metrics on its mux; all other paths return 404. When metrics are disabled, or go to OTLP alone (OTEL_METRICS_EXPORTER=otlp, #1893), NewListener returns nil so callers can mount it unconditionally and observe a nil receiver as the "disabled" signal.
type LogProvider ¶ added in v1.141.0
type LogProvider struct {
// contains filtered or unexported fields
}
LogProvider owns the OpenTelemetry LoggerProvider and OTLP exporter that the slog bridge writes through. A nil *LogProvider is a valid no-op: Handler returns nil and Shutdown returns nil, so the composition root holds and uses it unconditionally.
func NewLogProvider ¶ added in v1.141.0
func NewLogProvider(cfg LogsConfig) (*LogProvider, error)
NewLogProvider builds a LogProvider from cfg. When cfg.Exporter is not otlp it returns (nil, nil), mirroring NewTracer. The exporter connects lazily and records are batched, so an unreachable collector never delays startup or blocks a log call.
func NewLogProviderFromSDK ¶ added in v1.141.0
func NewLogProviderFromSDK(provider *sdklog.LoggerProvider) *LogProvider
NewLogProviderFromSDK wraps an already-constructed LoggerProvider. It is the seam NewLogProvider uses and the one tests use to inject an in-memory exporter. Always returns a non-nil *LogProvider.
func (*LogProvider) Handler ¶ added in v1.141.0
func (p *LogProvider) Handler() slog.Handler
Handler is the slog.Handler that forwards records to the provider through the otelslog bridge: each record becomes a log record carrying the resource, the record's attributes, and the trace and span ids of the span its context carries. Nil-safe: a disabled provider returns nil, which the composition root reads as "no second sink".
type LogsConfig ¶ added in v1.141.0
type LogsConfig struct {
// Exporter is where log records go beside stderr.
Exporter LogsExporter
// OTLP is the collector the records are exported to; shared with the
// tracer and the metrics OTLP reader.
OTLP OTLPEndpoint
}
LogsConfig holds the operator-configurable knobs for log export.
func LogsConfigFromEnv ¶ added in v1.141.0
func LogsConfigFromEnv() LogsConfig
LogsConfigFromEnv reads the log-export configuration. Export is off unless OTEL_LOGS_EXPORTER=otlp.
type LogsExporter ¶ added in v1.141.0
type LogsExporter string
LogsExporter is the OTEL_LOGS_EXPORTER choice.
const ( LogsExporterNone LogsExporter = "none" LogsExporterOTLP LogsExporter = "otlp" )
The LogsExporter values.
type Metrics ¶
type Metrics struct {
// contains filtered or unexported fields
}
Metrics owns the OTel MeterProvider and the registered instruments. A nil *Metrics is a valid no-op recorder: every Record method becomes a fast nil-check, so call sites can record unconditionally without an enabled check.
func New ¶
New builds a Metrics instance from the supplied config. When cfg.Enabled is false New returns (nil, nil) so callers receive a no-op recorder without an error path; this keeps the boot sequence simple in cmd/mcp-data-platform when metrics are off. The nil-no-op shape is intentional and documented on every Record method — callers can invoke them unconditionally.
When enabled, New constructs a fresh prometheus.Registry (NOT the default registerer) so the platform's metrics are isolated from any other library that may publish to the default registry. The Go runtime and process collectors are registered explicitly so "go_goroutines", "process_cpu_seconds_total", and friends are available on the same /metrics endpoint without extra wiring.
func (*Metrics) DBMeterProvider ¶ added in v1.141.0
func (m *Metrics) DBMeterProvider(slow time.Duration) metric.MeterProvider
DBMeterProvider is the meter provider the SQL driver wrapper records statement durations through, so they reach /metrics and the OTLP push beside the platform's own series, with the slow-statement count taken from the same measurement. Nil-safe: a disabled recorder gives a no-op provider.
func (*Metrics) DecInflightToolCalls ¶
DecInflightToolCalls decrements the in-flight gauge. Nil-safe.
func (*Metrics) Enabled ¶
Enabled reports whether the recorder is active. The middleware uses this only to skip building label sets when nothing will be recorded; Record methods themselves are nil-safe.
func (*Metrics) Handler ¶
Handler returns the /metrics HTTP handler. Returns http.NotFoundHandler when m is nil so cmd/main can mount the handler unconditionally, and nil when the recorder pushes over OTLP alone (OTEL_METRICS_EXPORTER=otlp), which is how NewListener knows there is nothing to serve.
func (*Metrics) IncInflightToolCalls ¶
IncInflightToolCalls increments the in-flight gauge. Paired with DecInflightToolCalls in a defer at the tool-call middleware so the gauge cannot leak even on panic. Nil-safe.
func (*Metrics) IndexEmbedCall ¶ added in v1.134.1
func (m *Metrics) IndexEmbedCall(ctx context.Context, kind string, texts int, status string, d time.Duration)
IndexEmbedCall records one EmbedBatch call. Nil-safe.
func (*Metrics) IndexJobEnqueued ¶ added in v1.134.1
IndexJobEnqueued records one Enqueue. Nil-safe.
func (*Metrics) IndexJobFinished ¶ added in v1.134.1
func (m *Metrics) IndexJobFinished(ctx context.Context, kind, trigger, outcome string, d time.Duration)
IndexJobFinished records a job settling: the running gauge drops, the job is counted by outcome, and its duration is observed. Nil-safe.
func (*Metrics) IndexJobItems ¶ added in v1.134.1
IndexJobItems records one pass's plan. Nil-safe.
func (*Metrics) IndexJobStarted ¶ added in v1.134.1
IndexJobStarted marks a job executing on this replica. Nil-safe.
func (*Metrics) IndexLeasesReleased ¶ added in v1.134.1
IndexLeasesReleased records one reaper sweep's reclaimed leases. Nil-safe.
func (*Metrics) IndexUnitsDeferred ¶ added in v1.134.1
IndexUnitsDeferred records the parked units one reconciler sweep skipped for a kind. Nil-safe.
func (*Metrics) RecordAPIGatewayInbound ¶ added in v1.69.0
func (m *Metrics) RecordAPIGatewayInbound(ctx context.Context, attrs APIGatewayInboundAttrs, duration time.Duration)
RecordAPIGatewayInbound records one inbound REST-shim request observation. The request counter carries the identity label; the duration histogram deliberately omits it so the bucket series do not multiply by the identity dimension. Nil-safe.
func (*Metrics) RecordAPIGatewayOutbound ¶
func (m *Metrics) RecordAPIGatewayOutbound(ctx context.Context, attrs APIGatewayAttrs, duration time.Duration)
RecordAPIGatewayOutbound records one outbound HTTP observation. The call counter carries the persona label; the duration histogram deliberately omits it so the bucket series do not multiply by the principal dimension, mirroring RecordAPIGatewayInbound's treatment of identity. Nil-safe.
func (*Metrics) RecordArchiveExtraction ¶ added in v1.141.0
RecordArchiveExtraction records one extraction's members and bytes. Nil-safe.
func (*Metrics) RecordArchiveRefusal ¶ added in v1.141.0
RecordArchiveRefusal records one refused extraction. Nil-safe.
func (*Metrics) RecordAuditEventDropped ¶ added in v1.101.0
RecordAuditEventDropped records one audit event lost by the bounded async writer, by reason: a queue-full drop, a write that failed, or one abandoned at the per-write timeout (issue #884, reason #1897). The tool and request dimensions live in the loss's slog line, not in a high-cardinality metric. Nil-safe, so the writer records unconditionally without an enabled check.
func (*Metrics) RecordAuditWrite ¶ added in v1.141.0
RecordAuditWrite records one audit store write. Nil-safe.
func (*Metrics) RecordAuthAttempt ¶ added in v1.141.0
RecordAuthAttempt records one credential validation. An empty method records MetricLabelUnknown; an empty reason records none. Nil-safe.
func (*Metrics) RecordAuthFailOpen ¶ added in v1.141.0
RecordAuthFailOpen records one request the HTTP gate passed through while its credential could not be validated. Nil-safe.
func (*Metrics) RecordConfigInfo ¶ added in v1.141.0
func (m *Metrics) RecordConfigInfo(ctx context.Context, info ConfigInfo)
RecordConfigInfo sets mcp_platform_config_info to 1 under info's labels. A second call replaces the first series' value with 0, so a scrape shows one current configuration. Nil-safe.
func (*Metrics) RecordConfigWarning ¶ added in v1.141.0
RecordConfigWarning records n configuration warnings under code. Nil-safe.
func (*Metrics) RecordConnectionOAuthCredentials ¶ added in v1.141.0
RecordConnectionOAuthCredentials records how many connections of a kind are in a credential state. Nil-safe.
func (*Metrics) RecordConnectionOAuthRefresh ¶ added in v1.141.0
RecordConnectionOAuthRefresh records one refresher attempt. Nil-safe.
func (*Metrics) RecordDataHubRequest ¶ added in v1.70.0
func (m *Metrics) RecordDataHubRequest(ctx context.Context, operation, status string, duration time.Duration)
RecordDataHubRequest records one DataHub request observation. operation is a bounded label (get_entity, get_schema, get_lineage, get_glossary_term, search_tables, ...); status is a bounded status constant. Nil-safe.
func (*Metrics) RecordEgressBlocked ¶ added in v1.141.0
RecordEgressBlocked counts one fetch the egress guard refused. Nil-safe.
func (*Metrics) RecordEmbeddingCall ¶ added in v1.141.0
RecordEmbeddingCall records one embedding call under its model. Nil-safe.
func (*Metrics) RecordEmbeddingFallback ¶ added in v1.141.0
RecordEmbeddingFallback counts one search ranked lexically. Nil-safe.
func (*Metrics) RecordEnrichmentBytes ¶ added in v1.99.0
func (m *Metrics) RecordEnrichmentBytes(ctx context.Context, attrs ToolCallAttrs, bytes int)
RecordEnrichmentBytes records the per-response cross-enrichment overhead in bytes, labeled by tool, toolkit_kind, and persona (issue #761). Status is deliberately omitted: enrichment only runs on successful results, so a status dimension would carry no signal. Nil-safe; callers skip the call when bytes is zero.
func (*Metrics) RecordGatewaySessionRedial ¶ added in v1.141.0
RecordGatewaySessionRedial counts one re-dial after a dropped session. Nil-safe.
func (*Metrics) RecordGatewayUpstreamCall ¶ added in v1.141.0
func (m *Metrics) RecordGatewayUpstreamCall(ctx context.Context, connection, outcome string, d time.Duration)
RecordGatewayUpstreamCall records one forwarded MCP gateway call. Nil-safe.
func (*Metrics) RecordHTTPClientRequest ¶ added in v1.141.0
func (m *Metrics) RecordHTTPClientRequest(ctx context.Context, kind, connection, statusClass string, d time.Duration)
RecordHTTPClientRequest records one outbound HTTP request. Nil-safe.
func (*Metrics) RecordHTTPRateLimited ¶ added in v1.141.0
RecordHTTPRateLimited counts one 429 under the limiter that answered it. Nil-safe.
func (*Metrics) RecordHTTPServerRequest ¶ added in v1.141.0
func (m *Metrics) RecordHTTPServerRequest(ctx context.Context, route, method, statusClass string, d time.Duration)
RecordHTTPServerRequest records one answered inbound request under its route template. Nil-safe. The route is the pattern the mux matched, resolved by the caller (internal/httpobs), and the method is clamped by HTTPMethodLabel.
func (*Metrics) RecordJWKSFetch ¶ added in v1.141.0
RecordJWKSFetch records one JWKS fetch and, on success, its time as the last success. Nil-safe.
func (*Metrics) RecordKnowledgeChange ¶ added in v1.141.0
RecordKnowledgeChange records one apply_knowledge outcome. Nil-safe.
func (*Metrics) RecordListenConnected ¶ added in v1.141.0
RecordListenConnected records a LISTEN connection coming up or going down. A connection that comes back up after being down counts a reconnect. Nil-safe.
func (*Metrics) RecordListenNotification ¶ added in v1.141.0
RecordListenNotification marks a LISTEN connection delivering a notification, which resets its clock. Nil-safe.
func (*Metrics) RecordLoopIteration ¶ added in v1.141.0
RecordLoopIteration records one background loop iteration: its result and duration, and, when it succeeded, the time it did. Nil-safe.
func (*Metrics) RecordMCPRequest ¶ added in v1.141.0
RecordMCPRequest records one MCP request other than tools/call. Nil-safe. method is the MCP method name the SDK dispatched, which is a closed set (shared.go's method table); status is StatusOK or an error status.
func (*Metrics) RecordNotificationAttempt ¶ added in v1.141.0
RecordNotificationAttempt records one delivery attempt by transport kind and result (delivered, retry, failed). Nil-safe.
func (*Metrics) RecordOAuthIssuance ¶ added in v1.70.0
RecordOAuthIssuance records one OAuth token issuance. grantType is the OAuth grant (authorization_code, client_credentials, refresh_token); status is a bounded status constant. Nil-safe.
func (*Metrics) RecordOAuthRefresh ¶ added in v1.70.0
RecordOAuthRefresh records one OAuth token refresh outcome and its latency. status is a bounded status constant. Nil-safe.
func (*Metrics) RecordOAuthRegistration ¶ added in v1.141.0
RecordOAuthRegistration records one dynamic client registration. Nil-safe.
func (*Metrics) RecordRateLimitQueued ¶ added in v1.126.5
RecordRateLimitQueued records one tools/call from a script principal that the per-user rate limiter held until a token was available rather than refused (issue #1534). A queued call is not a refusal and is not counted as one; this counter is what lets an operator see that a deployment's scripts are running against the limit. Nil-safe, like RecordRateLimited.
func (*Metrics) RecordRateLimited ¶ added in v1.102.0
RecordRateLimited records one authenticated tools/call refused by the per-user rate limiter (issue #929). Like RecordAuditEventDropped it carries no labels — a single scalar is enough to alert on tool-call shedding, and the throttled identity and tool live in the refusal's slog line, not in a high-cardinality metric. Nil-safe, so the middleware records unconditionally without an enabled check.
func (*Metrics) RecordRowsPurged ¶ added in v1.141.0
RecordRowsPurged records the rows one retention sweep deleted. Nil-safe and a no-op for zero.
func (*Metrics) RecordS3Operation ¶ added in v1.70.0
func (m *Metrics) RecordS3Operation(ctx context.Context, operation, status string, duration time.Duration)
RecordS3Operation records one S3 operation observation. operation is the S3 tool/op name (list_buckets, get_object, ...); status is a bounded status constant. Nil-safe.
func (*Metrics) RecordScriptAdmissionRefused ¶ added in v1.135.0
RecordScriptAdmissionRefused records the run worker declining to claim another run, for the reason it gave (#1843). Nil-safe.
func (*Metrics) RecordScriptMissedFires ¶ added in v1.121.0
RecordScriptMissedFires records fires the misfire policy stepped over for one script. Nil-safe, and a no-op for zero, so the caller can hand it whatever the materializer counted.
func (*Metrics) RecordScriptQueueWait ¶ added in v1.135.0
RecordScriptQueueWait records how long a claimed run waited after it became due. A negative wait (a clock skewed between replicas) is recorded as zero. Nil-safe.
func (*Metrics) RecordScriptRun ¶ added in v1.121.0
func (m *Metrics) RecordScriptRun(ctx context.Context, attrs ScriptRunAttrs, duration time.Duration)
RecordScriptRun records one managed-script run reaching a terminal state. Nil-safe.
It is recorded where the run finishes rather than where it is enqueued, so the count is of executions rather than intentions, and the duration is what the run actually took.
func (*Metrics) RecordScriptRunFailure ¶ added in v1.141.0
RecordScriptRunFailure counts a failed run by its cause. Nil-safe.
func (*Metrics) RecordScriptRunReclaim ¶ added in v1.135.1
RecordScriptRunReclaim counts one run found with its lease expired and no worker reporting, by what became of it (#1860). Nil-safe.
func (*Metrics) RecordScriptRunShed ¶ added in v1.141.0
RecordScriptRunShed counts a run stopped and requeued by memory shedding. Nil-safe.
func (*Metrics) RecordScriptWorkerLoad ¶ added in v1.141.0
RecordScriptWorkerLoad records the share of a resource limit the run worker's admission read. reason is memory or cpu. Nil-safe.
func (*Metrics) RecordSearchResults ¶ added in v1.141.0
RecordSearchResults records the hits one search returned. Nil-safe.
func (*Metrics) RecordSessionResolution ¶ added in v1.98.0
RecordSessionResolution records one tool call labeled by how its session was resolved (source: explicit, transport, stdio, or none). Nil-safe.
func (*Metrics) RecordStorageOperation ¶ added in v1.141.0
func (m *Metrics) RecordStorageOperation(ctx context.Context, o StorageOperation)
RecordStorageOperation records one object operation. Nil-safe.
func (*Metrics) RecordThumbnailFailure ¶ added in v1.141.0
RecordThumbnailFailure counts a document whose tile was not drawn, by why: document (the page threw or did not finish), renderer (the renderer was unavailable), storage (the file could not be read or the tile written), or undrawable (its attempts ran out). Nil-safe.
func (*Metrics) RecordThumbnailRender ¶ added in v1.141.0
RecordThumbnailRender records one attempt to draw a document's tiles and how long it took. Nil-safe.
func (*Metrics) RecordThumbnailRenderer ¶ added in v1.141.0
RecordThumbnailRenderer records whether the renderer answered. Nil-safe.
func (*Metrics) RecordToolCall ¶
RecordToolCall records one tool-call observation. Nil-safe. The source label is on the call counter only: the duration histogram's bucket series would otherwise multiply by it.
func (*Metrics) RecordToolDenial ¶ added in v1.141.0
RecordToolDenial records one tool call the authorizer refused. Nil-safe.
func (*Metrics) RecordTrinoQuery ¶ added in v1.70.0
func (m *Metrics) RecordTrinoQuery(ctx context.Context, status, queryKind string, duration time.Duration)
RecordTrinoQuery records one Trino query observation. status is one of the bounded status constants (see status.go); query_kind is a bounded label such as the SQL verb or the originating tool. Nil-safe.
func (*Metrics) RecordUpstreamRetry ¶ added in v1.141.0
RecordUpstreamRetry counts one request issued again, or given up on when exhausted is set. Nil-safe.
func (*Metrics) RegisterAuditQueueDepth ¶ added in v1.141.0
RegisterAuditQueueDepth installs the function the audit writer's depth is read from on a scrape. Nil-safe.
func (*Metrics) RegisterBackgroundQueue ¶ added in v1.141.0
func (m *Metrics) RegisterBackgroundQueue(loop string, s BackgroundQueueSampler)
RegisterBackgroundQueue installs the sampler a queue's gauges are read from, under the loop name of the queue's consumer. Nil-safe; a later call for the same loop replaces the earlier sampler.
func (*Metrics) RegisterDBPool ¶ added in v1.70.0
RegisterDBPool adds a *sql.DB to the set whose pool stats are reported on each scrape under the given pool label. Call once per managed handle at startup. Nil-safe (no-op when metrics are disabled or db is nil); ignores a duplicate pool name so a double-registration cannot double-observe the same series.
func (*Metrics) RegisterIndexQueue ¶ added in v1.134.1
func (m *Metrics) RegisterIndexQueue(s IndexQueueSampler)
RegisterIndexQueue installs the sampler the database gauges are read from. Nil-safe; a later call replaces the earlier sampler.
func (*Metrics) RegisterPlatformState ¶ added in v1.141.0
func (m *Metrics) RegisterPlatformState(s PlatformStateSampler)
RegisterPlatformState installs the sampler the platform-state gauges are read from. Nil-safe; a later call replaces the earlier sampler.
func (*Metrics) ScriptRunFinished ¶ added in v1.121.0
ScriptRunFinished is the other half of ScriptRunStarted.
func (*Metrics) ScriptRunStarted ¶ added in v1.121.0
ScriptRunStarted and ScriptRunFinished bracket a run executing on this replica. They are separate from RecordScriptRun because a run that never finishes never records one, and a worker wedged on a run is exactly what an operator needs to see. Nil-safe.
func (*Metrics) Shutdown ¶
Shutdown flushes the meter provider and releases resources. Safe to call on a nil receiver so cmd/main's shutdown path stays branch-free. Idempotent: the underlying OTel meter provider rejects a second Shutdown with "reader is shutdown", so subsequent calls return the first-call result (typically nil) instead of re-invoking the provider.
func (*Metrics) StartOp ¶ added in v1.141.0
StartOp opens operation name: a span under the trace ctx carries (none when ctx carries no trace) and the start of its duration. End it with End. Nil-safe on m: the span still opens, nothing is counted.
func (*Metrics) WebhookAck ¶ added in v1.136.0
WebhookAck records how long a request took to acknowledge. Nil-safe.
func (*Metrics) WebhookBuffer ¶ added in v1.136.0
WebhookBuffer reports the events a source holds in memory. Nil-safe.
func (*Metrics) WebhookCompaction ¶ added in v1.136.0
WebhookCompaction counts one compaction with its result. Nil-safe.
func (*Metrics) WebhookDuplicatesDropped ¶ added in v1.136.0
WebhookDuplicatesDropped counts events removed at compaction. Nil-safe.
func (*Metrics) WebhookEvents ¶ added in v1.136.0
WebhookEvents counts acknowledged events. Nil-safe.
func (*Metrics) WebhookRequest ¶ added in v1.136.0
WebhookRequest counts one request with its outcome. Nil-safe.
type MetricsExporter ¶ added in v1.141.0
type MetricsExporter string
MetricsExporter is the OTEL_METRICS_EXPORTER choice.
const ( MetricsExporterPrometheus MetricsExporter = "prometheus" MetricsExporterOTLP MetricsExporter = "otlp" MetricsExporterBoth MetricsExporter = "both" )
The MetricsExporter values.
func (MetricsExporter) OTLP ¶ added in v1.141.0
func (e MetricsExporter) OTLP() bool
OTLP reports whether metrics are pushed over OTLP.
func (MetricsExporter) Prometheus ¶ added in v1.141.0
func (e MetricsExporter) Prometheus() bool
Prometheus reports whether the /metrics listener is served.
type OTLPEndpoint ¶ added in v1.141.0
type OTLPEndpoint struct {
// Endpoint is OTEL_EXPORTER_OTLP_ENDPOINT in either form the OpenTelemetry
// specification defines: "host:port", or a URL ("http://collector:4317")
// whose scheme chooses TLS.
Endpoint string
// Insecure is OTEL_EXPORTER_OTLP_INSECURE when it was set. Nil leaves the
// choice to the form: plaintext for "host:port" (the common in-cluster
// topology), the scheme for a URL. Set, it wins over both.
Insecure *bool
}
OTLPEndpoint is the collector address every OTLP exporter dials, as the operator wrote it. One value, parsed once, feeds the trace, metric and log exporters so the three cannot read the same variable differently (#1893).
func OTLPEndpointFromEnv ¶ added in v1.141.0
func OTLPEndpointFromEnv() OTLPEndpoint
OTLPEndpointFromEnv reads OTEL_EXPORTER_OTLP_ENDPOINT (default DefaultOTLPEndpoint) and OTEL_EXPORTER_OTLP_INSECURE.
type Op ¶ added in v1.141.0
type Op struct {
// contains filtered or unexported fields
}
Op is one platform operation in progress: its span and its start. The zero value is not used; StartOp returns one.
func (*Op) End ¶ added in v1.141.0
End counts the operation under result (ok or error), records its duration and ends its span, recording err on it (redacted).
type PlatformStateSample ¶ added in v1.141.0
type PlatformStateSample struct {
Dependencies []DependencySample
Connections []ConnectionStateCount
// Personas is the number of registered personas; PersonasKnown is false
// when the sampler has no registry to count.
Personas int64
PersonasKnown bool
IndexAges []IndexAgeSample
}
PlatformStateSample is what the platform-state sampler reports on a scrape.
type PlatformStateSampler ¶ added in v1.141.0
type PlatformStateSampler func(ctx context.Context) PlatformStateSample
PlatformStateSampler returns the platform's state. It is called on each scrape with a bounded context and must answer from a cache: a probe of a slow dependency must never hold a scrape.
type ScriptRunAttrs ¶ added in v1.121.0
ScriptRunAttrs identifies one managed-script run observation. Script is the script's NAME rather than its id, because a metric is read by a person: the cardinality is the number of scripts a deployment has, which is bounded by what humans wrote and reviewed.
type StorageOperation ¶ added in v1.141.0
type StorageOperation struct {
// Purpose is one of the StoragePurpose* values; Operation is put, get,
// get_range, list or delete.
Purpose, Operation string
// Reason is empty for a success and one of the StorageReason* classes for
// a failure.
Reason string
Duration time.Duration
// Written is the bytes a put stored.
Written int64
}
StorageOperation is one object operation on a bucket the platform owns.
type ToolCallAttrs ¶
type ToolCallAttrs struct {
Tool string
ToolkitKind string
Persona string
StatusCategory string
Source string
}
ToolCallAttrs is the bounded label set for tool-call metrics. The metrics layer never reads request bodies, user identifiers, or session IDs — those are span attributes (phase 2) and audit log fields, not Prometheus labels. Persona is bounded by the deployment's persona definitions and records MetricLabelUnknown when the call never reached persona resolution, which is the same value the api-gateway's outbound counter records for the same case (#1615). Source is how the call arrived (the audit event's source: mcp, admin, rest, script), so an agent's traffic, a portal replay, the REST shim and a managed script's run are separable; it is on the call counter only.
type Tracer ¶ added in v1.87.0
type Tracer struct {
// contains filtered or unexported fields
}
Tracer owns the OTel TracerProvider and OTLP exporter for distributed tracing. A nil *Tracer is a valid no-op: Start returns a no-op span and Shutdown is a no-op, so callers hold and use it unconditionally.
Unlike Metrics (which builds an isolated Prometheus registry), Tracer installs the GLOBAL OTel TracerProvider and propagator in NewTracer. That lets any call site — middleware steps, toolkit adapters — create child spans with otel.Tracer(InstrumentationScope) and have them nest into the request's span tree via context, without injecting a *Tracer everywhere. When tracing is disabled the global provider stays the default no-op, so those call sites cost nothing.
func NewTracer ¶ added in v1.87.0
func NewTracer(cfg TracingConfig) (*Tracer, error)
NewTracer builds a Tracer from cfg. When cfg.Enabled is false it returns (nil, nil) so callers receive a no-op recorder without an error path, mirroring metrics New.
The OTLP/gRPC exporter connects lazily: NewTracer does not block or fail when the collector is unreachable, so an unconfigured or down collector never delays or breaks platform startup. Spans are batched and dropped if they cannot be delivered.
func NewTracerFromProvider ¶ added in v1.87.0
func NewTracerFromProvider(provider *sdktrace.TracerProvider, cfg TracingConfig) *Tracer
NewTracerFromProvider wraps an already-constructed TracerProvider and installs it (plus the W3C/baggage propagator) as the OTel global. It is the seam NewTracer uses after building the OTLP provider, and the one tests use to inject an in-memory span recorder. Always returns a non-nil *Tracer.
func (*Tracer) Enabled ¶ added in v1.87.0
Enabled reports whether tracing is active. Call sites that build expensive span attributes can gate on this; Start itself is nil-safe.
func (*Tracer) IncludeUserEmail ¶ added in v1.141.0
IncludeUserEmail reports whether the tool-call span may carry the caller's email address (OTEL_TRACES_INCLUDE_USER_EMAIL, default false). The user id is always on the span; the address is personal data a trace backend would otherwise hold for every call (#1892). Nil-safe: a disabled tracer says no.
func (*Tracer) SamplerRatio ¶ added in v1.141.0
SamplerRatio is the head-sampling ratio applied to root spans, or 0 when tracing is off. Nil-safe.
type TracingConfig ¶ added in v1.87.0
type TracingConfig struct {
// Enabled gates the entire subsystem. When false, NewTracer returns
// a nil *Tracer whose methods are no-ops and no exporter, provider,
// or global TracerProvider override is constructed.
Enabled bool
// OTLP is the collector the spans are exported to; shared with the
// metrics OTLP reader and the log provider.
OTLP OTLPEndpoint
// SamplerArg is the head-based sampling ratio for root spans, [0,1].
SamplerArg float64
// IncludeUserEmail lets the tool-call span carry mcp.user_email.
IncludeUserEmail bool
}
TracingConfig holds the operator-configurable knobs for the tracing subsystem. Environment-only, mirroring the metrics Config.
func TracingConfigFromEnv ¶ added in v1.87.0
func TracingConfigFromEnv() TracingConfig
TracingConfigFromEnv reads the tracing configuration from environment variables. Tracing is disabled unless OTEL_TRACES_ENABLED is truthy. Unset or unparsable values fall back to defaults so a partial configuration still boots.