telemetry

package
v0.5.0-rc.2 Latest Latest
Warning

This package is not in the latest version of its module.

Go to latest
Published: Sep 4, 2026 License: MIT Imports: 21 Imported by: 0

Documentation

Index

Constants

This section is empty.

Variables

This section is empty.

Functions

func PrometheusHandler

func PrometheusHandler() http.Handler

Types

type AuthnMetrics added in v0.5.0

type AuthnMetrics struct {
	// TenantConflictsTotal counts client-asserted tenants that an
	// authenticated binding overrode. Labels: surface (http|ws|grpc), reason
	// (header|query|metadata|resource_attribute). A non-zero rate means a
	// client is sending a tenant it is not entitled to — misconfiguration at
	// best, a probe at worst.
	TenantConflictsTotal *prometheus.CounterVec

	// GRPCAuthFailuresTotal counts rejected gRPC calls by reason
	// (missing_header|bad_scheme|bad_key).
	GRPCAuthFailuresTotal *prometheus.CounterVec
}

AuthnMetrics covers the authenticated-tenant-identity surfaces (HTTP, WebSocket, gRPC). It lives apart from Metrics because it is wired from package-level hooks in authn/api/ingest rather than passed down the call graph, and because a duplicate registration must be impossible even when a test constructs the platform twice.

func NewAuthnMetrics added in v0.5.0

func NewAuthnMetrics() *AuthnMetrics

NewAuthnMetrics returns the process-wide authn metric set, registering the collectors on first call.

type HealthStats

type HealthStats struct {
	IngestionRate     int64               `json:"ingestion_rate"`
	DLQSize           int64               `json:"dlq_size"`
	ActiveConns       int64               `json:"active_connections"`
	DBLatencyP99Ms    float64             `json:"db_latency_p99_ms"`
	DBLatencyLastMs   float64             `json:"db_latency_last_ms"`
	LatencyProvenance *latency.Provenance `json:"latency_provenance,omitempty"`
	Goroutines        int                 `json:"goroutines"`
	HeapAllocMB       float64             `json:"heap_alloc_mb"`
	UptimeSeconds     float64             `json:"uptime_seconds"`
}

HealthStats is the JSON response for GET /api/health.

type Metrics

type Metrics struct {
	// --- Existing ---
	IngestionRate     prometheus.Counter
	ActiveConnections prometheus.Gauge
	DBLatency         prometheus.Histogram
	DLQSize           prometheus.Gauge

	// IngestDurationSeconds is the per-Export E2E latency observed inside
	// the OTLP servers (gRPC + HTTP), labeled by signal {traces,logs,metrics}.
	// Drives ingest SLOs: alert on p99 / error budget burn rather than on the
	// blunt OtelContext_grpc_request_duration_seconds aggregate.
	IngestDurationSeconds *prometheus.HistogramVec

	// --- gRPC ---
	GRPCRequestsTotal   *prometheus.CounterVec
	GRPCRequestDuration *prometheus.HistogramVec
	GRPCBatchSize       prometheus.Histogram

	// --- HTTP ---
	HTTPRequestsTotal   *prometheus.CounterVec
	HTTPRequestDuration *prometheus.HistogramVec

	// --- TSDB ---
	TSDBIngestTotal         prometheus.Counter
	TSDBFlushDuration       prometheus.Histogram
	TSDBBatchesDropped      prometheus.Counter
	TSDBCardinalityOverflow prometheus.Counter
	// TSDBCardinalityOverflowByTenant labels overflow events with the tenant ID
	// that triggered them, or the sentinel "__global__" when the global cap
	// (not a per-tenant cap) was the trigger. Use this to identify noisy
	// tenants: sum by (tenant_id) (rate(otelcontext_tsdb_cardinality_overflow_by_tenant_total[5m]))
	TSDBCardinalityOverflowByTenant *prometheus.CounterVec

	// --- WebSocket ---
	WSMessagesSent       *prometheus.CounterVec
	WSSlowClientsRemoved prometheus.Counter

	// --- DLQ ---
	DLQEnqueuedTotal prometheus.Counter
	DLQReplaySuccess prometheus.Counter
	DLQReplayFailure prometheus.Counter
	DLQDiskBytes     prometheus.Gauge

	// --- Storage ---
	HotDBSizeBytes prometheus.Gauge

	// --- Retention ---
	RetentionRowsPurgedTotal       *prometheus.CounterVec
	RetentionPurgeDurationSeconds  *prometheus.HistogramVec
	RetentionVacuumDurationSeconds *prometheus.HistogramVec
	RetentionRowsBehindGauge       *prometheus.GaugeVec

	// --- Postgres partitioning (DB_POSTGRES_PARTITIONING=daily) ---
	// PartitionsDropped counts daily logs partitions dropped during the
	// retention pass. Each drop is a near-instant DDL — alert when this
	// counter is flat for >1.5 retention periods (indicates a stuck loop).
	PartitionsDropped prometheus.Counter
	// PartitionsActive gauges the live partitions attached to logs.
	// Healthy steady-state ~ HOT_RETENTION_DAYS + DB_PARTITION_LOOKAHEAD_DAYS + 1.
	PartitionsActive prometheus.Gauge

	// --- Runtime ---
	GoGoroutines     prometheus.Gauge
	GoHeapAllocBytes prometheus.Gauge

	// --- Operational (Fix 6) ---
	PanicsRecoveredTotal          *prometheus.CounterVec
	MCPToolInvocationsTotal       *prometheus.CounterVec
	APIAuthFailuresTotal          *prometheus.CounterVec
	GraphRAGEventBufferDepth      prometheus.Gauge
	RetentionLastSuccessTimestamp *prometheus.GaugeVec
	RetentionConsecutiveFailures  *prometheus.GaugeVec
	DBUp                          *prometheus.GaugeVec

	// --- GraphRAG overflow ---
	GraphRAGEventsDroppedTotal *prometheus.CounterVec

	// GraphRAGTenantsEvictedTotal counts tenant store slices evicted after
	// exceeding GRAPHRAG_TENANT_IDLE_TTL. The default tenant is never
	// evicted; a steady non-zero rate on a single-tenant install means
	// rogue tenant IDs are reaching ingest.
	GraphRAGTenantsEvictedTotal prometheus.Counter

	// --- In-memory store census (OOM-survival work) ---
	// GraphRAGStoreEntities — live node counts per entity kind across tenants
	// (tenants|services|operations|traces|spans|log_clusters|metrics|anomalies).
	// GraphRAGStoreEdges — live edge counts per store (service|trace|signal|anomaly).
	// Together with the ring/drain gauges these attribute RSS growth to a
	// specific structure before a heap profile is needed.
	GraphRAGStoreEntities *prometheus.GaugeVec
	GraphRAGStoreEdges    *prometheus.GaugeVec
	TSDBRingSeriesActive  prometheus.Gauge
	// TSDBRingSeriesRejected — points refused a NEW ring series at the
	// tenant-scoped series cap (existing series keep recording).
	TSDBRingSeriesRejected prometheus.Counter
	DrainTemplatesActive   prometheus.Gauge

	// --- Async ingest pipeline (Phase 1 robustness work) ---
	// IngestPipelineQueueDepth — current queue depth, sampled on every Submit.
	// Labeled by signal so spikes can be attributed to traces vs logs.
	IngestPipelineQueueDepth *prometheus.GaugeVec
	// IngestPipelineQueueBytes — approximate bytes held by queued batches.
	// Reserved at Submit, released when a worker finishes the batch; the
	// byte cap (INGEST_PIPELINE_MAX_BYTES) rejects submissions above it.
	IngestPipelineQueueBytes prometheus.Gauge
	// IngestPipelineDroppedTotal — batches that did NOT reach the DB.
	// reason="soft_backpressure" — healthy batch dropped at >=90% fullness.
	// reason="queue_full"        — batch rejected at 100% capacity (client got 429/RESOURCE_EXHAUSTED).
	// reason="bytes_full"        — batch rejected at the byte cap (even priority batches).
	IngestPipelineDroppedTotal *prometheus.CounterVec
	// IngestPipelineDLQTotal — batches handed to the Dead Letter Queue after
	// the persist transaction failed, instead of being dropped silently.
	// result="enqueued"      — the complete batch is on disk awaiting replay.
	// result="enqueue_failed" — the DLQ itself rejected the write (disk full,
	//                           permissions); the batch IS lost.
	// result="no_sink"       — no DLQ wired into the pipeline; the batch IS lost.
	IngestPipelineDLQTotal *prometheus.CounterVec

	// ExemplarSubmitTotal — outcome of every raw-exemplar batch submission in
	// AGGREGATE_MODE=aggregate, where the durable aggregate commit is the
	// Export ACK and raw exemplar storage is bounded best-effort (#196).
	// outcome="queued" reason="none"        — the batch entered the raw pipeline.
	// outcome="dlq"    reason="queue_full"  — pipeline saturated, DLQ accepted
	//                                         the batch. Deferred, NOT lost.
	// outcome="lost"   reason="dlq_full"    — the DLQ refused it for capacity.
	// outcome="lost"   reason="dlq_error"   — no DLQ wired, or its write failed.
	// lost{reason="queue_full"} is never emitted: on the lost outcome the
	// reason names why the DLQ could not hold the batch, not why the primary
	// queue refused it. Intentional soft-backpressure drops are not counted
	// here — they are already on IngestPipelineDroppedTotal.
	ExemplarSubmitTotal *prometheus.CounterVec
	// ExemplarSubmitLostTotal — dedicated counter for the permanent-loss
	// subset of ExemplarSubmitTotal, so an alert can target loss without a
	// label matcher. reason="dlq_full"|"dlq_error".
	ExemplarSubmitLostTotal *prometheus.CounterVec

	// --- OTLP metric completeness (#199) ---
	// IngestMetricsUnsupportedTotal — metric data points the aggregate path
	// refused outright and reported in ExportMetricsPartialSuccess. Labeled by
	// the OTLP point type (summary|histogram|exponential_histogram) and the
	// reason: cumulative_temporality, unspecified_temporality, unsupported_type
	// or malformed_point. Every increment is a point that did NOT enter
	// aggregate accounting, and the client must not retry it.
	IngestMetricsUnsupportedTotal *prometheus.CounterVec
	// IngestMetricsSketchDroppedTotal — histogram points whose SCALARS were
	// kept but whose percentiles are unavailable, by reason:
	// negative_observations (negative buckets or a bucket that spans negative
	// values without a proving min>=0), scale_out_of_range (an
	// ExponentialHistogram below scale 0, which the positive-only scale-4
	// sketch cannot represent) or no_finite_boundaries.
	// This is NOT a rejection: count/sum/min/max still land in the aggregate.
	IngestMetricsSketchDroppedTotal *prometheus.CounterVec
	// IngestMetricsDimsRejectedTotal — metric points whose configured
	// dimension tuple was refused from series identity, by reason. Today the
	// only reason is unsupported_value_type: an array or kvlist attribute
	// value has no canonical scalar rendering, so it cannot be interned.
	// The point is still aggregated, under DimsID=0.
	IngestMetricsDimsRejectedTotal *prometheus.CounterVec
	// IngestReservedServicePrefixTotal — resources whose client-declared
	// service.name sits inside the reserved host/ namespace (#280). The name
	// is accepted as sent; the count tells an operator a client is minting
	// host entities by hand. Label signal=traces|logs|metrics.
	IngestReservedServicePrefixTotal *prometheus.CounterVec

	// HTTPOTLPThrottledTotal — count of HTTP 429s issued by the OTLP HTTP
	// receiver when the async ingest pipeline is full. Mirrors the gRPC
	// RESOURCE_EXHAUSTED path so operators see a single throttling signal
	// across both transports. Label `signal` is one of traces|logs|metrics.
	HTTPOTLPThrottledTotal *prometheus.CounterVec

	// --- DB pool (sampled every 5s from sql.DB.Stats) ---
	DBPoolOpenConnections prometheus.Gauge
	DBPoolInUse           prometheus.Gauge
	DBPoolIdle            prometheus.Gauge
	DBPoolWaitCount       prometheus.Gauge
	DBPoolWaitDuration    prometheus.Gauge // cumulative seconds

	// --- DLQ eviction (Task 8) ---
	DLQEvictedTotal      prometheus.Counter
	DLQEvictedBytesTotal prometheus.Counter

	// --- Dashboard p99 (Task 10) ---
	DashboardP99RowCapHitsTotal prometheus.Counter

	// --- Aggregate engine (AGGREGATE_MODE != legacy) ---
	// AggregateInputPointsTotal — points offered to the request-local reducer
	// per signal, counted BEFORE the sampler and severity gates. This is
	// accepted telemetry, not persisted telemetry.
	AggregateInputPointsTotal *prometheus.CounterVec
	// AggregateDeltasTotal — series deltas emitted by reduction. The gap
	// between this and input points is the whole value of the engine.
	AggregateDeltasTotal *prometheus.CounterVec
	// AggregateReductionRatio — input points per emitted delta, per Export
	// request. A ratio collapsing toward 1 means cardinality is exploding.
	AggregateReductionRatio *prometheus.HistogramVec
	// AggregateLatePointsTotal — points excluded from aggregates because they
	// fell outside the mutable-window horizon. reason="late" (older than the
	// allowed lateness) or reason="future" (beyond the tolerated skew).
	AggregateLatePointsTotal *prometheus.CounterVec
	// AggregateSeriesActive — budgeted series present in at least one mutable
	// window, per signal. This is what the AGGREGATE_MAX_SERIES* caps bound,
	// and it never exceeds them: the __other__ series a cap mints when it
	// binds are the reserve, counted by AggregateOverflowSeriesActive instead.
	AggregateSeriesActive *prometheus.GaugeVec
	// AggregateOverflowSeriesActive — live __other__ series per signal. This
	// is the unbudgeted reserve the caps spend; it is bounded by
	// (services x signals x status classes), not by AGGREGATE_MAX_SERIES*.
	AggregateOverflowSeriesActive *prometheus.GaugeVec
	// AggregateOverflowTotal — admissions rerouted to an __other__ series,
	// labeled by the cap that triggered it (tenant|service_names|
	// service_series|signal|global). Totals are preserved; identity is not.
	AggregateOverflowTotal *prometheus.CounterVec
	// AggregateShadowAcceptedTotal — telemetry accounted on the aggregate
	// side per signal. In shadow mode this is compared against the legacy
	// path's accepted counts; it must not move with the sampling rate.
	AggregateShadowAcceptedTotal *prometheus.CounterVec
	// AggregateShadowErrorsTotal — errors accounted on the aggregate side per
	// service. Cheap invariant only (#165): no per-series comparison.
	AggregateShadowErrorsTotal *prometheus.CounterVec
	// AggregateClosedWindows — windows past their lateness horizon that
	// memory still holds because the finalizer has not materialized them into
	// aggregate_buckets yet. Steady state is 0 or 1; a value that stays high
	// means finalization is behind or failing.
	AggregateClosedWindows prometheus.Gauge
	// AggregateClosedWindowsEvictedTotal — closed windows the closed-window
	// cap forced out of memory before finalization. Each one is lost data:
	// alert on any increase.
	AggregateClosedWindowsEvictedTotal prometheus.Counter

	// --- Durable aggregate store (#173) ---
	// AggregateCommitDurationSeconds — group-commit wall time. This IS the
	// ACK latency floor: an Export cannot return before its commit does.
	AggregateCommitDurationSeconds *prometheus.HistogramVec
	// AggregateCommitDeltas — delta rows per group commit. The pre-merge
	// ratio and the coalescing behaviour both show up here.
	AggregateCommitDeltas prometheus.Histogram
	// AggregateCommitsTotal — commits by result (ok|error).
	AggregateCommitsTotal *prometheus.CounterVec
	// AggregateCommitBytesTotal — delta payload written to the store.
	AggregateCommitBytesTotal prometheus.Counter
	// AggregateAdmissionRejectedTotal — ErrSaturated refusals by the bound
	// that tripped (bytes|waiters|deltas). Non-zero at sustained load is a
	// release-gate failure, not a tuning hint.
	AggregateAdmissionRejectedTotal *prometheus.CounterVec
	// AggregateFinalizeDurationSeconds — window finalization wall time.
	AggregateFinalizeDurationSeconds prometheus.Histogram
	// AggregateFinalizeRowsTotal — rows materialized/deleted by finalization,
	// by kind (buckets|deltas).
	AggregateFinalizeRowsTotal *prometheus.CounterVec
	// AggregatePurgeDurationSeconds — retention purge wall time on the
	// aggregate DB.
	AggregatePurgeDurationSeconds prometheus.Histogram
	// AggregatePurgeRowsTotal — rows purged by kind (buckets|deltas|baselines).
	AggregatePurgeRowsTotal *prometheus.CounterVec
	// AggregateDeltaLogRows and AggregateDeltaLogAgeSeconds are the delta-log
	// backlog health bounds from #160: alert when either climbs.
	AggregateDeltaLogRows       prometheus.Gauge
	AggregateDeltaLogAgeSeconds prometheus.Gauge
	// AggregateRecoveryDurationSeconds and AggregateRecoveryRows describe the
	// last startup recovery. The gate allows 30s.
	AggregateRecoveryDurationSeconds prometheus.Gauge
	AggregateRecoveryRows            *prometheus.GaugeVec

	// --- Aggregate identity lifecycle (#200) ---
	// AggregateGCRunsTotal — identity garbage-collection passes by result
	// (ok|error). A pass that fails leaves memory untouched, so a rising
	// error count is disk growth, not corruption.
	AggregateGCRunsTotal *prometheus.CounterVec
	// AggregateGCDurationSeconds — GC wall time by phase. phase="mark" is the
	// lock-free scan; phase="barrier" is the part that serializes with the
	// group commit and is therefore the only one inside the ACK budget.
	AggregateGCDurationSeconds *prometheus.HistogramVec
	// AggregateGCSweptTotal — identity rows deleted by GC, by table.
	AggregateGCSweptTotal *prometheus.CounterVec
	// AggregateGCRetained — identity rows the last pass kept, by table. The
	// ratio against the swept counter is what says whether the dictionary has
	// reached a steady state.
	AggregateGCRetained *prometheus.GaugeVec
	// AggregateIdentityOverflowTotal — identities routed to __other__ by an
	// identity BOUND rather than by a series cap, labeled by dictionary kind
	// and the bound that tripped (length|count).
	AggregateIdentityOverflowTotal *prometheus.CounterVec
	// AggregateTenantRejectedTotal — points DROPPED because their tenant
	// identity was refused. The tenant namespace never collapses into a
	// shared __other__, so this is a drop, not a degradation: alert on any
	// sustained value.
	AggregateTenantRejectedTotal *prometheus.CounterVec
	// ResourceRegistryEntries — live resource registry entries per tenant.
	// kind=pair counts service-host-workload entries, kind=host counts
	// distinct non-empty hosts (#279).
	ResourceRegistryEntries *prometheus.GaugeVec
	// ResourceRegistryOverflowTotal — registrations DROPPED by a per-tenant
	// registry bound (kind=host|pair). Never merged into a shared host.
	ResourceRegistryOverflowTotal *prometheus.CounterVec
	// --- Bounded exemplar retention (AGGREGATE_MODE=aggregate) ---
	// ExemplarEligibleTotal — telemetry that qualified for raw retention,
	// per signal and priority class. Eligible is not retained: the gap between
	// this and the drop counter is what makes aggregate completeness and raw
	// diagnostic coverage distinguishable during a storm (#161).
	ExemplarEligibleTotal *prometheus.CounterVec
	// ExemplarDroppedTotal — eligible telemetry refused raw persistence,
	// reason=budget_count|budget_bytes|stratum.
	ExemplarDroppedTotal *prometheus.CounterVec
	// ExemplarEvictionTotal — selected exemplars displaced by a better-ranked
	// trace. Each eviction is one trace's worth of bounded OVER-retention:
	// already-persisted spans are never deleted.
	ExemplarEvictionTotal prometheus.Counter
	// ExemplarTruncatedTotal — retained traces forced past their max spans or
	// max bytes. These persist truncated=true plus retained/observed counts,
	// and causal-analysis tools report partial coverage for them (#163).
	ExemplarTruncatedTotal prometheus.Counter

	// --- 8 GiB data budget and disk watchdog (#201 Q1/Q5) ---
	// DiskBudgetBytes — the ENFORCEMENT ceiling actually in effect: the lower
	// of DATA_DISK_BUDGET_MB and the usable volume capacity. A volume smaller
	// than the configured budget does not grow because the config says so.
	DiskBudgetBytes prometheus.Gauge
	// DiskUsedBytes / DiskUsedRatio — statfs allocation on the data volume and
	// its fraction of DiskBudgetBytes. This pair, not summed file sizes, is
	// what the shedding ladder reads.
	DiskUsedBytes prometheus.Gauge
	DiskUsedRatio prometheus.Gauge
	// DiskComponentBytes / DiskComponentHighWaterBytes — per-tier attribution
	// against the budget table (main_db, aggregate_db, dlq, wal). The
	// high-water gauge is what the seven-day gate (#202) validates the table
	// against; the instantaneous gauge alone hides the peak that mattered.
	DiskComponentBytes          *prometheus.GaugeVec
	DiskComponentHighWaterBytes *prometheus.GaugeVec
	// DiskSheddingState — 0 none, 1 errors_only, 2 raw_off.
	DiskSheddingState prometheus.Gauge
	// DiskSheddingTransitionsTotal — every state change, labeled from/to.
	// Flapping is visible here and nowhere else.
	DiskSheddingTransitionsTotal *prometheus.CounterVec
	// ExemplarRowsPurgedTotal / ExemplarPurgeDurationSeconds — throughput of
	// the exemplar-tier purge (EXEMPLAR_RETENTION_DAYS), separate from the
	// HOT_RETENTION_DAYS counters so the two retentions are distinguishable.
	ExemplarRowsPurgedTotal      *prometheus.CounterVec
	ExemplarPurgeDurationSeconds prometheus.Histogram
	// contains filtered or unexported fields
}

Metrics holds all internal Prometheus metrics for OtelContext self-monitoring.

func New

func New() *Metrics

New creates and registers all OtelContext internal metrics.

func (*Metrics) DecrementActiveConns

func (m *Metrics) DecrementActiveConns()

func (*Metrics) DisableTSDBCollectors added in v0.5.0

func (m *Metrics) DisableTSDBCollectors()

DisableTSDBCollectors unregisters the collectors that only the legacy TSDB aggregator and ring buffer can move. Call it exactly once, at startup, when that path is not constructed (AGGREGATE_MODE=aggregate, #194 finding 10).

Leaving them registered would publish TSDB ingest, drop and cardinality series pinned at 0, which a dashboard reads as "no overflow" rather than "no TSDB". The aggregate engine reports its own admission and cardinality caps; these must not shadow them. The struct fields stay non-nil so any residual call site is inert rather than a nil dereference.

func (*Metrics) GetHealthStats

func (m *Metrics) GetHealthStats() HealthStats

func (*Metrics) HealthHandler

func (m *Metrics) HealthHandler() http.HandlerFunc

func (*Metrics) HealthWSHandler

func (m *Metrics) HealthWSHandler() http.HandlerFunc

HealthWSHandler returns an HTTP handler that upgrades to WebSocket and pushes HealthStats snapshots every 3 seconds. An immediate snapshot is sent on connection so the client never has to wait for the first tick.

func (*Metrics) IncrementActiveConns

func (m *Metrics) IncrementActiveConns()

func (*Metrics) ObserveDBLatency

func (m *Metrics) ObserveDBLatency(seconds float64)

func (*Metrics) ObserveIngestDuration

func (m *Metrics) ObserveIngestDuration(signal string, d time.Duration)

ObserveIngestDuration records an end-to-end OTLP Export latency for the given signal. Callers should pass time.Since(start) measured from the very start of the Export handler. Nil-safe so the OTLP servers can be wired without a Metrics instance during tests.

func (*Metrics) RecordExemplarDropped added in v0.5.0

func (m *Metrics) RecordExemplarDropped(signal, reason string)

RecordExemplarDropped implements ingest.ExemplarMetrics.

func (*Metrics) RecordExemplarEligible added in v0.5.0

func (m *Metrics) RecordExemplarEligible(signal, class string)

RecordExemplarEligible implements ingest.ExemplarMetrics.

func (*Metrics) RecordExemplarEviction added in v0.5.0

func (m *Metrics) RecordExemplarEviction()

RecordExemplarEviction implements ingest.ExemplarMetrics.

func (*Metrics) RecordExemplarTruncation added in v0.5.0

func (m *Metrics) RecordExemplarTruncation()

RecordExemplarTruncation implements ingest.ExemplarMetrics.

func (*Metrics) RecordIngestion

func (m *Metrics) RecordIngestion(count int)

func (*Metrics) RecordMetricDimsRejected added in v0.5.0

func (m *Metrics) RecordMetricDimsRejected(reason string, n uint64)

RecordMetricDimsRejected counts metric points whose dimension tuple was refused from series identity.

func (*Metrics) RecordMetricSketchDropped added in v0.5.0

func (m *Metrics) RecordMetricSketchDropped(reason string)

RecordMetricSketchDropped counts one histogram point kept for its scalars with percentiles suppressed.

func (*Metrics) RecordMetricUnsupported added in v0.5.0

func (m *Metrics) RecordMetricUnsupported(pointType, reason string, n int)

RecordMetricUnsupported counts one OTLP metric data point refused by the aggregate path. pointType is the OTLP point type, reason names why. A nil *Metrics is a no-op: every ingest unit test passes one.

func (*Metrics) RecordReservedServicePrefix added in v0.5.0

func (m *Metrics) RecordReservedServicePrefix(signal string)

RecordReservedServicePrefix counts one resource whose client-declared service.name starts with the reserved host/ prefix. Nil-safe.

func (*Metrics) RecordResourceRegistryOverflow added in v0.5.0

func (m *Metrics) RecordResourceRegistryOverflow(tenant, kind string)

RecordResourceRegistryOverflow counts one registration refused by the tenant's registry bound of kind (host|pair). Nil-safe.

func (*Metrics) RegisterReadCache added in v0.5.0

func (m *Metrics) RegisterReadCache(name string, size func() int)

RegisterReadCache exposes a read cache's live entry count as otelcontext_read_cache_entries{cache=name}. Safe on a nil receiver.

func (*Metrics) SampleDBPoolStats

func (m *Metrics) SampleDBPoolStats(sqlDB *sql.DB)

SampleDBPoolStats writes the live pool stats into the DBPool* gauges. Safe to call from a ticker goroutine. A nil receiver or a nil *sql.DB is a no-op so callers don't need to guard at every call site.

WaitCount and WaitDuration from sql.DBStats are cumulative values (always monotonically increasing) — operators should compute rate() over them.

func (*Metrics) SetActiveConnections

func (m *Metrics) SetActiveConnections(n int)

func (*Metrics) SetDLQSize

func (m *Metrics) SetDLQSize(n int)

func (*Metrics) SetResourceRegistryEntries added in v0.5.0

func (m *Metrics) SetResourceRegistryEntries(tenant, kind string, n int)

SetResourceRegistryEntries publishes the tenant's live registry count of kind (host|pair). Nil-safe.

func (*Metrics) StartRuntimeMetrics

func (m *Metrics) StartRuntimeMetrics()

StartRuntimeMetrics samples Go runtime stats every 15 seconds.

Jump to

Keyboard shortcuts

? : This menu
/ : Search site
f or F : Jump to
y or Y : Canonical URL