Documentation
¶
Index ¶
- func PrometheusHandler() http.Handler
- type AuthnMetrics
- type HealthStats
- type Metrics
- func (m *Metrics) DecrementActiveConns()
- func (m *Metrics) DisableTSDBCollectors()
- func (m *Metrics) GetHealthStats() HealthStats
- func (m *Metrics) HealthHandler() http.HandlerFunc
- func (m *Metrics) HealthWSHandler() http.HandlerFunc
- func (m *Metrics) IncrementActiveConns()
- func (m *Metrics) ObserveDBLatency(seconds float64)
- func (m *Metrics) ObserveIngestDuration(signal string, d time.Duration)
- func (m *Metrics) RecordExemplarDropped(signal, reason string)
- func (m *Metrics) RecordExemplarEligible(signal, class string)
- func (m *Metrics) RecordExemplarEviction()
- func (m *Metrics) RecordExemplarTruncation()
- func (m *Metrics) RecordIngestion(count int)
- func (m *Metrics) RecordMetricDimsRejected(reason string, n uint64)
- func (m *Metrics) RecordMetricSketchDropped(reason string)
- func (m *Metrics) RecordMetricUnsupported(pointType, reason string, n int)
- func (m *Metrics) RecordReservedServicePrefix(signal string)
- func (m *Metrics) RecordResourceRegistryOverflow(tenant, kind string)
- func (m *Metrics) RegisterReadCache(name string, size func() int)
- func (m *Metrics) SampleDBPoolStats(sqlDB *sql.DB)
- func (m *Metrics) SetActiveConnections(n int)
- func (m *Metrics) SetDLQSize(n int)
- func (m *Metrics) SetResourceRegistryEntries(tenant, kind string, n int)
- func (m *Metrics) StartRuntimeMetrics()
Constants ¶
This section is empty.
Variables ¶
This section is empty.
Functions ¶
func PrometheusHandler ¶
Types ¶
type AuthnMetrics ¶ added in v0.5.0
type AuthnMetrics struct {
// TenantConflictsTotal counts client-asserted tenants that an
// authenticated binding overrode. Labels: surface (http|ws|grpc), reason
// (header|query|metadata|resource_attribute). A non-zero rate means a
// client is sending a tenant it is not entitled to — misconfiguration at
// best, a probe at worst.
TenantConflictsTotal *prometheus.CounterVec
// GRPCAuthFailuresTotal counts rejected gRPC calls by reason
// (missing_header|bad_scheme|bad_key).
GRPCAuthFailuresTotal *prometheus.CounterVec
}
AuthnMetrics covers the authenticated-tenant-identity surfaces (HTTP, WebSocket, gRPC). It lives apart from Metrics because it is wired from package-level hooks in authn/api/ingest rather than passed down the call graph, and because a duplicate registration must be impossible even when a test constructs the platform twice.
func NewAuthnMetrics ¶ added in v0.5.0
func NewAuthnMetrics() *AuthnMetrics
NewAuthnMetrics returns the process-wide authn metric set, registering the collectors on first call.
type HealthStats ¶
type HealthStats struct {
IngestionRate int64 `json:"ingestion_rate"`
DLQSize int64 `json:"dlq_size"`
ActiveConns int64 `json:"active_connections"`
DBLatencyP99Ms float64 `json:"db_latency_p99_ms"`
DBLatencyLastMs float64 `json:"db_latency_last_ms"`
LatencyProvenance *latency.Provenance `json:"latency_provenance,omitempty"`
Goroutines int `json:"goroutines"`
HeapAllocMB float64 `json:"heap_alloc_mb"`
UptimeSeconds float64 `json:"uptime_seconds"`
}
HealthStats is the JSON response for GET /api/health.
type Metrics ¶
type Metrics struct {
// --- Existing ---
IngestionRate prometheus.Counter
ActiveConnections prometheus.Gauge
DBLatency prometheus.Histogram
DLQSize prometheus.Gauge
// IngestDurationSeconds is the per-Export E2E latency observed inside
// the OTLP servers (gRPC + HTTP), labeled by signal {traces,logs,metrics}.
// Drives ingest SLOs: alert on p99 / error budget burn rather than on the
// blunt OtelContext_grpc_request_duration_seconds aggregate.
IngestDurationSeconds *prometheus.HistogramVec
// --- gRPC ---
GRPCRequestsTotal *prometheus.CounterVec
GRPCRequestDuration *prometheus.HistogramVec
GRPCBatchSize prometheus.Histogram
// --- HTTP ---
HTTPRequestsTotal *prometheus.CounterVec
HTTPRequestDuration *prometheus.HistogramVec
// --- TSDB ---
TSDBIngestTotal prometheus.Counter
TSDBFlushDuration prometheus.Histogram
TSDBBatchesDropped prometheus.Counter
TSDBCardinalityOverflow prometheus.Counter
// TSDBCardinalityOverflowByTenant labels overflow events with the tenant ID
// that triggered them, or the sentinel "__global__" when the global cap
// (not a per-tenant cap) was the trigger. Use this to identify noisy
// tenants: sum by (tenant_id) (rate(otelcontext_tsdb_cardinality_overflow_by_tenant_total[5m]))
TSDBCardinalityOverflowByTenant *prometheus.CounterVec
// --- WebSocket ---
WSMessagesSent *prometheus.CounterVec
WSSlowClientsRemoved prometheus.Counter
// --- DLQ ---
DLQEnqueuedTotal prometheus.Counter
DLQReplaySuccess prometheus.Counter
DLQReplayFailure prometheus.Counter
DLQDiskBytes prometheus.Gauge
// --- Storage ---
HotDBSizeBytes prometheus.Gauge
// --- Retention ---
RetentionRowsPurgedTotal *prometheus.CounterVec
RetentionPurgeDurationSeconds *prometheus.HistogramVec
RetentionVacuumDurationSeconds *prometheus.HistogramVec
RetentionRowsBehindGauge *prometheus.GaugeVec
// --- Postgres partitioning (DB_POSTGRES_PARTITIONING=daily) ---
// PartitionsDropped counts daily logs partitions dropped during the
// retention pass. Each drop is a near-instant DDL — alert when this
// counter is flat for >1.5 retention periods (indicates a stuck loop).
PartitionsDropped prometheus.Counter
// PartitionsActive gauges the live partitions attached to logs.
// Healthy steady-state ~ HOT_RETENTION_DAYS + DB_PARTITION_LOOKAHEAD_DAYS + 1.
PartitionsActive prometheus.Gauge
// --- Runtime ---
GoGoroutines prometheus.Gauge
GoHeapAllocBytes prometheus.Gauge
// --- Operational (Fix 6) ---
PanicsRecoveredTotal *prometheus.CounterVec
MCPToolInvocationsTotal *prometheus.CounterVec
APIAuthFailuresTotal *prometheus.CounterVec
GraphRAGEventBufferDepth prometheus.Gauge
RetentionLastSuccessTimestamp *prometheus.GaugeVec
RetentionConsecutiveFailures *prometheus.GaugeVec
DBUp *prometheus.GaugeVec
// --- GraphRAG overflow ---
GraphRAGEventsDroppedTotal *prometheus.CounterVec
// GraphRAGTenantsEvictedTotal counts tenant store slices evicted after
// exceeding GRAPHRAG_TENANT_IDLE_TTL. The default tenant is never
// evicted; a steady non-zero rate on a single-tenant install means
// rogue tenant IDs are reaching ingest.
GraphRAGTenantsEvictedTotal prometheus.Counter
// --- In-memory store census (OOM-survival work) ---
// GraphRAGStoreEntities — live node counts per entity kind across tenants
// (tenants|services|operations|traces|spans|log_clusters|metrics|anomalies).
// GraphRAGStoreEdges — live edge counts per store (service|trace|signal|anomaly).
// Together with the Drain gauge these attribute RSS growth to a
// specific structure before a heap profile is needed.
GraphRAGStoreEntities *prometheus.GaugeVec
GraphRAGStoreEdges *prometheus.GaugeVec
// TSDBRingSeriesActive and TSDBRingSeriesRejected remain registered for
// metric-name compatibility. Production does not instantiate the legacy
// ring, so both collectors remain zero.
TSDBRingSeriesActive prometheus.Gauge
TSDBRingSeriesRejected prometheus.Counter
DrainTemplatesActive prometheus.Gauge
// --- Async ingest pipeline (Phase 1 robustness work) ---
// IngestPipelineQueueDepth — current queue depth, sampled on every Submit.
// Labeled by signal so spikes can be attributed to traces vs logs.
IngestPipelineQueueDepth *prometheus.GaugeVec
// IngestPipelineQueueBytes — approximate bytes held by queued batches.
// Reserved at Submit, released when a worker finishes the batch; the
// byte cap (INGEST_PIPELINE_MAX_BYTES) rejects submissions above it.
IngestPipelineQueueBytes prometheus.Gauge
// IngestPipelineDroppedTotal — batches that did NOT reach the DB.
// reason="soft_backpressure" — healthy batch dropped at >=90% fullness.
// reason="queue_full" — batch rejected at 100% capacity (client got 429/RESOURCE_EXHAUSTED).
// reason="bytes_full" — batch rejected at the byte cap (even priority batches).
IngestPipelineDroppedTotal *prometheus.CounterVec
// IngestPipelineDLQTotal — batches handed to the Dead Letter Queue after
// the persist transaction failed, instead of being dropped silently.
// result="enqueued" — the complete batch is on disk awaiting replay.
// result="enqueue_failed" — the DLQ itself rejected the write (disk full,
// permissions); the batch IS lost.
// result="no_sink" — no DLQ wired into the pipeline; the batch IS lost.
IngestPipelineDLQTotal *prometheus.CounterVec
// ExemplarSubmitTotal — outcome of every raw-exemplar batch submission in
// AGGREGATE_MODE=aggregate, where the durable aggregate commit is the
// Export ACK and raw exemplar storage is bounded best-effort (#196).
// outcome="queued" reason="none" — the batch entered the raw pipeline.
// outcome="dlq" reason="queue_full" — pipeline saturated, DLQ accepted
// the batch. Deferred, NOT lost.
// outcome="lost" reason="dlq_full" — the DLQ refused it for capacity.
// outcome="lost" reason="dlq_error" — no DLQ wired, or its write failed.
// lost{reason="queue_full"} is never emitted: on the lost outcome the
// reason names why the DLQ could not hold the batch, not why the primary
// queue refused it. Intentional soft-backpressure drops are not counted
// here — they are already on IngestPipelineDroppedTotal.
ExemplarSubmitTotal *prometheus.CounterVec
// ExemplarSubmitLostTotal — dedicated counter for the permanent-loss
// subset of ExemplarSubmitTotal, so an alert can target loss without a
// label matcher. reason="dlq_full"|"dlq_error".
ExemplarSubmitLostTotal *prometheus.CounterVec
// --- OTLP metric completeness (#199) ---
// IngestMetricsUnsupportedTotal — metric data points the aggregate path
// refused outright and reported in ExportMetricsPartialSuccess. Labeled by
// the OTLP point type (summary|histogram|exponential_histogram) and the
// reason: cumulative_temporality, unspecified_temporality, unsupported_type
// or malformed_point. Every increment is a point that did NOT enter
// aggregate accounting, and the client must not retry it.
IngestMetricsUnsupportedTotal *prometheus.CounterVec
// IngestMetricsSketchDroppedTotal — histogram points whose SCALARS were
// kept but whose percentiles are unavailable, by reason:
// negative_observations (negative buckets or a bucket that spans negative
// values without a proving min>=0), scale_out_of_range (an
// ExponentialHistogram below scale 0, which the positive-only scale-4
// sketch cannot represent) or no_finite_boundaries.
// This is NOT a rejection: count/sum/min/max still land in the aggregate.
IngestMetricsSketchDroppedTotal *prometheus.CounterVec
// IngestMetricsDimsRejectedTotal — metric points whose configured
// dimension tuple was refused from series identity, by reason. Today the
// only reason is unsupported_value_type: an array or kvlist attribute
// value has no canonical scalar rendering, so it cannot be interned.
// The point is still aggregated, under DimsID=0.
IngestMetricsDimsRejectedTotal *prometheus.CounterVec
// IngestReservedServicePrefixTotal — resources whose client-declared
// service.name sits inside the reserved host/ namespace (#280). The name
// is accepted as sent; the count tells an operator a client is minting
// host entities by hand. Label signal=traces|logs|metrics.
IngestReservedServicePrefixTotal *prometheus.CounterVec
// HTTPOTLPThrottledTotal — count of HTTP 429s issued by the OTLP HTTP
// receiver when the async ingest pipeline is full. Mirrors the gRPC
// RESOURCE_EXHAUSTED path so operators see a single throttling signal
// across both transports. Label `signal` is one of traces|logs|metrics.
HTTPOTLPThrottledTotal *prometheus.CounterVec
// --- DB pool (sampled every 5s from sql.DB.Stats) ---
DBPoolOpenConnections prometheus.Gauge
DBPoolInUse prometheus.Gauge
DBPoolIdle prometheus.Gauge
DBPoolWaitCount prometheus.Gauge
DBPoolWaitDuration prometheus.Gauge // cumulative seconds
// --- DLQ eviction (Task 8) ---
DLQEvictedTotal prometheus.Counter
DLQEvictedBytesTotal prometheus.Counter
// --- Dashboard p99 (Task 10) ---
DashboardP99RowCapHitsTotal prometheus.Counter
// --- Aggregate engine (AGGREGATE_MODE != legacy) ---
// AggregateInputPointsTotal — points offered to the request-local reducer
// per signal, counted BEFORE the sampler and severity gates. This is
// accepted telemetry, not persisted telemetry.
AggregateInputPointsTotal *prometheus.CounterVec
// AggregateDeltasTotal — series deltas emitted by reduction. The gap
// between this and input points is the whole value of the engine.
AggregateDeltasTotal *prometheus.CounterVec
// AggregateReductionRatio — input points per emitted delta, per Export
// request. A ratio collapsing toward 1 means cardinality is exploding.
AggregateReductionRatio *prometheus.HistogramVec
// AggregateLatePointsTotal — points excluded from aggregates because they
// fell outside the mutable-window horizon. reason="late" (older than the
// allowed lateness) or reason="future" (beyond the tolerated skew).
AggregateLatePointsTotal *prometheus.CounterVec
// AggregateSeriesActive — budgeted series present in at least one mutable
// window, per signal. This is what the AGGREGATE_MAX_SERIES* caps bound,
// and it never exceeds them: the __other__ series a cap mints when it
// binds are the reserve, counted by AggregateOverflowSeriesActive instead.
AggregateSeriesActive *prometheus.GaugeVec
// AggregateOverflowSeriesActive — live __other__ series per signal. This
// is the unbudgeted reserve the caps spend; it is bounded by
// (services x signals x status classes), not by AGGREGATE_MAX_SERIES*.
AggregateOverflowSeriesActive *prometheus.GaugeVec
// AggregateOverflowTotal — admissions rerouted to an __other__ series,
// labeled by the cap that triggered it (tenant|service_names|
// service_series|signal|global). Totals are preserved; identity is not.
AggregateOverflowTotal *prometheus.CounterVec
// AggregateShadowAcceptedTotal — telemetry accounted on the aggregate
// side per signal. In shadow mode this is compared against the legacy
// path's accepted counts; it must not move with the sampling rate.
AggregateShadowAcceptedTotal *prometheus.CounterVec
// AggregateShadowErrorsTotal — errors accounted on the aggregate side per
// service. Cheap invariant only (#165): no per-series comparison.
AggregateShadowErrorsTotal *prometheus.CounterVec
// AggregateClosedWindows — windows past their lateness horizon that
// memory still holds because the finalizer has not materialized them into
// aggregate_buckets yet. Steady state is 0 or 1; a value that stays high
// means finalization is behind or failing.
AggregateClosedWindows prometheus.Gauge
// AggregateClosedWindowsEvictedTotal — closed windows the closed-window
// cap forced out of memory before finalization. Each one is lost data:
// alert on any increase.
AggregateClosedWindowsEvictedTotal prometheus.Counter
// --- Durable aggregate store (#173) ---
// AggregateCommitDurationSeconds — group-commit wall time. This IS the
// ACK latency floor: an Export cannot return before its commit does.
AggregateCommitDurationSeconds *prometheus.HistogramVec
// AggregateCommitDeltas — delta rows per group commit. The pre-merge
// ratio and the coalescing behaviour both show up here.
AggregateCommitDeltas prometheus.Histogram
// AggregateCommitsTotal — commits by result (ok|error).
AggregateCommitsTotal *prometheus.CounterVec
// AggregateCommitBytesTotal — delta payload written to the store.
AggregateCommitBytesTotal prometheus.Counter
// AggregateAdmissionRejectedTotal — ErrSaturated refusals by the bound
// that tripped (bytes|waiters|deltas). Non-zero at sustained load is a
// release-gate failure, not a tuning hint.
AggregateAdmissionRejectedTotal *prometheus.CounterVec
// AggregateFinalizeDurationSeconds — window finalization wall time.
AggregateFinalizeDurationSeconds prometheus.Histogram
// AggregateFinalizeRowsTotal — rows materialized/deleted by finalization,
// by kind (buckets|deltas).
AggregateFinalizeRowsTotal *prometheus.CounterVec
// AggregatePurgeDurationSeconds — retention purge wall time on the
// aggregate DB.
AggregatePurgeDurationSeconds prometheus.Histogram
// AggregatePurgeRowsTotal — rows purged by kind (buckets|deltas|baselines).
AggregatePurgeRowsTotal *prometheus.CounterVec
// AggregateDeltaLogRows and AggregateDeltaLogAgeSeconds are the delta-log
// backlog health bounds from #160: alert when either climbs.
AggregateDeltaLogRows prometheus.Gauge
AggregateDeltaLogAgeSeconds prometheus.Gauge
// AggregateRecoveryDurationSeconds and AggregateRecoveryRows describe the
// last startup recovery. The gate allows 30s.
AggregateRecoveryDurationSeconds prometheus.Gauge
AggregateRecoveryRows *prometheus.GaugeVec
// --- Aggregate identity lifecycle (#200) ---
// AggregateGCRunsTotal — identity garbage-collection passes by result
// (ok|error). A pass that fails leaves memory untouched, so a rising
// error count is disk growth, not corruption.
AggregateGCRunsTotal *prometheus.CounterVec
// AggregateGCDurationSeconds — GC wall time by phase. phase="mark" is the
// lock-free scan; phase="barrier" is the part that serializes with the
// group commit and is therefore the only one inside the ACK budget.
AggregateGCDurationSeconds *prometheus.HistogramVec
// AggregateGCSweptTotal — identity rows deleted by GC, by table.
AggregateGCSweptTotal *prometheus.CounterVec
// AggregateGCRetained — identity rows the last pass kept, by table. The
// ratio against the swept counter is what says whether the dictionary has
// reached a steady state.
AggregateGCRetained *prometheus.GaugeVec
// AggregateIdentityOverflowTotal — identities routed to __other__ by an
// identity BOUND rather than by a series cap, labeled by dictionary kind
// and the bound that tripped (length|count).
AggregateIdentityOverflowTotal *prometheus.CounterVec
// AggregateTenantRejectedTotal — points DROPPED because their tenant
// identity was refused. The tenant namespace never collapses into a
// shared __other__, so this is a drop, not a degradation: alert on any
// sustained value.
AggregateTenantRejectedTotal *prometheus.CounterVec
// ResourceRegistryEntries — live resource registry entries per tenant.
// kind=pair counts service-host-workload entries, kind=host counts
// distinct non-empty hosts (#279).
ResourceRegistryEntries *prometheus.GaugeVec
// ResourceRegistryOverflowTotal — registrations DROPPED by a per-tenant
// registry bound (kind=host|pair). Never merged into a shared host.
ResourceRegistryOverflowTotal *prometheus.CounterVec
// --- Bounded exemplar retention (AGGREGATE_MODE=aggregate) ---
// ExemplarEligibleTotal — telemetry that qualified for raw retention,
// per signal and priority class. Eligible is not retained: the gap between
// this and the drop counter is what makes aggregate completeness and raw
// diagnostic coverage distinguishable during a storm (#161).
ExemplarEligibleTotal *prometheus.CounterVec
// ExemplarDroppedTotal — eligible telemetry refused raw persistence,
// reason=budget_count|budget_bytes|stratum.
ExemplarDroppedTotal *prometheus.CounterVec
// ExemplarEvictionTotal — selected exemplars displaced by a better-ranked
// trace. Each eviction is one trace's worth of bounded OVER-retention:
// already-persisted spans are never deleted.
ExemplarEvictionTotal prometheus.Counter
// ExemplarTruncatedTotal — retained traces forced past their max spans or
// max bytes. These persist truncated=true plus retained/observed counts,
// and causal-analysis tools report partial coverage for them (#163).
ExemplarTruncatedTotal prometheus.Counter
// --- 8 GiB data budget and disk watchdog (#201 Q1/Q5) ---
// DiskBudgetBytes — the ENFORCEMENT ceiling actually in effect: the lower
// of DATA_DISK_BUDGET_MB and the usable volume capacity. A volume smaller
// than the configured budget does not grow because the config says so.
DiskBudgetBytes prometheus.Gauge
// DiskUsedBytes / DiskUsedRatio — statfs allocation on the data volume and
// its fraction of DiskBudgetBytes. This pair, not summed file sizes, is
// what the shedding ladder reads.
DiskUsedBytes prometheus.Gauge
DiskUsedRatio prometheus.Gauge
// DiskComponentBytes / DiskComponentHighWaterBytes — per-tier attribution
// against the budget table (main_db, aggregate_db, dlq, wal). The
// high-water gauge is what the seven-day gate (#202) validates the table
// against; the instantaneous gauge alone hides the peak that mattered.
DiskComponentBytes *prometheus.GaugeVec
DiskComponentHighWaterBytes *prometheus.GaugeVec
// DiskSheddingState — 0 none, 1 errors_only, 2 raw_off.
DiskSheddingState prometheus.Gauge
// DiskSheddingTransitionsTotal — every state change, labeled from/to.
// Flapping is visible here and nowhere else.
DiskSheddingTransitionsTotal *prometheus.CounterVec
// ExemplarRowsPurgedTotal / ExemplarPurgeDurationSeconds — throughput of
// the exemplar-tier purge (EXEMPLAR_RETENTION_DAYS), separate from the
// HOT_RETENTION_DAYS counters so the two retentions are distinguishable.
ExemplarRowsPurgedTotal *prometheus.CounterVec
ExemplarPurgeDurationSeconds prometheus.Histogram
// contains filtered or unexported fields
}
Metrics holds all internal Prometheus metrics for OtelContext self-monitoring.
func (*Metrics) DecrementActiveConns ¶
func (m *Metrics) DecrementActiveConns()
func (*Metrics) DisableTSDBCollectors ¶ added in v0.5.0
func (m *Metrics) DisableTSDBCollectors()
DisableTSDBCollectors unregisters the collectors that only the legacy TSDB aggregator and ring buffer can move. Call it exactly once, at startup, when that path is not constructed (AGGREGATE_MODE=aggregate, #194 finding 10).
Leaving them registered would publish TSDB ingest, drop and cardinality series pinned at 0, which a dashboard reads as "no overflow" rather than "no TSDB". The aggregate engine reports its own admission and cardinality caps; these must not shadow them. The struct fields stay non-nil so any residual call site is inert rather than a nil dereference.
func (*Metrics) GetHealthStats ¶
func (m *Metrics) GetHealthStats() HealthStats
func (*Metrics) HealthHandler ¶
func (m *Metrics) HealthHandler() http.HandlerFunc
func (*Metrics) HealthWSHandler ¶
func (m *Metrics) HealthWSHandler() http.HandlerFunc
HealthWSHandler returns an HTTP handler that upgrades to WebSocket and pushes HealthStats snapshots every 3 seconds. An immediate snapshot is sent on connection so the client never has to wait for the first tick.
func (*Metrics) IncrementActiveConns ¶
func (m *Metrics) IncrementActiveConns()
func (*Metrics) ObserveDBLatency ¶
func (*Metrics) ObserveIngestDuration ¶
ObserveIngestDuration records an end-to-end OTLP Export latency for the given signal. Callers should pass time.Since(start) measured from the very start of the Export handler. Nil-safe so the OTLP servers can be wired without a Metrics instance during tests.
func (*Metrics) RecordExemplarDropped ¶ added in v0.5.0
RecordExemplarDropped implements ingest.ExemplarMetrics.
func (*Metrics) RecordExemplarEligible ¶ added in v0.5.0
RecordExemplarEligible implements ingest.ExemplarMetrics.
func (*Metrics) RecordExemplarEviction ¶ added in v0.5.0
func (m *Metrics) RecordExemplarEviction()
RecordExemplarEviction implements ingest.ExemplarMetrics.
func (*Metrics) RecordExemplarTruncation ¶ added in v0.5.0
func (m *Metrics) RecordExemplarTruncation()
RecordExemplarTruncation implements ingest.ExemplarMetrics.
func (*Metrics) RecordIngestion ¶
func (*Metrics) RecordMetricDimsRejected ¶ added in v0.5.0
RecordMetricDimsRejected counts metric points whose dimension tuple was refused from series identity.
func (*Metrics) RecordMetricSketchDropped ¶ added in v0.5.0
RecordMetricSketchDropped counts one histogram point kept for its scalars with percentiles suppressed.
func (*Metrics) RecordMetricUnsupported ¶ added in v0.5.0
RecordMetricUnsupported counts one OTLP metric data point refused by the aggregate path. pointType is the OTLP point type, reason names why. A nil *Metrics is a no-op: every ingest unit test passes one.
func (*Metrics) RecordReservedServicePrefix ¶ added in v0.5.0
RecordReservedServicePrefix counts one resource whose client-declared service.name starts with the reserved host/ prefix. Nil-safe.
func (*Metrics) RecordResourceRegistryOverflow ¶ added in v0.5.0
RecordResourceRegistryOverflow counts one registration refused by the tenant's registry bound of kind (host|pair). Nil-safe.
func (*Metrics) RegisterReadCache ¶ added in v0.5.0
RegisterReadCache exposes a read cache's live entry count as otelcontext_read_cache_entries{cache=name}. Safe on a nil receiver.
func (*Metrics) SampleDBPoolStats ¶
SampleDBPoolStats writes the live pool stats into the DBPool* gauges. Safe to call from a ticker goroutine. A nil receiver or a nil *sql.DB is a no-op so callers don't need to guard at every call site.
WaitCount and WaitDuration from sql.DBStats are cumulative values (always monotonically increasing) — operators should compute rate() over them.
func (*Metrics) SetActiveConnections ¶
func (*Metrics) SetDLQSize ¶
func (*Metrics) SetResourceRegistryEntries ¶ added in v0.5.0
SetResourceRegistryEntries publishes the tenant's live registry count of kind (host|pair). Nil-safe.
func (*Metrics) StartRuntimeMetrics ¶
func (m *Metrics) StartRuntimeMetrics()
StartRuntimeMetrics samples Go runtime stats every 15 seconds.