Documentation
¶
Index ¶
- Constants
- Variables
- func AgentProcessID() string
- func DecodePush(w http.ResponseWriter, req *http.Request) (map[string]*dto.MetricFamily, error)
- func ProtocolFromIPHeader(proto byte) string
- func ReexportsFamily(name string) bool
- func RegisterAgentMetrics()
- type MetricsStore
- func (s *MetricsStore) ForEachResult(fn func(connID string, result *StoreResult) bool)
- func (s *MetricsStore) Push(connID string, families map[string]*dto.MetricFamily)
- func (s *MetricsStore) Register(target StoreTarget)
- func (s *MetricsStore) Results() map[string]*StoreResult
- func (s *MetricsStore) Snapshot() []StoreResult
- func (s *MetricsStore) Start(ctx context.Context) error
- func (s *MetricsStore) Unregister(connID string)
- type ProtocolCounters
- type ReexportCollector
- type ReexportOption
- type StoreResult
- type StoreTarget
Constants ¶
const MaxPushBytes = 1 << 20
MaxPushBytes caps how much of a pushed metrics body is read. An agent exposition body is a few tens of kilobytes, so a body this large is either broken or hostile, and the server must not buffer it either way.
const QueryParamAgentProcessID = "agent_process_id"
QueryParamAgentProcessID is the CONNECT-IP query-string key the agent uses to tell the server its stable per-process ID. Referenced by both the client and the server handler so a rename can't silently break the wire.
const RelayRTTMetric = "tunnel_relay_rtt_seconds"
RelayRTTMetric is the family name of TunnelRelayRTTSeconds. Servers that read the round trip time out of a pushed family look it up by name, so the name is exported here instead of being written out again by each reader.
const ( // StaleResultTimeout is how long a pushed result remains valid. // Results older than this are skipped during collection. StaleResultTimeout = 60 * time.Second )
Variables ¶
var ( // TunnelAgentInfo is an info metric that exports version labels. Always set to 1. TunnelAgentInfo = prometheus.NewGaugeVec( prometheus.GaugeOpts{ Name: "tunnel_agent_info", Help: "Agent build information. Always 1.", }, []string{"version", "build_date", "commit"}, ) // TunnelAgentUptimeSeconds reports agent process uptime. TunnelAgentUptimeSeconds = prometheus.NewGaugeFunc( prometheus.GaugeOpts{ Name: "tunnel_agent_uptime_seconds", Help: "Seconds since the agent process started.", }, func() float64 { return time.Since(startTime).Seconds() }, ) // TunnelRelayRTTSeconds reports the latency to each relay the agent holds a // live session with: the smoothed RTT QUIC continuously measures on the // control connection (falling back to probe/connect time before the first // measurement). The series is deleted when the session ends. TunnelRelayRTTSeconds = prometheus.NewGaugeVec( prometheus.GaugeOpts{ Name: RelayRTTMetric, Help: "Smoothed round-trip time to the connected relay, as measured by QUIC.", }, []string{"relay"}, ) // TunnelRelayPacketsLost counts packets QUIC declared lost on the control // connection, by relay and loss reason ("timeout", "reordering", "other"). // Loss rate is usually a better "tunnel feels slow" signal than RTT. TunnelRelayPacketsLost = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_relay_packets_lost_total", Help: "Packets declared lost by QUIC on the relay control connection.", }, []string{"relay", "reason"}, ) // TunnelRelayPTOs counts QUIC probe timeouts on the control connection — // full stalls where nothing was ACKed for a whole probe interval. A rising // rate means the path to that relay is dying, not just congested. TunnelRelayPTOs = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_relay_ptos_total", Help: "QUIC probe timeouts (PTO) on the relay control connection.", }, []string{"relay"}, ) // TunnelConnectionReconnects counts reconnection attempts across all connections. TunnelConnectionReconnects = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_connection_reconnects_total", Help: "Total number of tunnel reconnection attempts.", }, ) // TunnelServer metrics. TunnelPingRequests = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_ping_requests_total", Help: "Total number of ping requests for latency probing.", }, ) TunnelConnectionRequests = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_connection_requests_total", Help: "Total number of connection requests to the tunnel server.", }, ) TunnelConnectionsActive = prometheus.NewGauge( prometheus.GaugeOpts{ Name: "tunnel_connections_active", Help: "Number of currently active tunnel connections.", }, ) TunnelConnectionFailures = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_connection_failures_total", Help: "Total number of failed connection attempts.", }, []string{"reason"}, ) // TunnelSessionClosures counts connections removed after their QUIC control // session closed. This is the primary per-connection liveness signal. TunnelSessionClosures = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_session_closures_total", Help: "Tunnel connections removed after their QUIC control session closed.", }, ) // TunnelCleanupPending is the number of connection allocations held until // their Tunnel deletion is confirmed. TunnelCleanupPending = prometheus.NewGauge( prometheus.GaugeOpts{ Name: "tunnel_cleanup_pending", Help: "Connection allocations waiting for Tunnel deletion confirmation.", }, ) // TunnelCleanupRetries counts Tunnel deletion attempts after a failure. TunnelCleanupRetries = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_cleanup_retries_total", Help: "Tunnel deletion attempts that failed and were scheduled for retry.", }, ) // TunnelSlotLosses counts leased slots that lost their backing authority. TunnelSlotLosses = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_slot_losses_total", Help: "Leased relay slots that lost their backing lease authority.", }, ) TunnelNodesManaged = prometheus.NewGauge( prometheus.GaugeOpts{ Name: "tunnel_nodes_managed_total", Help: "Number of currently managed tunnel nodes.", }, ) // MuxedConn metrics. TunnelPacketsSent = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_packets_sent_total", Help: "Total number of packets sent through the tunnel.", }, ) TunnelBytesSent = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_bytes_sent_total", Help: "Total number of bytes sent through the tunnel.", }, ) // TunnelPacketsSentErrors tracks packet send errors with labels. // Common error_type values: "invalid_ip", "no_tunnel", "invalid_connection_type", "write_error" TunnelPacketsSentErrors = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_packets_sent_errors_total", Help: "Total number of packets sent through the tunnel with errors.", }, []string{"error_type"}, ) TunnelPacketsReceived = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_packets_received_total", Help: "Total number of packets received from the tunnel.", }, ) TunnelBytesReceived = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_bytes_received_total", Help: "Total number of bytes received from the tunnel.", }, ) // TunnelPacketsReceivedErrors tracks packet receive errors with labels. // Common error_type values: "read_error", "connection_closed" TunnelPacketsReceivedErrors = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_packets_received_errors_total", Help: "Total number of packets received from the tunnel with errors.", }, []string{"error_type"}, ) // TunnelPacketsDropped tracks packets that were dropped. // Common reason values: "channel_full", "channel_closed" TunnelPacketsDropped = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_packets_dropped_total", Help: "Total number of packets dropped by the tunnel.", }, []string{"reason"}, ) // Per-protocol packet and byte counters. // Protocol values: "tcp", "udp", "icmp", "other". // Direction values: "tx", "rx". TunnelPacketsByProtocol = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_packets_by_protocol_total", Help: "Total packets broken down by IP protocol and direction.", }, []string{"protocol", "direction"}, ) TunnelBytesByProtocol = prometheus.NewCounterVec( prometheus.CounterOpts{ Name: "tunnel_bytes_by_protocol_total", Help: "Total bytes broken down by IP protocol and direction.", }, []string{"protocol", "direction"}, ) // TunnelConnectIPICMPReturned counts ICMP packets that CONNECT-IP // synthesized in response to an outbound-write failure (notably the QUIC // DatagramTooLargeError path that emits an ICMPv6 Packet-Too-Big back at // the sender for PMTUD). Cross-reference with tunnel_icmp6_packet_too_big_* // to detect PTB generation vs. propagation issues. // TODO(APO-543): remove once PMTUD-over-tunnel is verified healthy. TunnelConnectIPICMPReturned = prometheus.NewCounter( prometheus.CounterOpts{ Name: "tunnel_connect_ip_icmp_returned_total", Help: "ICMP packets synthesized by connect-ip-go on WritePacket failure (e.g. QUIC DatagramTooLargeError → ICMPv6 PTB).", }, ) )
var ErrPushTooLarge = errors.New("push body is over the size limit")
ErrPushTooLarge reports a pushed metrics body that goes over MaxPushBytes. Callers match it with errors.Is to tell an oversized body from a body that is simply not the exposition format: the first gets 413, the second 400.
Functions ¶
func AgentProcessID ¶
func AgentProcessID() string
AgentProcessID returns the stable per-process ID for this agent.
func DecodePush ¶ added in v0.23.0
func DecodePush(w http.ResponseWriter, req *http.Request) (map[string]*dto.MetricFamily, error)
DecodePush reads a pushed metrics body in Prometheus text exposition format and returns the metric families it holds, keyed by family name. The read is capped at MaxPushBytes, so every push handler gets the same limit. A body over the cap comes back as an error that wraps ErrPushTooLarge, and the caller answers with 413. A body that is not the exposition format comes back as a plain error, and the caller answers with 400.
func ProtocolFromIPHeader ¶
ProtocolFromIPHeader returns a protocol label from the IP next-header/protocol byte.
func ReexportsFamily ¶ added in v0.23.0
ReexportsFamily reports whether a pushed metric family is re-exported. See reexportFamilyPrefix for why the rest of an agent's registry is dropped.
func RegisterAgentMetrics ¶
func RegisterAgentMetrics()
RegisterAgentMetrics registers agent-only metrics (info and uptime) with the controller-runtime metrics registry. This must be called explicitly by agent processes — server processes (tunnelproxy) should NOT call this, as they re-export agent metrics via the AgentScraper/ReexportCollector path instead.
Types ¶
type MetricsStore ¶
type MetricsStore struct {
// contains filtered or unexported fields
}
MetricsStore accepts pushed metrics from tunnel agents and makes them available for re-export via Prometheus. It replaces the old pull-based AgentScraper: instead of scraping agents over the overlay network, agents POST their metrics to the tunnelproxy over the existing HTTP/3 connection.
func NewMetricsStore ¶
func NewMetricsStore() *MetricsStore
NewMetricsStore creates a new MetricsStore.
func (*MetricsStore) ForEachResult ¶
func (s *MetricsStore) ForEachResult(fn func(connID string, result *StoreResult) bool)
ForEachResult iterates over results under a read lock. The callback must not call other MetricsStore methods (deadlock). Keep callbacks short: an agent push waits for the read lock to be released. Use Snapshot for work that is more than a copy.
func (*MetricsStore) Push ¶
func (s *MetricsStore) Push(connID string, families map[string]*dto.MetricFamily)
Push stores a new set of metric families for the given connection, replacing any previously pushed data. Unknown connection IDs are silently ignored (the push may arrive between Unregister and the next push tick).
func (*MetricsStore) Register ¶
func (s *MetricsStore) Register(target StoreTarget)
Register records a new agent connection. Metadata from the target is attached to every metric when re-exported. The result starts with empty Families until the first push arrives.
func (*MetricsStore) Results ¶
func (s *MetricsStore) Results() map[string]*StoreResult
Results returns a snapshot of all current results.
func (*MetricsStore) Snapshot ¶ added in v0.23.0
func (s *MetricsStore) Snapshot() []StoreResult
Snapshot copies every stored result out under a short read lock, so callers can walk the results with no lock held. The copies keep the same Families map the store holds, which stays safe to read after the lock is released: Push replaces the whole map (see Push above) and never writes into a map it already handed out, so the map a caller holds is never changed under it.
func (*MetricsStore) Start ¶
func (s *MetricsStore) Start(ctx context.Context) error
Start blocks until ctx is cancelled. It exists for compatibility with callers that run the store in an errgroup alongside other long-running goroutines.
func (*MetricsStore) Unregister ¶
func (s *MetricsStore) Unregister(connID string)
Unregister removes an agent connection and its cached metrics.
type ProtocolCounters ¶
type ProtocolCounters struct {
Packets prometheus.Counter
Bytes prometheus.Counter
}
ProtocolCounters holds pre-resolved counters for a (protocol, direction) pair to avoid per-packet WithLabelValues lookups on the hot path.
func GetProtocolCounters ¶
func GetProtocolCounters(proto, direction string) *ProtocolCounters
GetProtocolCounters returns pre-resolved counters for the given protocol and direction. Returns nil if the protocol string is empty.
type ReexportCollector ¶
type ReexportCollector struct {
// contains filtered or unexported fields
}
ReexportCollector implements prometheus.Collector by iterating over pushed agent metrics and re-emitting them with tunnel_node, agent, conn_id, agent_process_id, and project_id labels injected. It should be registered with the tunnelproxy's Prometheus registry so agent metrics appear on the tunnelproxy's /metrics endpoint.
func NewReexportCollector ¶
func NewReexportCollector(store *MetricsStore, opts ...ReexportOption) *ReexportCollector
NewReexportCollector creates a new ReexportCollector backed by the given store.
func (*ReexportCollector) Collect ¶
func (c *ReexportCollector) Collect(ch chan<- prometheus.Metric)
Collect implements prometheus.Collector. It takes a snapshot of the store results under a short read lock, then converts the snapshot with no lock held: the conversion builds a descriptor per series and can only go as fast as the scraper reads the channel, while an agent push needs the write lock. A slow scrape must not hold pushes off for its whole duration.
func (*ReexportCollector) Describe ¶
func (c *ReexportCollector) Describe(ch chan<- *prometheus.Desc)
Describe implements prometheus.Collector. Because the set of metrics is dynamic (depends on what the agents export), we emit no fixed descriptors and mark this as an unchecked collector.
type ReexportOption ¶
type ReexportOption func(*ReexportCollector)
ReexportOption configures a ReexportCollector.
func WithReexportPrefix ¶
func WithReexportPrefix(prefix string) ReexportOption
WithReexportPrefix sets a prefix added to all re-exported metric names. Defaults to "apoxy_".
type StoreResult ¶
type StoreResult struct {
Target StoreTarget
// RegisteredAt is the time the connection was registered with the store.
// Used to compute per-connection uptime independent of the agent process
// uptime (which is shared across all conns of the same agent).
RegisteredAt time.Time
Families map[string]*dto.MetricFamily
PushedAt time.Time
}
StoreResult holds the parsed metric families from a single agent push.
type StoreTarget ¶
type StoreTarget struct {
ConnID string
TunnelNode string
// AgentName is the legacy label value used for the "agent" label on
// re-exported metrics. In the tunnelproxy CONNECT-IP path it is set to
// ConnID (so "agent" is really per-conn). Retained for dashboard
// backwards-compatibility; prefer ConnID and AgentProcessID for new queries.
AgentName string
// AgentProcessID is a stable UUID for the agent *process* the connection
// originates from. Multiple conns opened by the same agent process share
// this value; use it to deduplicate process-scoped metrics like
// tunnel_agent_uptime_seconds across min-conns > 1 sessions. May be empty
// if the agent is old enough to not send it.
AgentProcessID string
ProjectID string
}
StoreTarget describes an agent whose metrics are stored.