proxy

package
v2.2.1 Latest Latest
Warning

This package is not in the latest version of its module.

Go to latest
Published: Oct 2, 2026 License: AGPL-3.0 Imports: 68 Imported by: 0

Documentation

Index

Constants

View Source
const (
	AuthTargetLLM            = "llm"
	AuthTargetDatasource     = "datasource"
	AuthTargetTool           = "tool"
	AuthTargetModelRouter    = "model_router"
	AuthTargetSemanticRouter = "semantic_router"
	AuthTargetPlugin         = "plugin" // a custom_endpoint plugin's /plugins/{slug}/ routes
)

Kinds of endpoint an auth plugin can be attached to (AuthTarget.Kind).

View Source
const (
	CredTypeBearer = "bearer"  // Authorization: Bearer ...
	CredTypeAPIKey = "api_key" // a vendor key header, x-api-key, a bare Authorization value or ?apiKey=
)

Credential types passed to CustomAuth.

View Source
const (
	AuthMethodAppKey = authidentity.MethodAppKey
	AuthMethodOAuth  = authidentity.MethodOAuth
	AuthMethodPlugin = authidentity.MethodPlugin
)

Authentication methods (AuthIdentity.Method).

View Source
const (
	LLMPRefix        = "/llm/"
	DatasourcePrefix = "/datasource/"
	ToolPrefix       = "/tools/"
	MCPSuffix        = "/mcp"
)
View Source
const DefaultUnifiedRouterBasePath = "/v1"

DefaultUnifiedRouterBasePath is where the unified ingress lives unless the embedder moves it. It matches the OpenAI/OpenRouter convention so an unmodified SDK pointed at the gateway root works with no path configuration.

Variables

View Source
var (
	// ErrRouteNoMatch: the request names nothing the router routes (400).
	ErrRouteNoMatch = errors.New("no route matches the requested model")
	// ErrRouteNoCandidates: the router would route the request, but to
	// nothing this App may use (403).
	ErrRouteNoCandidates = errors.New("the router has no target this app may use")
	// ErrRouteUnavailable: every target the router could pick is inactive or
	// unloaded (503).
	ErrRouteUnavailable = errors.New("the router has no available target")
)

Functions

func AnalyzeCompletionResponse

func AnalyzeCompletionResponse(service services.ServiceInterface, llm *models.LLM, app *models.App, response models.ITokenResponse, ctx context.Context, r *http.Request, timestamp time.Time)

func AnalyzeResponse

func AnalyzeResponse(service services.ServiceInterface, llm *models.LLM, app *models.App, statusCode int, body []byte, reqBody []byte, r *http.Request)

func AnalyzeStreamingResponse

func AnalyzeStreamingResponse(service services.ServiceInterface, llm *models.LLM, app *models.App, statusCode int, responses []byte, reqBody []byte, r *http.Request, chunks [][]byte, timestamp time.Time, contentEncoding string)

func AnthropicModelExtractor

func AnthropicModelExtractor(r *http.Request, body []byte) (string, error)

func AnthropicValidator

func AnthropicValidator(r *http.Request) (string, error)

func AzureModelExtractor

func AzureModelExtractor(r *http.Request, body []byte) (string, error)

func BedrockModelExtractor

func BedrockModelExtractor(r *http.Request, body []byte) (string, error)

BedrockModelExtractor extracts the model ID from a Bedrock Converse API request. The model ID is sent in the URL path (e.g., /model/{modelId}/converse) or request body.

func BedrockValidator

func BedrockValidator(r *http.Request) (string, error)

BedrockValidator extracts the gateway app API key from the AWS SigV4 Authorization header. Clients set their gateway app API key as the AWS_ACCESS_KEY_ID; the gateway extracts it from the Credential field of the SigV4 header. Format: AWS4-HMAC-SHA256 Credential=ACCESS_KEY_ID/date/region/service/aws4_request, ...

func DummyValidator

func DummyValidator(r *http.Request) (string, error)

func ExecuteFinalResponseFilters

func ExecuteFinalResponseFilters(
	llm *models.LLM,
	service services.ServiceInterface,
	statusCode int,
	chunkIndex int,
	fullText string,
	r *http.Request,
) (blocked bool, blockMessage string, err error)

ExecuteFinalResponseFilters runs the response filters once more when a streaming response has completed, over the whole accumulated text. Per-chunk evaluation can legitimately skip a response that never reaches a script's buffer threshold or a guardrail's cadence; this pass makes sure every streamed response is looked at in full at least once. The chunks are already with the client, so a block here ends the stream with an error event and is recorded, rather than withholding content.

func ExecuteResponseFilters

func ExecuteResponseFilters(
	llm *models.LLM,
	service services.ServiceInterface,
	responseBody []byte,
	statusCode int,
	isStreaming bool,
	isChunk bool,
	chunkIndex int,
	currentBuffer string,
	r *http.Request,
) (blocked bool, blockMessage string, err error)

ExecuteResponseFilters executes response-side filters on LLM responses Returns whether the response should be blocked and an optional block message

func ExtractBedrockModelIDFromPath

func ExtractBedrockModelIDFromPath(path string) string

ExtractBedrockModelIDFromPath extracts the model ID from a Bedrock URL path. Native Bedrock API paths contain /model/{modelId}/converse or /model/{modelId}/converse-stream.

func GoogleAIModelExtractor

func GoogleAIModelExtractor(r *http.Request, body []byte) (string, error)

func GoogleAIValidator

func GoogleAIValidator(r *http.Request) (string, error)

GoogleAIValidator extracts and validates the API key from the incoming request. It checks both the 'x-goog-api-key' header and the 'key' query parameter.

func HuggingFaceModelExtractor

func HuggingFaceModelExtractor(r *http.Request, body []byte) (string, error)

func HuggingFaceValidator

func HuggingFaceValidator(r *http.Request) (string, error)

func IsInternalHop

func IsInternalHop(r *http.Request) bool

IsInternalHop reports whether r is the gateway calling itself on its loopback hop: it carries this process's hop token and arrives over a loopback connection.

func MockValidator

func MockValidator(r *http.Request) (string, error)

func NewUnifiedRouterHandler

func NewUnifiedRouterHandler(next http.Handler, basePath string) http.Handler

NewUnifiedRouterHandler returns the handler for the unified router ingress. next receives the rewritten request and is expected to be the /ai/ handler chain (auth middleware included).

The handler deliberately does NOT resolve the route prefix itself: every well-formed request is rewritten and forwarded so that authentication runs before route resolution. An anonymous caller therefore always gets 401 and cannot enumerate configured route slugs from 404-vs-401 responses; the /ai/ shim produces the vendor-not-found 404 after auth.

func NormalizeUnifiedRouterBasePath

func NormalizeUnifiedRouterBasePath(basePath string) string

NormalizeUnifiedRouterBasePath cleans a configured ingress base path into the form the router matches on: a leading slash, no trailing slash, no empty or dot segments. Empty or unusable input falls back to DefaultUnifiedRouterBasePath: a gateway with the ingress silently mounted at "/" would swallow every other route, and a segment carrying router metacharacters would register a wildcard or path-variable route instead of the literal prefix the operator asked for.

Embedders that need the ingress gone entirely disable it (Config.DisableUnifiedRouter) rather than passing an empty path.

func OpenAIModelExtractor

func OpenAIModelExtractor(r *http.Request, body []byte) (string, error)

func OpenAIValidator

func OpenAIValidator(r *http.Request) (string, error)

func ParseUnifiedModel

func ParseUnifiedModel(model string) (route string, bareModel string, err error)

ParseUnifiedModel splits an OpenRouter-style model string into its vendor route prefix and the bare model name. The split is on the FIRST slash so model names that themselves contain slashes (e.g. fine-tune identifiers) survive intact.

func VertexModelExtractor

func VertexModelExtractor(r *http.Request, body []byte) (string, error)

func VertexValidator

func VertexValidator(r *http.Request) (string, error)

func WithAuthIdentity

func WithAuthIdentity(ctx context.Context, id *AuthIdentity) context.Context

WithAuthIdentity returns ctx carrying id.

Types

type APIError

type APIError struct {
	Code           any         `json:"code,omitempty"`
	Message        string      `json:"message"`
	Param          *string     `json:"param,omitempty"`
	Type           string      `json:"type"`
	HTTPStatus     string      `json:"-"`
	HTTPStatusCode int         `json:"-"`
	InnerError     *InnerError `json:"innererror,omitempty"`
}

type AnthropicCacheControl

type AnthropicCacheControl struct {
	Type string `json:"type"` // "ephemeral"
}

AnthropicCacheControl marks a block as a prompt-cache breakpoint.

type AnthropicContentBlock

type AnthropicContentBlock struct {
	Type string `json:"type"` // "text" | "tool_use" | "tool_result"

	// text
	Text string `json:"text,omitempty"`

	// tool_use (assistant asking to call a tool)
	ID    string          `json:"id,omitempty"`
	Name  string          `json:"name,omitempty"`
	Input json.RawMessage `json:"input,omitempty"`

	// tool_result (user returning a tool's output)
	ToolUseID string          `json:"tool_use_id,omitempty"`
	Content   json.RawMessage `json:"content,omitempty"` // string | []AnthropicContentBlock
	IsError   bool            `json:"is_error,omitempty"`

	// prompt caching marker
	CacheControl *AnthropicCacheControl `json:"cache_control,omitempty"`
}

AnthropicContentBlock is one block within a message's content array (or a system block). Different block types populate different fields.

type AnthropicMessage

type AnthropicMessage struct {
	Role    string          `json:"role"` // "user" | "assistant"
	Content json.RawMessage `json:"content"`
}

AnthropicMessage is a single conversation turn. Content is either a plain string or an array of typed content blocks.

type AnthropicMessageExtractor

type AnthropicMessageExtractor struct{}

AnthropicMessageExtractor extracts messages from Anthropic-format requests

func (*AnthropicMessageExtractor) ExtractMessages

func (e *AnthropicMessageExtractor) ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)

ExtractMessages parses Anthropic's format (separate system field + messages array)

func (*AnthropicMessageExtractor) VendorName

func (e *AnthropicMessageExtractor) VendorName() string

VendorName returns "anthropic"

type AnthropicMessageReconstructor

type AnthropicMessageReconstructor struct{}

AnthropicMessageReconstructor rebuilds Anthropic-format requests

func (*AnthropicMessageReconstructor) Reconstruct

func (r *AnthropicMessageReconstructor) Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)

func (*AnthropicMessageReconstructor) VendorName

func (r *AnthropicMessageReconstructor) VendorName() string

type AnthropicMessagesRequest

type AnthropicMessagesRequest struct {
	Model         string               `json:"model"`
	Messages      []AnthropicMessage   `json:"messages"`
	System        json.RawMessage      `json:"system,omitempty"` // string | []AnthropicContentBlock
	MaxTokens     int                  `json:"max_tokens"`
	Temperature   *float64             `json:"temperature,omitempty"`
	TopP          *float64             `json:"top_p,omitempty"`
	StopSequences []string             `json:"stop_sequences,omitempty"`
	Tools         []AnthropicTool      `json:"tools,omitempty"`
	ToolChoice    *AnthropicToolChoice `json:"tool_choice,omitempty"`
	Stream        bool                 `json:"stream,omitempty"`
}

AnthropicMessagesRequest is the body of POST /anthropic/{routeId}/v1/messages.

type AnthropicMessagesResponse

type AnthropicMessagesResponse struct {
	ID           string                   `json:"id"`
	Type         string                   `json:"type"` // "message"
	Role         string                   `json:"role"` // "assistant"
	Model        string                   `json:"model"`
	Content      []AnthropicResponseBlock `json:"content"`
	StopReason   string                   `json:"stop_reason,omitempty"`
	StopSequence *string                  `json:"stop_sequence"`
	Usage        AnthropicUsage           `json:"usage"`
}

AnthropicMessagesResponse is the non-streaming response, and also the message skeleton embedded in the streaming "message_start" event.

type AnthropicModelListEntry

type AnthropicModelListEntry struct {
	Type        string `json:"type"` // "model"
	ID          string `json:"id"`
	DisplayName string `json:"display_name"`
}

AnthropicModelListEntry is one model in the list.

type AnthropicModelListResponse

type AnthropicModelListResponse struct {
	Data    []AnthropicModelListEntry `json:"data"`
	HasMore bool                      `json:"has_more"`
}

AnthropicModelListResponse is the body of GET /anthropic/{routeId}/v1/models, the list Claude Code reads for gateway model discovery. Data is never null.

type AnthropicResponseBlock

type AnthropicResponseBlock struct {
	Type  string          `json:"type"`
	Text  string          `json:"text,omitempty"`
	ID    string          `json:"id,omitempty"`
	Name  string          `json:"name,omitempty"`
	Input json.RawMessage `json:"input,omitempty"`
}

AnthropicResponseBlock is an output content block (text or tool_use).

type AnthropicTool

type AnthropicTool struct {
	Name         string                 `json:"name"`
	Description  string                 `json:"description,omitempty"`
	InputSchema  map[string]any         `json:"input_schema,omitempty"`
	CacheControl *AnthropicCacheControl `json:"cache_control,omitempty"`
}

AnthropicTool is a tool definition the model may call.

type AnthropicToolChoice

type AnthropicToolChoice struct {
	Type string `json:"type"` // "auto" | "any" | "tool"
	Name string `json:"name,omitempty"`
}

AnthropicToolChoice controls whether/which tool the model must use.

type AnthropicUsage

type AnthropicUsage struct {
	InputTokens              int `json:"input_tokens"`
	OutputTokens             int `json:"output_tokens"`
	CacheCreationInputTokens int `json:"cache_creation_input_tokens,omitempty"`
	CacheReadInputTokens     int `json:"cache_read_input_tokens,omitempty"`
}

AnthropicUsage carries token counts. input_tokens is reported in message_start, output_tokens in message_delta (streaming) or both together (non-streaming).

type AudioConfig

type AudioConfig struct {
	Voice  string `json:"voice"`
	Format string `json:"format"`
}

type AuthHooks

type AuthHooks struct {
	// PreAuth runs BEFORE credential extraction and validation.
	// Only called for non-OAuth requests (LLM proxy requests).
	// Has NO access to authenticated user/app data.
	// Return true to block the request.
	PreAuth func(w http.ResponseWriter, r *http.Request) bool

	// CustomAuth lets auth plugins authenticate a request in place of app
	// keys. It is called with the endpoint the request addresses (an LLM,
	// datasource, tool, router or plugin endpoint) and the credential the
	// request presented, for every request that is not a Studio OAuth access
	// token.
	//
	// An endpoint with auth plugins attached is authenticated by them alone:
	// the hook answers AuthAccepted or AuthRejected, and app keys are not
	// tried. An endpoint without any gets AuthNotHandled, and app keys apply
	// as usual. An error means the plugins could not be asked (none could be
	// loaded, say); the request is refused with a 503.
	CustomAuth func(r *http.Request, target AuthTarget, credential, credType string) (AuthResult, error)

	// PostAuth runs AFTER successful authentication (standard or custom).
	// Only called for non-OAuth requests (LLM proxy requests).
	// Receives the authenticated app ID.
	// Return true to block the request.
	PostAuth func(w http.ResponseWriter, r *http.Request, appID uint) bool
}

AuthHooks provides extension points in the authentication lifecycle. These hooks are for LLM proxy requests, NOT for OAuth/MCP flows. All hooks are optional (nil checks are performed before calling).

type AuthIdentity

type AuthIdentity = authidentity.Identity

AuthIdentity is who authenticated a request and how. The credential validator puts it on the request context for every method; analytics, the post-auth hook and the plugins read it. It lives in pkg/authidentity so analytics can read it without importing the proxy.

func AuthIdentityFromContext

func AuthIdentityFromContext(ctx context.Context) *AuthIdentity

AuthIdentityFromContext returns the identity the credential validator recorded, or nil.

type AuthOutcome

type AuthOutcome int

AuthOutcome is what the auth plugins decided.

const (
	// AuthNotHandled: the endpoint has no auth plugins; app keys apply.
	AuthNotHandled AuthOutcome = iota
	// AuthAccepted: a plugin authenticated the request as AuthResult.AppID.
	AuthAccepted
	// AuthRejected: the endpoint's plugins all refused the credential.
	AuthRejected
)

type AuthResult

type AuthResult struct {
	Outcome AuthOutcome
	AppID   uint
	// Subject is who the call is for, when the credential says (the auth
	// response's user id: a delegated token's sub, say). For audit only.
	Subject string
	// Claims the plugin returned alongside, for audit and post-auth plugins.
	Claims     map[string]string
	PluginID   uint
	PluginName string
	// Reason is why the plugins refused, for the log. It is not sent to the
	// caller.
	Reason string
}

AuthResult is the outcome of CustomAuth.

type AuthTarget

type AuthTarget struct {
	Kind string // AuthTarget*
	ID   uint
	Slug string
}

AuthTarget is the endpoint a request addresses, resolved from its path.

type BadRequestError

type BadRequestError struct {
	// contains filtered or unexported fields
}

func (*BadRequestError) Error

func (e *BadRequestError) Error() string

type CORSResponseHook

type CORSResponseHook struct{}

CORSResponseHook adds CORS headers to responses

func NewCORSResponseHook

func NewCORSResponseHook() *CORSResponseHook

func (*CORSResponseHook) GetName

func (h *CORSResponseHook) GetName() string

func (*CORSResponseHook) OnBeforeWrite

func (*CORSResponseHook) OnBeforeWriteHeaders

func (h *CORSResponseHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)

func (*CORSResponseHook) OnStreamComplete

OnStreamComplete is called after a streaming response finishes

type ChatCompletionChoice

type ChatCompletionChoice struct {
	Index        int                   `json:"index"`
	Message      ChatCompletionMessage `json:"message"`
	FinishReason string                `json:"finish_reason"`
}

type ChatCompletionChunk

type ChatCompletionChunk struct {
	ID      string                      `json:"id"`
	Object  string                      `json:"object"` // "chat.completion.chunk"
	Created int64                       `json:"created"`
	Model   string                      `json:"model"`
	Choices []ChatCompletionChunkChoice `json:"choices"`
	Usage   *CompletionUsage            `json:"usage,omitempty"` // Only in final chunk
}

ChatCompletionChunk represents a streaming chunk in OpenAI format

type ChatCompletionChunkChoice

type ChatCompletionChunkChoice struct {
	Index        int                 `json:"index"`
	Delta        ChatCompletionDelta `json:"delta"`
	FinishReason *string             `json:"finish_reason"` // null until final chunk
}

ChatCompletionChunkChoice represents a choice in a streaming chunk

type ChatCompletionDelta

type ChatCompletionDelta struct {
	Role      string                        `json:"role,omitempty"`    // Only in first chunk
	Content   string                        `json:"content,omitempty"` // Streaming text content
	ToolCalls []ChatCompletionToolCallDelta `json:"tool_calls,omitempty"`
}

ChatCompletionDelta represents the delta content in a streaming chunk

type ChatCompletionErrorDetail

type ChatCompletionErrorDetail struct {
	Message string `json:"message"`
	Type    string `json:"type"`
}

ChatCompletionErrorDetail contains error details

type ChatCompletionFunctionDelta

type ChatCompletionFunctionDelta struct {
	Name      string `json:"name,omitempty"`
	Arguments string `json:"arguments,omitempty"`
}

ChatCompletionFunctionDelta is the function half of a streamed tool call.

type ChatCompletionMessage

type ChatCompletionMessage struct {
	Role      string                   `json:"role"`
	Content   string                   `json:"content"`
	Name      string                   `json:"name,omitempty"`
	ToolCalls []map[string]interface{} `json:"tool_calls,omitempty"`
}

type ChatCompletionRequest

type ChatCompletionRequest struct {
	Messages            []Message       `json:"messages"`
	Model               string          `json:"model"`
	Store               *bool           `json:"store,omitempty"`
	Metadata            map[string]any  `json:"metadata,omitempty"`
	FrequencyPenalty    *float64        `json:"frequency_penalty,omitempty"`
	LogitBias           map[string]int  `json:"logit_bias,omitempty"`
	LogProbs            *bool           `json:"logprobs,omitempty"`
	TopLogProbs         *int            `json:"top_logprobs,omitempty"`
	MaxCompletionTokens *int            `json:"max_completion_tokens,omitempty"`
	MaxTokens           *int            `json:"max_tokens,omitempty"`
	ReasoningEffort     *string         `json:"reasoning_effort,omitempty"`
	N                   *int            `json:"n,omitempty"`
	Modalities          []string        `json:"modalities,omitempty"`
	Audio               *AudioConfig    `json:"audio,omitempty"`
	PresencePenalty     *float64        `json:"presence_penalty,omitempty"`
	ResponseFormat      *ResponseFormat `json:"response_format,omitempty"`
	Seed                *int            `json:"seed,omitempty"`
	ServiceTier         *string         `json:"service_tier,omitempty"`
	Stop                any             `json:"stop,omitempty"`
	Stream              *bool           `json:"stream,omitempty"`
	StreamOptions       *StreamOptions  `json:"stream_options,omitempty"`
	Temperature         *float64        `json:"temperature,omitempty"`
	TopP                *float64        `json:"top_p,omitempty"`
	Tools               []Tool          `json:"tools,omitempty"`
	ToolChoice          any             `json:"tool_choice,omitempty"`
	ParallelToolCalls   *bool           `json:"parallel_tool_calls,omitempty"`
	User                *string         `json:"user,omitempty"`
}

Models

func (*ChatCompletionRequest) CompletionCount

func (r *ChatCompletionRequest) CompletionCount() int

func (*ChatCompletionRequest) GetMessageContent

func (r *ChatCompletionRequest) GetMessageContent(msg Message) (string, []map[string]any, bool)

func (*ChatCompletionRequest) GetMessages

func (r *ChatCompletionRequest) GetMessages() []llms.MessageContent

func (*ChatCompletionRequest) GetStop

func (r *ChatCompletionRequest) GetStop() ([]string, bool)

Getter functions for ambiguous fields

func (*ChatCompletionRequest) GetToolChoice

func (r *ChatCompletionRequest) GetToolChoice() (string, *FunctionConfig, bool)

func (*ChatCompletionRequest) OutputTokenLimit

func (r *ChatCompletionRequest) OutputTokenLimit() *int

CompletionCount is the number of completions the caller asked for: the request's `n`, defaulting to OpenAI's own default of 1. OutputTokenLimit is the cap on generated tokens the client asked for: max_completion_tokens when set, otherwise max_tokens (the older name, still what most clients send); nil when neither is.

func (*ChatCompletionRequest) ToLangchainOptions

func (r *ChatCompletionRequest) ToLangchainOptions(conf *models.LLM) []llms.CallOption

type ChatCompletionResponse

type ChatCompletionResponse struct {
	ID      string                 `json:"id"`
	Object  string                 `json:"object"`
	Created int64                  `json:"created"`
	Model   string                 `json:"model"`
	Choices []ChatCompletionChoice `json:"choices"`
	Usage   CompletionUsage        `json:"usage"`
}

func NewChatCompletionResponse

func NewChatCompletionResponse(llmResponse *llms.ContentResponse, model string, n int) *ChatCompletionResponse

NewChatCompletionResponse converts a langchaingo ContentResponse into the OpenAI chat.completion object.

n is the number of completions the caller asked for (the request's `n`, or 1 when it is omitted). It matters because langchaingo drivers do not agree on what a ContentChoice represents: OpenAI's returns one per requested completion, while Anthropic's returns one per *content block* of a single turn - so a reply carrying text and a tool call arrives as two choices.

OpenAI indexes `choices` by n and puts every part of one turn in the same message. Passing the block-per-choice shape straight through produced choices[0] with finish_reason:"tool_calls" and no tool_calls array, plus a choices[1] holding the call: an SDK reads choices[0], finds nothing to invoke, and breaks. Anything beyond the requested n is therefore folded back into a single turn.

type ChatCompletionStreamError

type ChatCompletionStreamError struct {
	Error ChatCompletionErrorDetail `json:"error"`
}

ChatCompletionStreamError represents an error in SSE format

type ChatCompletionToolCallDelta

type ChatCompletionToolCallDelta struct {
	Index    int                          `json:"index"`
	ID       string                       `json:"id,omitempty"`
	Type     string                       `json:"type,omitempty"`
	Function *ChatCompletionFunctionDelta `json:"function,omitempty"`
}

ChatCompletionToolCallDelta is one fragment of a streamed tool call.

Index is what makes reassembly possible and is required on every fragment: it is the only thing tying an arguments fragment back to the call it belongs to when a model emits several in parallel. Id and Function.Name appear on the opening fragment; subsequent fragments carry argument text only, and the concatenation of those fragments must parse as JSON.

type CompletionChoice

type CompletionChoice struct {
	Text         string    `json:"text"`
	Index        int       `json:"index"`
	LogProbs     *LogProbs `json:"logprobs"`
	FinishReason string    `json:"finish_reason"`
}

type CompletionResponse

type CompletionResponse struct {
	ID      string             `json:"id"`
	Object  string             `json:"object"`
	Created int64              `json:"created"`
	Model   string             `json:"model"`
	Choices []CompletionChoice `json:"choices"`
	Usage   CompletionUsage    `json:"usage"`
}

type CompletionUsage

type CompletionUsage struct {
	PromptTokens     int `json:"prompt_tokens"`
	CompletionTokens int `json:"completion_tokens"`
	TotalTokens      int `json:"total_tokens"`
}

type Config

type Config struct {
	Port int

	// TLSEnabled reports that the listener this proxy is served from terminates
	// TLS. The /ai/ and unified-router handlers re-enter the gateway with a
	// loopback HTTP call to /llm/call/{slug} on Port (see getInternalLLMBaseURL);
	// that hop must speak HTTPS when the listener does, otherwise the server
	// rejects it with "client sent an HTTP request to an HTTPS server" and every
	// OpenAI-compatible request fails. The loopback connects to 127.0.0.1 and the
	// serving certificate is issued for the public hostname, so it does not
	// verify the certificate: it is talking to its own process.
	TLSEnabled bool

	// LLMTimeout is the timeout for upstream LLM HTTP requests (both streaming and REST).
	// Defaults to 5 minutes if zero, suitable for long-running agentic workloads.
	LLMTimeout time.Duration

	// Server timeouts for the standalone proxy HTTP server (used by proxy.Start()).
	// These are not used when the proxy is mounted into another server.
	// Zero values use defaults: ReadTimeout=5m, WriteTimeout=10m, IdleTimeout=5m.
	ServerReadTimeout  time.Duration
	ServerWriteTimeout time.Duration
	ServerIdleTimeout  time.Duration

	// DebugHTTPProxy logs every request the standalone proxy server
	// receives. DEBUG_HTTP_PROXY=true also turns it on.
	DebugHTTPProxy bool

	// Datasource endpoint limits (zero values use defaults)
	DatasourceMaxBodyBytes  int64 // Max request body size in bytes (default: 1MB)
	DatasourceMaxResults    int   // Max documents returned per query (default: 100)
	DatasourceMaxEmbedTexts int   // Max texts per embedding request (default: 100)

	// UnifiedRouterBasePath is where the OpenRouter-style single-endpoint ingress
	// is mounted: "{base}/chat/completions", "{base}/completions" and "{base}/models".
	// Empty uses DefaultUnifiedRouterBasePath ("/v1").
	//
	// This exists because the proxy is embedded in host gateways (e.g. Tyk's API
	// Gateway) that already own "/v1" or mount the proxy under their own prefix.
	// The value is normalized by NormalizeUnifiedRouterBasePath; the base path of
	// the internal per-route shims (/ai/{slug}/v1/...) is unaffected by it.
	UnifiedRouterBasePath string

	// DisableUnifiedRouter removes the unified ingress entirely: no base-path
	// interception and no {base}/models route, so those paths fall through to the
	// normal authenticated router (404). Per-route endpoints (/ai/, /llm/,
	// /anthropic/) are unaffected. Hosts that expose their own single endpoint,
	// or that must not have the proxy claim a shared path prefix, set this.
	DisableUnifiedRouter bool

	// ServerTiming adds a Server-Timing header and trailer to LLM responses that
	// split each request into gateway time and upstream time (see
	// server_timing.go). Off by default; meant for benchmarking and diagnosis.
	ServerTiming bool
}

type ContentFilterHook

type ContentFilterHook struct {
	// contains filtered or unexported fields
}

ContentFilterHook demonstrates content filtering

func NewContentFilterHook

func NewContentFilterHook(blockedWords []string) *ContentFilterHook

func (*ContentFilterHook) GetName

func (h *ContentFilterHook) GetName() string

func (*ContentFilterHook) OnBeforeWrite

func (*ContentFilterHook) OnBeforeWriteHeaders

func (h *ContentFilterHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)

func (*ContentFilterHook) OnStreamComplete

OnStreamComplete is called after a streaming response finishes

type ContentFilterResults

type ContentFilterResults struct {
	Hate     Hate     `json:"hate,omitempty"`
	SelfHarm SelfHarm `json:"self_harm,omitempty"`
	Sexual   Sexual   `json:"sexual,omitempty"`
	Violence Violence `json:"violence,omitempty"`
}

type CreateCompletionRequest

type CreateCompletionRequest struct {
	Model            string      `json:"model"`
	Prompt           string      `json:"prompt,omitempty"`
	MaxTokens        *int        `json:"max_tokens,omitempty"`
	Temperature      *float64    `json:"temperature,omitempty"`
	TopP             *float64    `json:"top_p,omitempty"`
	N                *int        `json:"n,omitempty"`
	Stream           *bool       `json:"stream,omitempty"`
	LogProbs         *int        `json:"logprobs,omitempty"`
	Echo             *bool       `json:"echo,omitempty"`
	Stop             interface{} `json:"stop,omitempty"`
	PresencePenalty  *float64    `json:"presence_penalty,omitempty"`
	FrequencyPenalty *float64    `json:"frequency_penalty,omitempty"`
	User             string      `json:"user,omitempty"`
}

type CredentialExtractor

type CredentialExtractor func(r *http.Request) (string, error)

type CredentialValidator

type CredentialValidator struct {
	// contains filtered or unexported fields
}

func NewCredentialValidator

func NewCredentialValidator(service services.ServiceInterface, proxy *Proxy) *CredentialValidator

func (*CredentialValidator) CheckAPICredential

func (cv *CredentialValidator) CheckAPICredential(apiKey, dsSlug, llmSlug, routeID, toolSlug string, r *http.Request) (bool, *http.Request)

Renamed from CheckCredential to CheckAPICredential to differentiate

func (*CredentialValidator) Middleware

func (cv *CredentialValidator) Middleware(next http.Handler) http.Handler

func (*CredentialValidator) RegisterValidator

func (cv *CredentialValidator) RegisterValidator(vendor string, validator CredentialExtractor)

func (*CredentialValidator) SetAuthHooks

func (cv *CredentialValidator) SetAuthHooks(hooks *AuthHooks)

SetAuthHooks sets authentication lifecycle hooks

func (*CredentialValidator) SetPostAuthCallback

func (cv *CredentialValidator) SetPostAuthCallback(callback PostAuthCallback)

SetPostAuthCallback is deprecated, use SetAuthHooks instead Kept for backward compatibility

type DatasourceDocument

type DatasourceDocument struct {
	PageContent string         `json:"PageContent"`
	Metadata    map[string]any `json:"Metadata"`
	Score       float32        `json:"Score"`
}

DatasourceDocument is the REST response format for datasource search results. Field names match the original schema.Document marshaling for backward compatibility.

type DefaultResponseHookManager

type DefaultResponseHookManager struct {
	// contains filtered or unexported fields
}

DefaultResponseHookManager provides a basic implementation of ResponseHookManager

func NewDefaultResponseHookManager

func NewDefaultResponseHookManager() *DefaultResponseHookManager

NewDefaultResponseHookManager creates a new default response hook manager

func (*DefaultResponseHookManager) AddHook

func (m *DefaultResponseHookManager) AddHook(hook ResponseHook)

AddHook adds a response hook to the manager

func (*DefaultResponseHookManager) ExecuteOnBeforeWrite

ExecuteOnBeforeWrite executes all registered body hooks

func (*DefaultResponseHookManager) ExecuteOnBeforeWriteHeaders

func (m *DefaultResponseHookManager) ExecuteOnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)

ExecuteOnBeforeWriteHeaders executes all registered header hooks

func (*DefaultResponseHookManager) ExecuteOnStreamComplete

ExecuteOnStreamComplete executes all registered stream complete hooks

func (*DefaultResponseHookManager) HasHooks

func (m *DefaultResponseHookManager) HasHooks() bool

HasHooks returns true if there are any hooks configured

type EmbeddingRequest

type EmbeddingRequest struct {
	Texts []string `json:"texts"`
}

EmbeddingRequest is the request body for embedding generation.

type EmbeddingResponse

type EmbeddingResponse struct {
	Vectors [][]float32 `json:"vectors"`
}

EmbeddingResponse is the response containing generated embedding vectors.

type EndpointMap

type EndpointMap struct {
	LLMs        map[string]string
	Datasources map[string]string
}

type ErrorResponse

type ErrorResponse struct {
	Status  int    `json:"status"`
	Message string `json:"message"`
	Error   string `json:"error,omitempty"`
}

type ExampleResponseHook

type ExampleResponseHook struct {
	// contains filtered or unexported fields
}

ExampleResponseHook demonstrates how to create a custom response hook

func NewExampleResponseHook

func NewExampleResponseHook(name string) *ExampleResponseHook

NewExampleResponseHook creates a new example response hook

func (*ExampleResponseHook) GetName

func (h *ExampleResponseHook) GetName() string

GetName returns the hook name

func (*ExampleResponseHook) OnBeforeWrite

OnBeforeWrite modifies response body content

func (*ExampleResponseHook) OnBeforeWriteHeaders

func (h *ExampleResponseHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)

OnBeforeWriteHeaders adds custom headers to responses

func (*ExampleResponseHook) OnStreamComplete

OnStreamComplete is called after a streaming response finishes

type FunctionConfig

type FunctionConfig struct {
	Description string         `json:"description,omitempty"`
	Name        string         `json:"name"`
	Parameters  map[string]any `json:"parameters,omitempty"`
	Strict      *bool          `json:"strict,omitempty"`
}

type GoogleAIMessageExtractor

type GoogleAIMessageExtractor struct{}

GoogleAIMessageExtractor extracts messages from Google AI/Vertex format requests

func (*GoogleAIMessageExtractor) ExtractMessages

func (e *GoogleAIMessageExtractor) ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)

ExtractMessages parses Google AI's contents/parts structure

func (*GoogleAIMessageExtractor) VendorName

func (e *GoogleAIMessageExtractor) VendorName() string

VendorName returns "google_ai"

type GoogleAIMessageReconstructor

type GoogleAIMessageReconstructor struct{}

GoogleAIMessageReconstructor rebuilds Google AI/Vertex-format requests

func (*GoogleAIMessageReconstructor) Reconstruct

func (r *GoogleAIMessageReconstructor) Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)

func (*GoogleAIMessageReconstructor) VendorName

func (r *GoogleAIMessageReconstructor) VendorName() string

type Hate

type Hate struct {
	Filtered bool   `json:"filtered"`
	Severity string `json:"severity,omitempty"`
}

type HeadersRequest

type HeadersRequest struct {
	Headers map[string]string `json:"headers"`
	Context *PluginContext    `json:"context"`
}

HeadersRequest represents a request to modify response headers

type HeadersResponse

type HeadersResponse struct {
	Modified bool              `json:"modified"`
	Headers  map[string]string `json:"headers"`
}

HeadersResponse represents the response from header modification hooks

type InnerError

type InnerError struct {
	Code                 string               `json:"code,omitempty"`
	ContentFilterResults ContentFilterResults `json:"content_filter_result,omitempty"`
}

type InternalRoutingTransport

type InternalRoutingTransport struct {
	// contains filtered or unexported fields
}

InternalRoutingTransport intercepts SDK HTTP calls for internal routing. When the /ai/ (OpenAI compatibility) endpoint routes requests through /llm/, this transport: 1. Strips vendor-specific auth headers set by the SDK 2. Passes through the original client's Authorization header

This allows /llm/ to authenticate the request using the client's credentials and then set the correct vendor auth (from stored LLM config) before forwarding.

func NewInternalRoutingTransport

func NewInternalRoutingTransport(originalAuth string, serverTLS bool) *InternalRoutingTransport

NewInternalRoutingTransport creates a transport that passes through the original client auth header while stripping any vendor-specific auth headers set by the SDK.

serverTLS must be true when the listener being called back terminates TLS: the loopback then dials 127.0.0.1 over HTTPS without verifying the certificate, which is issued for the public hostname rather than the loopback address.

This builds a private connection pool; the proxy's request path uses newInternalRoutingTransport with a shared pool instead.

func (*InternalRoutingTransport) RoundTrip

func (t *InternalRoutingTransport) RoundTrip(req *http.Request) (*http.Response, error)

type LogProbs

type LogProbs struct {
	Tokens        []string             `json:"tokens"`
	TokenLogProbs []float32            `json:"token_logprobs"`
	TopLogProbs   []map[string]float32 `json:"top_logprobs"`
	TextOffset    []int                `json:"text_offset"`
}

type MCPServerCache

type MCPServerCache struct {
	SSEServer        *server.SSEServer
	StreamableServer *server.StreamableHTTPServer
	MCPServer        *server.MCPServer
	ToolVersion      string
	OperationHash    string
}

type Message

type Message struct {
	Role    string `json:"role"`
	Content any    `json:"content"`
}

type MessageExtractor

type MessageExtractor interface {
	// ExtractMessages parses the request body and returns normalized messages
	ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)

	// VendorName returns the vendor this extractor handles
	VendorName() string
}

MessageExtractor converts vendor-specific request formats to langchaingo MessageContent

type MessageExtractorRegistry

type MessageExtractorRegistry struct {
	// contains filtered or unexported fields
}

MessageExtractorRegistry manages extractors for different vendors

func NewMessageExtractorRegistry

func NewMessageExtractorRegistry() *MessageExtractorRegistry

NewMessageExtractorRegistry creates a new registry for message extractors

func (*MessageExtractorRegistry) Extract

func (r *MessageExtractorRegistry) Extract(vendor string, req *http.Request, body []byte) ([]llms.MessageContent, error)

Extract uses the appropriate extractor to parse messages from the request

func (*MessageExtractorRegistry) Register

func (r *MessageExtractorRegistry) Register(extractor MessageExtractor)

Register adds a message extractor to the registry

type MessageReconstructor

type MessageReconstructor interface {
	// Reconstruct takes normalized messages and the original request body,
	// returning a new request body with modified messages
	Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)

	// VendorName returns the vendor this reconstructor handles
	VendorName() string
}

MessageReconstructor rebuilds vendor-specific request bodies from normalized messages

type MessageReconstructorRegistry

type MessageReconstructorRegistry struct {
	// contains filtered or unexported fields
}

MessageReconstructorRegistry manages reconstructors for different vendors

func NewMessageReconstructorRegistry

func NewMessageReconstructorRegistry() *MessageReconstructorRegistry

NewMessageReconstructorRegistry creates a new registry

func (*MessageReconstructorRegistry) Reconstruct

func (r *MessageReconstructorRegistry) Reconstruct(vendor string, messages []map[string]interface{}, originalBody []byte) ([]byte, error)

Reconstruct uses the appropriate reconstructor for the vendor

func (*MessageReconstructorRegistry) Register

func (r *MessageReconstructorRegistry) Register(reconstructor MessageReconstructor)

Register adds a reconstructor to the registry

type MetadataQuery

type MetadataQuery struct {
	Filter     map[string]string `json:"filter"`
	FilterMode string            `json:"filter_mode"`
	Limit      int               `json:"limit"`
	Offset     int               `json:"offset"`
}

MetadataQuery is the request body for metadata-only document queries.

type MetadataResults

type MetadataResults struct {
	Documents  []DatasourceDocument `json:"documents"`
	TotalCount int                  `json:"total_count"`
}

MetadataResults is the response for metadata queries with pagination.

type ModelNameExtractor

type ModelNameExtractor func(r *http.Request, body []byte) (string, error)

type ModelValidator

type ModelValidator struct {
	// contains filtered or unexported fields
}

func NewModelValidator

func NewModelValidator(allowedModels []string) *ModelValidator

func (*ModelValidator) IsModelAllowed

func (mv *ModelValidator) IsModelAllowed(modelName string) bool

IsModelAllowed reports whether a model name is permitted by the configured patterns.

Two behaviours worth stating plainly, because both surprise people:

  1. An empty pattern list allows EVERY model. An empty Allowed Models field is not a deny-all.
  2. Patterns are matched UNANCHORED, i.e. anywhere within the model name. So "gpt-4.*" -- the example the UI itself suggests -- also matches "legacy-gpt-4o" and "not-really-gpt-4". Anchor explicitly with ^ and $ if you mean the whole name.

The unanchored behaviour is kept deliberately: existing configurations rely on substring matching, and silently anchoring them would start rejecting models that are allowed today.

The rule itself lives in pkg/modelmatch so save-time validation in the service layer and request-time validation here cannot drift apart.

func (*ModelValidator) RegisterExtractor

func (mv *ModelValidator) RegisterExtractor(vendor string, extractor ModelNameExtractor)

func (*ModelValidator) ValidateRequest

func (mv *ModelValidator) ValidateRequest(body []byte) error

type OAIErrorResponse

type OAIErrorResponse struct {
	Error *APIError `json:"error,omitempty"`
}

type OllamaMessageExtractor

type OllamaMessageExtractor struct {
	OpenAIMessageExtractor
}

OllamaMessageExtractor uses OpenAI format

func (*OllamaMessageExtractor) VendorName

func (e *OllamaMessageExtractor) VendorName() string

VendorName returns "ollama"

type OllamaMessageReconstructor

type OllamaMessageReconstructor struct {
	OpenAIMessageReconstructor
}

OllamaMessageReconstructor uses OpenAI format

func (*OllamaMessageReconstructor) VendorName

func (r *OllamaMessageReconstructor) VendorName() string

type OpenAIMessageExtractor

type OpenAIMessageExtractor struct{}

OpenAIMessageExtractor extracts messages from OpenAI-format requests

func (*OpenAIMessageExtractor) ExtractMessages

func (e *OpenAIMessageExtractor) ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)

ExtractMessages wraps the existing ChatCompletionRequest.GetMessages() method

func (*OpenAIMessageExtractor) VendorName

func (e *OpenAIMessageExtractor) VendorName() string

VendorName returns "openai"

type OpenAIMessageReconstructor

type OpenAIMessageReconstructor struct{}

OpenAIMessageReconstructor rebuilds OpenAI-format requests

func (*OpenAIMessageReconstructor) Reconstruct

func (r *OpenAIMessageReconstructor) Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)

func (*OpenAIMessageReconstructor) VendorName

func (r *OpenAIMessageReconstructor) VendorName() string

type OpenAPICache

type OpenAPICache struct {
	Document    libopenapi.Document
	Operations  []string
	ToolVersion string
	CreatedAt   time.Time
}

type PluginContext

type PluginContext struct {
	RequestID string            `json:"request_id"`
	LLMSlug   string            `json:"llm_slug"`
	LLMID     uint              `json:"llm_id"`
	AppID     uint              `json:"app_id"`
	UserID    uint              `json:"user_id"`
	Metadata  map[string]string `json:"metadata"`
}

PluginContext provides context information for hooks

type PostAuthCallback

type PostAuthCallback func(w http.ResponseWriter, r *http.Request, appID uint) bool

type Proxy

type Proxy struct {
	// contains filtered or unexported fields
}

func New

func New(gatewayService services.ServiceInterface, budgetService services.BudgetServiceInterface, cfg *Config) *Proxy

New creates a new Proxy instance using the unified services interface. This is the new interface-based constructor that supports flexible backends.

func NewProxy

func NewProxy(service *services.Service, cfg *Config, budgetService services.BudgetServiceInterface) *Proxy

NewProxy creates a new Proxy instance using the existing concrete services. This is the legacy constructor that maintains backward compatibility.

func (*Proxy) AddFilter

func (p *Proxy) AddFilter(filter *models.Filter)

func (*Proxy) AddResponseHook

func (p *Proxy) AddResponseHook(hook ResponseHook)

AddResponseHook adds a response hook to the proxy (for embeddable AI Gateway)

func (*Proxy) CreateChatCompletionHandler

func (p *Proxy) CreateChatCompletionHandler(w http.ResponseWriter, r *http.Request)

func (*Proxy) CreateCompletionHandler

func (p *Proxy) CreateCompletionHandler(w http.ResponseWriter, r *http.Request)

Handlers

func (*Proxy) GetDatasource

func (p *Proxy) GetDatasource(name string) (*models.Datasource, bool)

func (*Proxy) GetLLM

func (p *Proxy) GetLLM(name string) (*models.LLM, bool)

func (*Proxy) GetLLMByID

func (p *Proxy) GetLLMByID(id uint) (*models.LLM, bool)

GetLLMByID looks a loaded LLM up by its database id. Waterfall rungs are stored by id, so this is the lookup the loop needs; the slug map is what everything else uses.

func (*Proxy) GetResponseHookManager

func (p *Proxy) GetResponseHookManager() ResponseHookManager

GetResponseHookManager returns the response hook manager for external configuration

func (*Proxy) Handler

func (p *Proxy) Handler() http.Handler

Handler returns the HTTP handler for the proxy, allowing it to be used with existing HTTP servers instead of starting its own server

func (*Proxy) Reload

func (p *Proxy) Reload() error

func (*Proxy) SetAuthHooks

func (p *Proxy) SetAuthHooks(hooks *AuthHooks)

SetAuthHooks sets authentication lifecycle hooks

func (*Proxy) SetPostAuthCallback

func (p *Proxy) SetPostAuthCallback(callback PostAuthCallback)

SetPostAuthCallback is deprecated, use SetAuthHooks instead Kept for backward compatibility

func (*Proxy) SetRouteResolver

func (p *Proxy) SetRouteResolver(r RouteResolver)

SetRouteResolver installs the router implementation. Called once by the host before the proxy serves traffic; nil disables routers.

func (*Proxy) Start

func (p *Proxy) Start() error

func (*Proxy) Stop

func (p *Proxy) Stop(ctx context.Context) error

Stop shuts down the server Start is running. Called before Start has begun listening, it makes Start return http.ErrServerClosed instead.

func (*Proxy) UnifiedRouterBasePath

func (p *Proxy) UnifiedRouterBasePath() string

UnifiedRouterBasePath reports where the OpenRouter-style unified ingress is mounted on this proxy's handler ("/v1" by default), or "" when it is disabled. Hosts that mount Handler() behind their own router use this to forward exactly the prefix the proxy claims, instead of hardcoding "/v1".

func (*Proxy) WaitForAnalytics

func (p *Proxy) WaitForAnalytics(ctx context.Context) error

WaitForAnalytics waits until the analysis of every response written so far has finished, or ctx is done. It runs after the response is sent, so a host shutting down calls it after draining its HTTP server and before stopping whatever the analysis records into (the analytics handler, data collection plugins such as the edge's analytics pulse).

type ResponseFormat

type ResponseFormat struct {
	Type       string `json:"type"`
	JSONSchema *any   `json:"json_schema,omitempty"`
}

type ResponseHook

type ResponseHook interface {
	// OnBeforeWriteHeaders is called before response headers are written
	OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)

	// OnBeforeWrite is called before response body is written (REST-only)
	OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)

	// OnStreamComplete is called after a streaming response finishes (streaming-only)
	OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)

	// GetName returns the name of this response hook
	GetName() string
}

ResponseHook defines the interface that response hook implementations must satisfy

type ResponseHookManager

type ResponseHookManager interface {
	// ExecuteOnBeforeWriteHeaders is called before response headers are written
	ExecuteOnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)

	// ExecuteOnBeforeWrite is called before response body is written (REST-only)
	ExecuteOnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)

	// ExecuteOnStreamComplete is called after a streaming response finishes (streaming-only)
	ExecuteOnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
}

ResponseHookManager defines the interface for managing response hooks

type ResponseWriteRequest

type ResponseWriteRequest struct {
	Body    []byte            `json:"body"`
	Headers map[string]string `json:"headers"`
	Context *PluginContext    `json:"context"`
}

ResponseWriteRequest represents a request to modify response body (REST-only)

type ResponseWriteResponse

type ResponseWriteResponse struct {
	Modified bool              `json:"modified"`
	Body     []byte            `json:"body"`
	Headers  map[string]string `json:"headers"`
}

ResponseWriteResponse represents the response from body modification hooks

type RouteDecision

type RouteDecision struct {
	Router RouterRef
	LLMID  uint
	Model  string
	// Pool is the Model Router pool that matched.
	Pool string
	// Route is the route that was chosen, for routers that have named routes.
	Route string
	// Reason says why this target was chosen, from a small fixed set
	// ("model_pattern", ...), for analytics and the X-Tyk-Route-Reason header.
	Reason string
	// Selection is how the target was picked among the candidates (a Model
	// Router pool's algorithm: "round_robin", "weighted"), for analytics.
	Selection string
	// SourceModel is the model the caller asked the router for; Model is what
	// the chosen LLM is asked for. Set by resolveRoute.
	SourceModel string
	// Score is the similarity that decided a Semantic Router's embedding
	// match, 0 otherwise.
	Score float64
	// ShadowRoute is, for a Semantic Router in shadow mode, the route the
	// classifier picked while the default route served the request.
	ShadowRoute string
}

RouteDecision is a resolver's answer: the LLM and model that serve the request, and why.

type RouteRequest

type RouteRequest struct {
	Router RouterRef
	App    *models.App
	// Model is the model the caller asked for, with the router prefix
	// already stripped by the ingress.
	Model string
	// Body is the OpenAI-shaped request body as the caller sent it.
	Body []byte
	// Header is the caller's request header (session affinity and the like).
	Header http.Header
	// Allow restricts the LLMs the resolver may pick. Nil allows every LLM
	// the router can reach.
	Allow func(llmID uint) bool
}

RouteRequest is what a resolver is asked to route.

type RouteResolver

type RouteResolver interface {
	// Lookup reports whether slug names a router.
	Lookup(slug string) (RouterRef, bool)
	// Resolve picks the LLM and model for a request. Errors are
	// ErrRouteNoMatch, ErrRouteNoCandidates or ErrRouteUnavailable, possibly
	// wrapped.
	Resolve(ctx context.Context, req RouteRequest) (*RouteDecision, error)
	// Reaches reports whether the router can send a request to the LLM.
	Reaches(ref RouterRef, llmID uint) bool
	// Models lists the model names the router advertises (GET /v1/models).
	Models(ref RouterRef) []string
}

RouteResolver resolves router slugs. Implementations must be safe for concurrent use.

type RouterKind

type RouterKind string

RouterKind names the kind of router a slug resolves to.

const (
	// RouterKindModel is the Enterprise Model Router: pools matched on the
	// requested model name, vendors picked by round robin or weight.
	RouterKindModel RouterKind = "model_router"
	// RouterKindSemantic is the Enterprise Semantic Router: named routes
	// picked by classifying the prompt (keywords, embeddings, an LLM judge).
	RouterKindSemantic RouterKind = "semantic_router"
)

type RouterRef

type RouterRef struct {
	Kind RouterKind
	ID   uint
	Slug string
}

RouterRef identifies a router.

type SearchQuery

type SearchQuery struct {
	Query string `json:"query"`
	N     int    `json:"n"`
}

type SearchResults

type SearchResults struct {
	Documents []DatasourceDocument `json:"documents"`
}

type SelfHarm

type SelfHarm struct {
	Filtered bool   `json:"filtered"`
	Severity string `json:"severity,omitempty"`
}

type Sexual

type Sexual struct {
	Filtered bool   `json:"filtered"`
	Severity string `json:"severity,omitempty"`
}

type StreamCompleteRequest

type StreamCompleteRequest struct {
	AccumulatedResponse []byte            `json:"accumulated_response"` // Full SSE response (all chunks concatenated)
	Headers             map[string]string `json:"headers"`              // Response headers from upstream
	StatusCode          int               `json:"status_code"`          // HTTP status code
	Context             *PluginContext    `json:"context"`              // Plugin context with request metadata
	ChunkCount          int               `json:"chunk_count"`          // Number of chunks received
	RequestBody         []byte            `json:"request_body"`         // Original request body (for cache key generation)
}

StreamCompleteRequest represents a request to process a completed streaming response

type StreamCompleteResponse

type StreamCompleteResponse struct {
	Handled      bool   `json:"handled"`       // Plugin processed the response
	Cached       bool   `json:"cached"`        // Response was cached (for metrics/logging)
	ErrorMessage string `json:"error_message"` // Error description if any
}

StreamCompleteResponse represents the response from stream complete hooks

type StreamOptions

type StreamOptions struct {
	IncludeUsage bool `json:"include_usage"`
}

type Tool

type Tool struct {
	Type     string         `json:"type"`
	Function FunctionConfig `json:"function"`
}

func (Tool) ToLangchainTool

func (t Tool) ToLangchainTool() llms.Tool

type ValidationError

type ValidationError struct {
	// contains filtered or unexported fields
}

func (*ValidationError) Error

func (e *ValidationError) Error() string

type VectorSearchQuery

type VectorSearchQuery struct {
	Embedding           []float32 `json:"embedding"`
	N                   int       `json:"n"`
	SimilarityThreshold float64   `json:"similarity_threshold"`
}

VectorSearchQuery is the request body for vector-based similarity search.

type VertexMessageExtractor

type VertexMessageExtractor struct {
	GoogleAIMessageExtractor
}

VertexMessageExtractor is an alias for GoogleAIMessageExtractor (same format)

func (*VertexMessageExtractor) VendorName

func (e *VertexMessageExtractor) VendorName() string

VendorName returns "vertex"

type VertexMessageReconstructor

type VertexMessageReconstructor struct {
	GoogleAIMessageReconstructor
}

VertexMessageReconstructor is an alias for GoogleAIMessageReconstructor

func (*VertexMessageReconstructor) VendorName

func (r *VertexMessageReconstructor) VendorName() string

type Violence

type Violence struct {
	Filtered bool   `json:"filtered"`
	Severity string `json:"severity,omitempty"`
}

Jump to

Keyboard shortcuts

? : This menu
/ : Search site
f or F : Jump to
y or Y : Canonical URL