Documentation
¶
Index ¶
- Constants
- Variables
- func AnalyzeCompletionResponse(service services.ServiceInterface, llm *models.LLM, app *models.App, ...)
- func AnalyzeResponse(service services.ServiceInterface, llm *models.LLM, app *models.App, ...)
- func AnalyzeStreamingResponse(service services.ServiceInterface, llm *models.LLM, app *models.App, ...)
- func AnthropicModelExtractor(r *http.Request, body []byte) (string, error)
- func AnthropicValidator(r *http.Request) (string, error)
- func AzureModelExtractor(r *http.Request, body []byte) (string, error)
- func BedrockModelExtractor(r *http.Request, body []byte) (string, error)
- func BedrockValidator(r *http.Request) (string, error)
- func DummyValidator(r *http.Request) (string, error)
- func ExecuteFinalResponseFilters(llm *models.LLM, service services.ServiceInterface, statusCode int, ...) (blocked bool, blockMessage string, err error)
- func ExecuteResponseFilters(llm *models.LLM, service services.ServiceInterface, responseBody []byte, ...) (blocked bool, blockMessage string, err error)
- func ExtractBedrockModelIDFromPath(path string) string
- func GoogleAIModelExtractor(r *http.Request, body []byte) (string, error)
- func GoogleAIValidator(r *http.Request) (string, error)
- func HuggingFaceModelExtractor(r *http.Request, body []byte) (string, error)
- func HuggingFaceValidator(r *http.Request) (string, error)
- func IsInternalHop(r *http.Request) bool
- func MockValidator(r *http.Request) (string, error)
- func NewUnifiedRouterHandler(next http.Handler, basePath string) http.Handler
- func NormalizeUnifiedRouterBasePath(basePath string) string
- func OpenAIModelExtractor(r *http.Request, body []byte) (string, error)
- func OpenAIValidator(r *http.Request) (string, error)
- func ParseUnifiedModel(model string) (route string, bareModel string, err error)
- func VertexModelExtractor(r *http.Request, body []byte) (string, error)
- func VertexValidator(r *http.Request) (string, error)
- func WithAuthIdentity(ctx context.Context, id *AuthIdentity) context.Context
- type APIError
- type AnthropicCacheControl
- type AnthropicContentBlock
- type AnthropicMessage
- type AnthropicMessageExtractor
- type AnthropicMessageReconstructor
- type AnthropicMessagesRequest
- type AnthropicMessagesResponse
- type AnthropicModelListEntry
- type AnthropicModelListResponse
- type AnthropicResponseBlock
- type AnthropicTool
- type AnthropicToolChoice
- type AnthropicUsage
- type AudioConfig
- type AuthHooks
- type AuthIdentity
- type AuthOutcome
- type AuthResult
- type AuthTarget
- type BadRequestError
- type CORSResponseHook
- func (h *CORSResponseHook) GetName() string
- func (h *CORSResponseHook) OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
- func (h *CORSResponseHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
- func (h *CORSResponseHook) OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
- type ChatCompletionChoice
- type ChatCompletionChunk
- type ChatCompletionChunkChoice
- type ChatCompletionDelta
- type ChatCompletionErrorDetail
- type ChatCompletionFunctionDelta
- type ChatCompletionMessage
- type ChatCompletionRequest
- func (r *ChatCompletionRequest) CompletionCount() int
- func (r *ChatCompletionRequest) GetMessageContent(msg Message) (string, []map[string]any, bool)
- func (r *ChatCompletionRequest) GetMessages() []llms.MessageContent
- func (r *ChatCompletionRequest) GetStop() ([]string, bool)
- func (r *ChatCompletionRequest) GetToolChoice() (string, *FunctionConfig, bool)
- func (r *ChatCompletionRequest) OutputTokenLimit() *int
- func (r *ChatCompletionRequest) ToLangchainOptions(conf *models.LLM) []llms.CallOption
- type ChatCompletionResponse
- type ChatCompletionStreamError
- type ChatCompletionToolCallDelta
- type CompletionChoice
- type CompletionResponse
- type CompletionUsage
- type Config
- type ContentFilterHook
- func (h *ContentFilterHook) GetName() string
- func (h *ContentFilterHook) OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
- func (h *ContentFilterHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
- func (h *ContentFilterHook) OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
- type ContentFilterResults
- type CreateCompletionRequest
- type CredentialExtractor
- type CredentialValidator
- func (cv *CredentialValidator) CheckAPICredential(apiKey, dsSlug, llmSlug, routeID, toolSlug string, r *http.Request) (bool, *http.Request)
- func (cv *CredentialValidator) Middleware(next http.Handler) http.Handler
- func (cv *CredentialValidator) RegisterValidator(vendor string, validator CredentialExtractor)
- func (cv *CredentialValidator) SetAuthHooks(hooks *AuthHooks)
- func (cv *CredentialValidator) SetPostAuthCallback(callback PostAuthCallback)
- type DatasourceDocument
- type DefaultResponseHookManager
- func (m *DefaultResponseHookManager) AddHook(hook ResponseHook)
- func (m *DefaultResponseHookManager) ExecuteOnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
- func (m *DefaultResponseHookManager) ExecuteOnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
- func (m *DefaultResponseHookManager) ExecuteOnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
- func (m *DefaultResponseHookManager) HasHooks() bool
- type EmbeddingRequest
- type EmbeddingResponse
- type EndpointMap
- type ErrorResponse
- type ExampleResponseHook
- func (h *ExampleResponseHook) GetName() string
- func (h *ExampleResponseHook) OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
- func (h *ExampleResponseHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
- func (h *ExampleResponseHook) OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
- type FunctionConfig
- type GoogleAIMessageExtractor
- type GoogleAIMessageReconstructor
- type Hate
- type HeadersRequest
- type HeadersResponse
- type InnerError
- type InternalRoutingTransport
- type LogProbs
- type MCPServerCache
- type Message
- type MessageExtractor
- type MessageExtractorRegistry
- type MessageReconstructor
- type MessageReconstructorRegistry
- type MetadataQuery
- type MetadataResults
- type ModelNameExtractor
- type ModelValidator
- type OAIErrorResponse
- type OllamaMessageExtractor
- type OllamaMessageReconstructor
- type OpenAIMessageExtractor
- type OpenAIMessageReconstructor
- type OpenAPICache
- type PluginContext
- type PostAuthCallback
- type Proxy
- func (p *Proxy) AddFilter(filter *models.Filter)
- func (p *Proxy) AddResponseHook(hook ResponseHook)
- func (p *Proxy) CreateChatCompletionHandler(w http.ResponseWriter, r *http.Request)
- func (p *Proxy) CreateCompletionHandler(w http.ResponseWriter, r *http.Request)
- func (p *Proxy) GetDatasource(name string) (*models.Datasource, bool)
- func (p *Proxy) GetLLM(name string) (*models.LLM, bool)
- func (p *Proxy) GetLLMByID(id uint) (*models.LLM, bool)
- func (p *Proxy) GetResponseHookManager() ResponseHookManager
- func (p *Proxy) Handler() http.Handler
- func (p *Proxy) Reload() error
- func (p *Proxy) SetAuthHooks(hooks *AuthHooks)
- func (p *Proxy) SetPostAuthCallback(callback PostAuthCallback)
- func (p *Proxy) SetRouteResolver(r RouteResolver)
- func (p *Proxy) Start() error
- func (p *Proxy) Stop(ctx context.Context) error
- func (p *Proxy) UnifiedRouterBasePath() string
- func (p *Proxy) WaitForAnalytics(ctx context.Context) error
- type ResponseFormat
- type ResponseHook
- type ResponseHookManager
- type ResponseWriteRequest
- type ResponseWriteResponse
- type RouteDecision
- type RouteRequest
- type RouteResolver
- type RouterKind
- type RouterRef
- type SearchQuery
- type SearchResults
- type SelfHarm
- type Sexual
- type StreamCompleteRequest
- type StreamCompleteResponse
- type StreamOptions
- type Tool
- type ValidationError
- type VectorSearchQuery
- type VertexMessageExtractor
- type VertexMessageReconstructor
- type Violence
Constants ¶
const ( AuthTargetLLM = "llm" AuthTargetDatasource = "datasource" AuthTargetTool = "tool" AuthTargetModelRouter = "model_router" AuthTargetSemanticRouter = "semantic_router" AuthTargetPlugin = "plugin" // a custom_endpoint plugin's /plugins/{slug}/ routes )
Kinds of endpoint an auth plugin can be attached to (AuthTarget.Kind).
const ( CredTypeBearer = "bearer" // Authorization: Bearer ... CredTypeAPIKey = "api_key" // a vendor key header, x-api-key, a bare Authorization value or ?apiKey= )
Credential types passed to CustomAuth.
const ( AuthMethodAppKey = authidentity.MethodAppKey AuthMethodOAuth = authidentity.MethodOAuth AuthMethodPlugin = authidentity.MethodPlugin )
Authentication methods (AuthIdentity.Method).
const ( LLMPRefix = "/llm/" DatasourcePrefix = "/datasource/" ToolPrefix = "/tools/" MCPSuffix = "/mcp" )
const DefaultUnifiedRouterBasePath = "/v1"
DefaultUnifiedRouterBasePath is where the unified ingress lives unless the embedder moves it. It matches the OpenAI/OpenRouter convention so an unmodified SDK pointed at the gateway root works with no path configuration.
Variables ¶
var ( // ErrRouteNoMatch: the request names nothing the router routes (400). ErrRouteNoMatch = errors.New("no route matches the requested model") // ErrRouteNoCandidates: the router would route the request, but to // nothing this App may use (403). ErrRouteNoCandidates = errors.New("the router has no target this app may use") // unloaded (503). ErrRouteUnavailable = errors.New("the router has no available target") )
Functions ¶
func AnalyzeResponse ¶
func AnthropicModelExtractor ¶
func BedrockModelExtractor ¶
BedrockModelExtractor extracts the model ID from a Bedrock Converse API request. The model ID is sent in the URL path (e.g., /model/{modelId}/converse) or request body.
func BedrockValidator ¶
BedrockValidator extracts the gateway app API key from the AWS SigV4 Authorization header. Clients set their gateway app API key as the AWS_ACCESS_KEY_ID; the gateway extracts it from the Credential field of the SigV4 header. Format: AWS4-HMAC-SHA256 Credential=ACCESS_KEY_ID/date/region/service/aws4_request, ...
func ExecuteFinalResponseFilters ¶
func ExecuteFinalResponseFilters( llm *models.LLM, service services.ServiceInterface, statusCode int, chunkIndex int, fullText string, r *http.Request, ) (blocked bool, blockMessage string, err error)
ExecuteFinalResponseFilters runs the response filters once more when a streaming response has completed, over the whole accumulated text. Per-chunk evaluation can legitimately skip a response that never reaches a script's buffer threshold or a guardrail's cadence; this pass makes sure every streamed response is looked at in full at least once. The chunks are already with the client, so a block here ends the stream with an error event and is recorded, rather than withholding content.
func ExecuteResponseFilters ¶
func ExecuteResponseFilters( llm *models.LLM, service services.ServiceInterface, responseBody []byte, statusCode int, isStreaming bool, isChunk bool, chunkIndex int, currentBuffer string, r *http.Request, ) (blocked bool, blockMessage string, err error)
ExecuteResponseFilters executes response-side filters on LLM responses Returns whether the response should be blocked and an optional block message
func ExtractBedrockModelIDFromPath ¶
ExtractBedrockModelIDFromPath extracts the model ID from a Bedrock URL path. Native Bedrock API paths contain /model/{modelId}/converse or /model/{modelId}/converse-stream.
func GoogleAIModelExtractor ¶
func GoogleAIValidator ¶
GoogleAIValidator extracts and validates the API key from the incoming request. It checks both the 'x-goog-api-key' header and the 'key' query parameter.
func IsInternalHop ¶
IsInternalHop reports whether r is the gateway calling itself on its loopback hop: it carries this process's hop token and arrives over a loopback connection.
func NewUnifiedRouterHandler ¶
NewUnifiedRouterHandler returns the handler for the unified router ingress. next receives the rewritten request and is expected to be the /ai/ handler chain (auth middleware included).
The handler deliberately does NOT resolve the route prefix itself: every well-formed request is rewritten and forwarded so that authentication runs before route resolution. An anonymous caller therefore always gets 401 and cannot enumerate configured route slugs from 404-vs-401 responses; the /ai/ shim produces the vendor-not-found 404 after auth.
func NormalizeUnifiedRouterBasePath ¶
NormalizeUnifiedRouterBasePath cleans a configured ingress base path into the form the router matches on: a leading slash, no trailing slash, no empty or dot segments. Empty or unusable input falls back to DefaultUnifiedRouterBasePath: a gateway with the ingress silently mounted at "/" would swallow every other route, and a segment carrying router metacharacters would register a wildcard or path-variable route instead of the literal prefix the operator asked for.
Embedders that need the ingress gone entirely disable it (Config.DisableUnifiedRouter) rather than passing an empty path.
func ParseUnifiedModel ¶
ParseUnifiedModel splits an OpenRouter-style model string into its vendor route prefix and the bare model name. The split is on the FIRST slash so model names that themselves contain slashes (e.g. fine-tune identifiers) survive intact.
func WithAuthIdentity ¶
func WithAuthIdentity(ctx context.Context, id *AuthIdentity) context.Context
WithAuthIdentity returns ctx carrying id.
Types ¶
type AnthropicCacheControl ¶
type AnthropicCacheControl struct {
Type string `json:"type"` // "ephemeral"
}
AnthropicCacheControl marks a block as a prompt-cache breakpoint.
type AnthropicContentBlock ¶
type AnthropicContentBlock struct {
Type string `json:"type"` // "text" | "tool_use" | "tool_result"
// text
Text string `json:"text,omitempty"`
// tool_use (assistant asking to call a tool)
ID string `json:"id,omitempty"`
Name string `json:"name,omitempty"`
Input json.RawMessage `json:"input,omitempty"`
// tool_result (user returning a tool's output)
ToolUseID string `json:"tool_use_id,omitempty"`
Content json.RawMessage `json:"content,omitempty"` // string | []AnthropicContentBlock
IsError bool `json:"is_error,omitempty"`
// prompt caching marker
CacheControl *AnthropicCacheControl `json:"cache_control,omitempty"`
}
AnthropicContentBlock is one block within a message's content array (or a system block). Different block types populate different fields.
type AnthropicMessage ¶
type AnthropicMessage struct {
Role string `json:"role"` // "user" | "assistant"
Content json.RawMessage `json:"content"`
}
AnthropicMessage is a single conversation turn. Content is either a plain string or an array of typed content blocks.
type AnthropicMessageExtractor ¶
type AnthropicMessageExtractor struct{}
AnthropicMessageExtractor extracts messages from Anthropic-format requests
func (*AnthropicMessageExtractor) ExtractMessages ¶
func (e *AnthropicMessageExtractor) ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)
ExtractMessages parses Anthropic's format (separate system field + messages array)
func (*AnthropicMessageExtractor) VendorName ¶
func (e *AnthropicMessageExtractor) VendorName() string
VendorName returns "anthropic"
type AnthropicMessageReconstructor ¶
type AnthropicMessageReconstructor struct{}
AnthropicMessageReconstructor rebuilds Anthropic-format requests
func (*AnthropicMessageReconstructor) Reconstruct ¶
func (r *AnthropicMessageReconstructor) Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)
func (*AnthropicMessageReconstructor) VendorName ¶
func (r *AnthropicMessageReconstructor) VendorName() string
type AnthropicMessagesRequest ¶
type AnthropicMessagesRequest struct {
Model string `json:"model"`
Messages []AnthropicMessage `json:"messages"`
System json.RawMessage `json:"system,omitempty"` // string | []AnthropicContentBlock
MaxTokens int `json:"max_tokens"`
Temperature *float64 `json:"temperature,omitempty"`
TopP *float64 `json:"top_p,omitempty"`
StopSequences []string `json:"stop_sequences,omitempty"`
Tools []AnthropicTool `json:"tools,omitempty"`
ToolChoice *AnthropicToolChoice `json:"tool_choice,omitempty"`
Stream bool `json:"stream,omitempty"`
}
AnthropicMessagesRequest is the body of POST /anthropic/{routeId}/v1/messages.
type AnthropicMessagesResponse ¶
type AnthropicMessagesResponse struct {
ID string `json:"id"`
Type string `json:"type"` // "message"
Role string `json:"role"` // "assistant"
Model string `json:"model"`
Content []AnthropicResponseBlock `json:"content"`
StopReason string `json:"stop_reason,omitempty"`
StopSequence *string `json:"stop_sequence"`
Usage AnthropicUsage `json:"usage"`
}
AnthropicMessagesResponse is the non-streaming response, and also the message skeleton embedded in the streaming "message_start" event.
type AnthropicModelListEntry ¶
type AnthropicModelListEntry struct {
Type string `json:"type"` // "model"
ID string `json:"id"`
DisplayName string `json:"display_name"`
}
AnthropicModelListEntry is one model in the list.
type AnthropicModelListResponse ¶
type AnthropicModelListResponse struct {
Data []AnthropicModelListEntry `json:"data"`
HasMore bool `json:"has_more"`
}
AnthropicModelListResponse is the body of GET /anthropic/{routeId}/v1/models, the list Claude Code reads for gateway model discovery. Data is never null.
type AnthropicResponseBlock ¶
type AnthropicResponseBlock struct {
Type string `json:"type"`
Text string `json:"text,omitempty"`
ID string `json:"id,omitempty"`
Name string `json:"name,omitempty"`
Input json.RawMessage `json:"input,omitempty"`
}
AnthropicResponseBlock is an output content block (text or tool_use).
type AnthropicTool ¶
type AnthropicTool struct {
Name string `json:"name"`
Description string `json:"description,omitempty"`
InputSchema map[string]any `json:"input_schema,omitempty"`
CacheControl *AnthropicCacheControl `json:"cache_control,omitempty"`
}
AnthropicTool is a tool definition the model may call.
type AnthropicToolChoice ¶
type AnthropicToolChoice struct {
Type string `json:"type"` // "auto" | "any" | "tool"
Name string `json:"name,omitempty"`
}
AnthropicToolChoice controls whether/which tool the model must use.
type AnthropicUsage ¶
type AnthropicUsage struct {
InputTokens int `json:"input_tokens"`
OutputTokens int `json:"output_tokens"`
CacheCreationInputTokens int `json:"cache_creation_input_tokens,omitempty"`
CacheReadInputTokens int `json:"cache_read_input_tokens,omitempty"`
}
AnthropicUsage carries token counts. input_tokens is reported in message_start, output_tokens in message_delta (streaming) or both together (non-streaming).
type AudioConfig ¶
type AuthHooks ¶
type AuthHooks struct {
// PreAuth runs BEFORE credential extraction and validation.
// Only called for non-OAuth requests (LLM proxy requests).
// Has NO access to authenticated user/app data.
// Return true to block the request.
PreAuth func(w http.ResponseWriter, r *http.Request) bool
// CustomAuth lets auth plugins authenticate a request in place of app
// keys. It is called with the endpoint the request addresses (an LLM,
// datasource, tool, router or plugin endpoint) and the credential the
// request presented, for every request that is not a Studio OAuth access
// token.
//
// An endpoint with auth plugins attached is authenticated by them alone:
// the hook answers AuthAccepted or AuthRejected, and app keys are not
// tried. An endpoint without any gets AuthNotHandled, and app keys apply
// as usual. An error means the plugins could not be asked (none could be
// loaded, say); the request is refused with a 503.
CustomAuth func(r *http.Request, target AuthTarget, credential, credType string) (AuthResult, error)
// PostAuth runs AFTER successful authentication (standard or custom).
// Only called for non-OAuth requests (LLM proxy requests).
// Receives the authenticated app ID.
// Return true to block the request.
PostAuth func(w http.ResponseWriter, r *http.Request, appID uint) bool
}
AuthHooks provides extension points in the authentication lifecycle. These hooks are for LLM proxy requests, NOT for OAuth/MCP flows. All hooks are optional (nil checks are performed before calling).
type AuthIdentity ¶
type AuthIdentity = authidentity.Identity
AuthIdentity is who authenticated a request and how. The credential validator puts it on the request context for every method; analytics, the post-auth hook and the plugins read it. It lives in pkg/authidentity so analytics can read it without importing the proxy.
func AuthIdentityFromContext ¶
func AuthIdentityFromContext(ctx context.Context) *AuthIdentity
AuthIdentityFromContext returns the identity the credential validator recorded, or nil.
type AuthOutcome ¶
type AuthOutcome int
AuthOutcome is what the auth plugins decided.
const ( // AuthNotHandled: the endpoint has no auth plugins; app keys apply. AuthNotHandled AuthOutcome = iota // AuthAccepted: a plugin authenticated the request as AuthResult.AppID. AuthAccepted // AuthRejected: the endpoint's plugins all refused the credential. AuthRejected )
type AuthResult ¶
type AuthResult struct {
Outcome AuthOutcome
AppID uint
// Subject is who the call is for, when the credential says (the auth
// response's user id: a delegated token's sub, say). For audit only.
Subject string
// Claims the plugin returned alongside, for audit and post-auth plugins.
Claims map[string]string
PluginID uint
PluginName string
// Reason is why the plugins refused, for the log. It is not sent to the
// caller.
Reason string
}
AuthResult is the outcome of CustomAuth.
type AuthTarget ¶
AuthTarget is the endpoint a request addresses, resolved from its path.
type BadRequestError ¶
type BadRequestError struct {
// contains filtered or unexported fields
}
func (*BadRequestError) Error ¶
func (e *BadRequestError) Error() string
type CORSResponseHook ¶
type CORSResponseHook struct{}
CORSResponseHook adds CORS headers to responses
func NewCORSResponseHook ¶
func NewCORSResponseHook() *CORSResponseHook
func (*CORSResponseHook) GetName ¶
func (h *CORSResponseHook) GetName() string
func (*CORSResponseHook) OnBeforeWrite ¶
func (h *CORSResponseHook) OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
func (*CORSResponseHook) OnBeforeWriteHeaders ¶
func (h *CORSResponseHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
func (*CORSResponseHook) OnStreamComplete ¶
func (h *CORSResponseHook) OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
OnStreamComplete is called after a streaming response finishes
type ChatCompletionChoice ¶
type ChatCompletionChoice struct {
Index int `json:"index"`
Message ChatCompletionMessage `json:"message"`
FinishReason string `json:"finish_reason"`
}
type ChatCompletionChunk ¶
type ChatCompletionChunk struct {
ID string `json:"id"`
Object string `json:"object"` // "chat.completion.chunk"
Created int64 `json:"created"`
Model string `json:"model"`
Choices []ChatCompletionChunkChoice `json:"choices"`
Usage *CompletionUsage `json:"usage,omitempty"` // Only in final chunk
}
ChatCompletionChunk represents a streaming chunk in OpenAI format
type ChatCompletionChunkChoice ¶
type ChatCompletionChunkChoice struct {
Index int `json:"index"`
Delta ChatCompletionDelta `json:"delta"`
FinishReason *string `json:"finish_reason"` // null until final chunk
}
ChatCompletionChunkChoice represents a choice in a streaming chunk
type ChatCompletionDelta ¶
type ChatCompletionDelta struct {
Role string `json:"role,omitempty"` // Only in first chunk
Content string `json:"content,omitempty"` // Streaming text content
ToolCalls []ChatCompletionToolCallDelta `json:"tool_calls,omitempty"`
}
ChatCompletionDelta represents the delta content in a streaming chunk
type ChatCompletionErrorDetail ¶
ChatCompletionErrorDetail contains error details
type ChatCompletionFunctionDelta ¶
type ChatCompletionFunctionDelta struct {
Name string `json:"name,omitempty"`
Arguments string `json:"arguments,omitempty"`
}
ChatCompletionFunctionDelta is the function half of a streamed tool call.
type ChatCompletionMessage ¶
type ChatCompletionRequest ¶
type ChatCompletionRequest struct {
Messages []Message `json:"messages"`
Model string `json:"model"`
Store *bool `json:"store,omitempty"`
Metadata map[string]any `json:"metadata,omitempty"`
FrequencyPenalty *float64 `json:"frequency_penalty,omitempty"`
LogitBias map[string]int `json:"logit_bias,omitempty"`
LogProbs *bool `json:"logprobs,omitempty"`
TopLogProbs *int `json:"top_logprobs,omitempty"`
MaxCompletionTokens *int `json:"max_completion_tokens,omitempty"`
MaxTokens *int `json:"max_tokens,omitempty"`
ReasoningEffort *string `json:"reasoning_effort,omitempty"`
N *int `json:"n,omitempty"`
Modalities []string `json:"modalities,omitempty"`
Audio *AudioConfig `json:"audio,omitempty"`
PresencePenalty *float64 `json:"presence_penalty,omitempty"`
ResponseFormat *ResponseFormat `json:"response_format,omitempty"`
Seed *int `json:"seed,omitempty"`
ServiceTier *string `json:"service_tier,omitempty"`
Stop any `json:"stop,omitempty"`
Stream *bool `json:"stream,omitempty"`
StreamOptions *StreamOptions `json:"stream_options,omitempty"`
Temperature *float64 `json:"temperature,omitempty"`
TopP *float64 `json:"top_p,omitempty"`
Tools []Tool `json:"tools,omitempty"`
ToolChoice any `json:"tool_choice,omitempty"`
ParallelToolCalls *bool `json:"parallel_tool_calls,omitempty"`
User *string `json:"user,omitempty"`
}
Models
func (*ChatCompletionRequest) CompletionCount ¶
func (r *ChatCompletionRequest) CompletionCount() int
func (*ChatCompletionRequest) GetMessageContent ¶
func (*ChatCompletionRequest) GetMessages ¶
func (r *ChatCompletionRequest) GetMessages() []llms.MessageContent
func (*ChatCompletionRequest) GetStop ¶
func (r *ChatCompletionRequest) GetStop() ([]string, bool)
Getter functions for ambiguous fields
func (*ChatCompletionRequest) GetToolChoice ¶
func (r *ChatCompletionRequest) GetToolChoice() (string, *FunctionConfig, bool)
func (*ChatCompletionRequest) OutputTokenLimit ¶
func (r *ChatCompletionRequest) OutputTokenLimit() *int
CompletionCount is the number of completions the caller asked for: the request's `n`, defaulting to OpenAI's own default of 1. OutputTokenLimit is the cap on generated tokens the client asked for: max_completion_tokens when set, otherwise max_tokens (the older name, still what most clients send); nil when neither is.
func (*ChatCompletionRequest) ToLangchainOptions ¶
func (r *ChatCompletionRequest) ToLangchainOptions(conf *models.LLM) []llms.CallOption
type ChatCompletionResponse ¶
type ChatCompletionResponse struct {
ID string `json:"id"`
Object string `json:"object"`
Created int64 `json:"created"`
Model string `json:"model"`
Choices []ChatCompletionChoice `json:"choices"`
Usage CompletionUsage `json:"usage"`
}
func NewChatCompletionResponse ¶
func NewChatCompletionResponse(llmResponse *llms.ContentResponse, model string, n int) *ChatCompletionResponse
NewChatCompletionResponse converts a langchaingo ContentResponse into the OpenAI chat.completion object.
n is the number of completions the caller asked for (the request's `n`, or 1 when it is omitted). It matters because langchaingo drivers do not agree on what a ContentChoice represents: OpenAI's returns one per requested completion, while Anthropic's returns one per *content block* of a single turn - so a reply carrying text and a tool call arrives as two choices.
OpenAI indexes `choices` by n and puts every part of one turn in the same message. Passing the block-per-choice shape straight through produced choices[0] with finish_reason:"tool_calls" and no tool_calls array, plus a choices[1] holding the call: an SDK reads choices[0], finds nothing to invoke, and breaks. Anything beyond the requested n is therefore folded back into a single turn.
type ChatCompletionStreamError ¶
type ChatCompletionStreamError struct {
Error ChatCompletionErrorDetail `json:"error"`
}
ChatCompletionStreamError represents an error in SSE format
type ChatCompletionToolCallDelta ¶
type ChatCompletionToolCallDelta struct {
Index int `json:"index"`
ID string `json:"id,omitempty"`
Type string `json:"type,omitempty"`
Function *ChatCompletionFunctionDelta `json:"function,omitempty"`
}
ChatCompletionToolCallDelta is one fragment of a streamed tool call.
Index is what makes reassembly possible and is required on every fragment: it is the only thing tying an arguments fragment back to the call it belongs to when a model emits several in parallel. Id and Function.Name appear on the opening fragment; subsequent fragments carry argument text only, and the concatenation of those fragments must parse as JSON.
type CompletionChoice ¶
type CompletionResponse ¶
type CompletionResponse struct {
ID string `json:"id"`
Object string `json:"object"`
Created int64 `json:"created"`
Model string `json:"model"`
Choices []CompletionChoice `json:"choices"`
Usage CompletionUsage `json:"usage"`
}
type CompletionUsage ¶
type Config ¶
type Config struct {
Port int
// TLSEnabled reports that the listener this proxy is served from terminates
// TLS. The /ai/ and unified-router handlers re-enter the gateway with a
// loopback HTTP call to /llm/call/{slug} on Port (see getInternalLLMBaseURL);
// that hop must speak HTTPS when the listener does, otherwise the server
// rejects it with "client sent an HTTP request to an HTTPS server" and every
// OpenAI-compatible request fails. The loopback connects to 127.0.0.1 and the
// serving certificate is issued for the public hostname, so it does not
// verify the certificate: it is talking to its own process.
TLSEnabled bool
// LLMTimeout is the timeout for upstream LLM HTTP requests (both streaming and REST).
// Defaults to 5 minutes if zero, suitable for long-running agentic workloads.
LLMTimeout time.Duration
// Server timeouts for the standalone proxy HTTP server (used by proxy.Start()).
// These are not used when the proxy is mounted into another server.
// Zero values use defaults: ReadTimeout=5m, WriteTimeout=10m, IdleTimeout=5m.
ServerReadTimeout time.Duration
ServerWriteTimeout time.Duration
ServerIdleTimeout time.Duration
// DebugHTTPProxy logs every request the standalone proxy server
// receives. DEBUG_HTTP_PROXY=true also turns it on.
DebugHTTPProxy bool
// Datasource endpoint limits (zero values use defaults)
DatasourceMaxBodyBytes int64 // Max request body size in bytes (default: 1MB)
DatasourceMaxResults int // Max documents returned per query (default: 100)
DatasourceMaxEmbedTexts int // Max texts per embedding request (default: 100)
// UnifiedRouterBasePath is where the OpenRouter-style single-endpoint ingress
// is mounted: "{base}/chat/completions", "{base}/completions" and "{base}/models".
// Empty uses DefaultUnifiedRouterBasePath ("/v1").
//
// This exists because the proxy is embedded in host gateways (e.g. Tyk's API
// Gateway) that already own "/v1" or mount the proxy under their own prefix.
// The value is normalized by NormalizeUnifiedRouterBasePath; the base path of
// the internal per-route shims (/ai/{slug}/v1/...) is unaffected by it.
UnifiedRouterBasePath string
// DisableUnifiedRouter removes the unified ingress entirely: no base-path
// interception and no {base}/models route, so those paths fall through to the
// normal authenticated router (404). Per-route endpoints (/ai/, /llm/,
// /anthropic/) are unaffected. Hosts that expose their own single endpoint,
// or that must not have the proxy claim a shared path prefix, set this.
DisableUnifiedRouter bool
// ServerTiming adds a Server-Timing header and trailer to LLM responses that
// split each request into gateway time and upstream time (see
// server_timing.go). Off by default; meant for benchmarking and diagnosis.
ServerTiming bool
}
type ContentFilterHook ¶
type ContentFilterHook struct {
// contains filtered or unexported fields
}
ContentFilterHook demonstrates content filtering
func NewContentFilterHook ¶
func NewContentFilterHook(blockedWords []string) *ContentFilterHook
func (*ContentFilterHook) GetName ¶
func (h *ContentFilterHook) GetName() string
func (*ContentFilterHook) OnBeforeWrite ¶
func (h *ContentFilterHook) OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
func (*ContentFilterHook) OnBeforeWriteHeaders ¶
func (h *ContentFilterHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
func (*ContentFilterHook) OnStreamComplete ¶
func (h *ContentFilterHook) OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
OnStreamComplete is called after a streaming response finishes
type ContentFilterResults ¶
type CreateCompletionRequest ¶
type CreateCompletionRequest struct {
Model string `json:"model"`
Prompt string `json:"prompt,omitempty"`
MaxTokens *int `json:"max_tokens,omitempty"`
Temperature *float64 `json:"temperature,omitempty"`
TopP *float64 `json:"top_p,omitempty"`
N *int `json:"n,omitempty"`
Stream *bool `json:"stream,omitempty"`
LogProbs *int `json:"logprobs,omitempty"`
Echo *bool `json:"echo,omitempty"`
Stop interface{} `json:"stop,omitempty"`
PresencePenalty *float64 `json:"presence_penalty,omitempty"`
FrequencyPenalty *float64 `json:"frequency_penalty,omitempty"`
User string `json:"user,omitempty"`
}
type CredentialValidator ¶
type CredentialValidator struct {
// contains filtered or unexported fields
}
func NewCredentialValidator ¶
func NewCredentialValidator(service services.ServiceInterface, proxy *Proxy) *CredentialValidator
func (*CredentialValidator) CheckAPICredential ¶
func (cv *CredentialValidator) CheckAPICredential(apiKey, dsSlug, llmSlug, routeID, toolSlug string, r *http.Request) (bool, *http.Request)
Renamed from CheckCredential to CheckAPICredential to differentiate
func (*CredentialValidator) Middleware ¶
func (cv *CredentialValidator) Middleware(next http.Handler) http.Handler
func (*CredentialValidator) RegisterValidator ¶
func (cv *CredentialValidator) RegisterValidator(vendor string, validator CredentialExtractor)
func (*CredentialValidator) SetAuthHooks ¶
func (cv *CredentialValidator) SetAuthHooks(hooks *AuthHooks)
SetAuthHooks sets authentication lifecycle hooks
func (*CredentialValidator) SetPostAuthCallback ¶
func (cv *CredentialValidator) SetPostAuthCallback(callback PostAuthCallback)
SetPostAuthCallback is deprecated, use SetAuthHooks instead Kept for backward compatibility
type DatasourceDocument ¶
type DatasourceDocument struct {
PageContent string `json:"PageContent"`
Metadata map[string]any `json:"Metadata"`
Score float32 `json:"Score"`
}
DatasourceDocument is the REST response format for datasource search results. Field names match the original schema.Document marshaling for backward compatibility.
type DefaultResponseHookManager ¶
type DefaultResponseHookManager struct {
// contains filtered or unexported fields
}
DefaultResponseHookManager provides a basic implementation of ResponseHookManager
func NewDefaultResponseHookManager ¶
func NewDefaultResponseHookManager() *DefaultResponseHookManager
NewDefaultResponseHookManager creates a new default response hook manager
func (*DefaultResponseHookManager) AddHook ¶
func (m *DefaultResponseHookManager) AddHook(hook ResponseHook)
AddHook adds a response hook to the manager
func (*DefaultResponseHookManager) ExecuteOnBeforeWrite ¶
func (m *DefaultResponseHookManager) ExecuteOnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
ExecuteOnBeforeWrite executes all registered body hooks
func (*DefaultResponseHookManager) ExecuteOnBeforeWriteHeaders ¶
func (m *DefaultResponseHookManager) ExecuteOnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
ExecuteOnBeforeWriteHeaders executes all registered header hooks
func (*DefaultResponseHookManager) ExecuteOnStreamComplete ¶
func (m *DefaultResponseHookManager) ExecuteOnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
ExecuteOnStreamComplete executes all registered stream complete hooks
func (*DefaultResponseHookManager) HasHooks ¶
func (m *DefaultResponseHookManager) HasHooks() bool
HasHooks returns true if there are any hooks configured
type EmbeddingRequest ¶
type EmbeddingRequest struct {
Texts []string `json:"texts"`
}
EmbeddingRequest is the request body for embedding generation.
type EmbeddingResponse ¶
type EmbeddingResponse struct {
Vectors [][]float32 `json:"vectors"`
}
EmbeddingResponse is the response containing generated embedding vectors.
type ErrorResponse ¶
type ExampleResponseHook ¶
type ExampleResponseHook struct {
// contains filtered or unexported fields
}
ExampleResponseHook demonstrates how to create a custom response hook
func NewExampleResponseHook ¶
func NewExampleResponseHook(name string) *ExampleResponseHook
NewExampleResponseHook creates a new example response hook
func (*ExampleResponseHook) GetName ¶
func (h *ExampleResponseHook) GetName() string
GetName returns the hook name
func (*ExampleResponseHook) OnBeforeWrite ¶
func (h *ExampleResponseHook) OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
OnBeforeWrite modifies response body content
func (*ExampleResponseHook) OnBeforeWriteHeaders ¶
func (h *ExampleResponseHook) OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
OnBeforeWriteHeaders adds custom headers to responses
func (*ExampleResponseHook) OnStreamComplete ¶
func (h *ExampleResponseHook) OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
OnStreamComplete is called after a streaming response finishes
type FunctionConfig ¶
type GoogleAIMessageExtractor ¶
type GoogleAIMessageExtractor struct{}
GoogleAIMessageExtractor extracts messages from Google AI/Vertex format requests
func (*GoogleAIMessageExtractor) ExtractMessages ¶
func (e *GoogleAIMessageExtractor) ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)
ExtractMessages parses Google AI's contents/parts structure
func (*GoogleAIMessageExtractor) VendorName ¶
func (e *GoogleAIMessageExtractor) VendorName() string
VendorName returns "google_ai"
type GoogleAIMessageReconstructor ¶
type GoogleAIMessageReconstructor struct{}
GoogleAIMessageReconstructor rebuilds Google AI/Vertex-format requests
func (*GoogleAIMessageReconstructor) Reconstruct ¶
func (r *GoogleAIMessageReconstructor) Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)
func (*GoogleAIMessageReconstructor) VendorName ¶
func (r *GoogleAIMessageReconstructor) VendorName() string
type HeadersRequest ¶
type HeadersRequest struct {
Headers map[string]string `json:"headers"`
Context *PluginContext `json:"context"`
}
HeadersRequest represents a request to modify response headers
type HeadersResponse ¶
type HeadersResponse struct {
Modified bool `json:"modified"`
Headers map[string]string `json:"headers"`
}
HeadersResponse represents the response from header modification hooks
type InnerError ¶
type InnerError struct {
Code string `json:"code,omitempty"`
ContentFilterResults ContentFilterResults `json:"content_filter_result,omitempty"`
}
type InternalRoutingTransport ¶
type InternalRoutingTransport struct {
// contains filtered or unexported fields
}
InternalRoutingTransport intercepts SDK HTTP calls for internal routing. When the /ai/ (OpenAI compatibility) endpoint routes requests through /llm/, this transport: 1. Strips vendor-specific auth headers set by the SDK 2. Passes through the original client's Authorization header
This allows /llm/ to authenticate the request using the client's credentials and then set the correct vendor auth (from stored LLM config) before forwarding.
func NewInternalRoutingTransport ¶
func NewInternalRoutingTransport(originalAuth string, serverTLS bool) *InternalRoutingTransport
NewInternalRoutingTransport creates a transport that passes through the original client auth header while stripping any vendor-specific auth headers set by the SDK.
serverTLS must be true when the listener being called back terminates TLS: the loopback then dials 127.0.0.1 over HTTPS without verifying the certificate, which is issued for the public hostname rather than the loopback address.
This builds a private connection pool; the proxy's request path uses newInternalRoutingTransport with a shared pool instead.
type MCPServerCache ¶
type MessageExtractor ¶
type MessageExtractor interface {
// ExtractMessages parses the request body and returns normalized messages
ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)
// VendorName returns the vendor this extractor handles
VendorName() string
}
MessageExtractor converts vendor-specific request formats to langchaingo MessageContent
type MessageExtractorRegistry ¶
type MessageExtractorRegistry struct {
// contains filtered or unexported fields
}
MessageExtractorRegistry manages extractors for different vendors
func NewMessageExtractorRegistry ¶
func NewMessageExtractorRegistry() *MessageExtractorRegistry
NewMessageExtractorRegistry creates a new registry for message extractors
func (*MessageExtractorRegistry) Extract ¶
func (r *MessageExtractorRegistry) Extract(vendor string, req *http.Request, body []byte) ([]llms.MessageContent, error)
Extract uses the appropriate extractor to parse messages from the request
func (*MessageExtractorRegistry) Register ¶
func (r *MessageExtractorRegistry) Register(extractor MessageExtractor)
Register adds a message extractor to the registry
type MessageReconstructor ¶
type MessageReconstructor interface {
// Reconstruct takes normalized messages and the original request body,
// returning a new request body with modified messages
Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)
// VendorName returns the vendor this reconstructor handles
VendorName() string
}
MessageReconstructor rebuilds vendor-specific request bodies from normalized messages
type MessageReconstructorRegistry ¶
type MessageReconstructorRegistry struct {
// contains filtered or unexported fields
}
MessageReconstructorRegistry manages reconstructors for different vendors
func NewMessageReconstructorRegistry ¶
func NewMessageReconstructorRegistry() *MessageReconstructorRegistry
NewMessageReconstructorRegistry creates a new registry
func (*MessageReconstructorRegistry) Reconstruct ¶
func (r *MessageReconstructorRegistry) Reconstruct(vendor string, messages []map[string]interface{}, originalBody []byte) ([]byte, error)
Reconstruct uses the appropriate reconstructor for the vendor
func (*MessageReconstructorRegistry) Register ¶
func (r *MessageReconstructorRegistry) Register(reconstructor MessageReconstructor)
Register adds a reconstructor to the registry
type MetadataQuery ¶
type MetadataQuery struct {
Filter map[string]string `json:"filter"`
FilterMode string `json:"filter_mode"`
Limit int `json:"limit"`
Offset int `json:"offset"`
}
MetadataQuery is the request body for metadata-only document queries.
type MetadataResults ¶
type MetadataResults struct {
Documents []DatasourceDocument `json:"documents"`
TotalCount int `json:"total_count"`
}
MetadataResults is the response for metadata queries with pagination.
type ModelNameExtractor ¶
type ModelValidator ¶
type ModelValidator struct {
// contains filtered or unexported fields
}
func NewModelValidator ¶
func NewModelValidator(allowedModels []string) *ModelValidator
func (*ModelValidator) IsModelAllowed ¶
func (mv *ModelValidator) IsModelAllowed(modelName string) bool
IsModelAllowed reports whether a model name is permitted by the configured patterns.
Two behaviours worth stating plainly, because both surprise people:
- An empty pattern list allows EVERY model. An empty Allowed Models field is not a deny-all.
- Patterns are matched UNANCHORED, i.e. anywhere within the model name. So "gpt-4.*" -- the example the UI itself suggests -- also matches "legacy-gpt-4o" and "not-really-gpt-4". Anchor explicitly with ^ and $ if you mean the whole name.
The unanchored behaviour is kept deliberately: existing configurations rely on substring matching, and silently anchoring them would start rejecting models that are allowed today.
The rule itself lives in pkg/modelmatch so save-time validation in the service layer and request-time validation here cannot drift apart.
func (*ModelValidator) RegisterExtractor ¶
func (mv *ModelValidator) RegisterExtractor(vendor string, extractor ModelNameExtractor)
func (*ModelValidator) ValidateRequest ¶
func (mv *ModelValidator) ValidateRequest(body []byte) error
type OAIErrorResponse ¶
type OAIErrorResponse struct {
Error *APIError `json:"error,omitempty"`
}
type OllamaMessageExtractor ¶
type OllamaMessageExtractor struct {
OpenAIMessageExtractor
}
OllamaMessageExtractor uses OpenAI format
func (*OllamaMessageExtractor) VendorName ¶
func (e *OllamaMessageExtractor) VendorName() string
VendorName returns "ollama"
type OllamaMessageReconstructor ¶
type OllamaMessageReconstructor struct {
OpenAIMessageReconstructor
}
OllamaMessageReconstructor uses OpenAI format
func (*OllamaMessageReconstructor) VendorName ¶
func (r *OllamaMessageReconstructor) VendorName() string
type OpenAIMessageExtractor ¶
type OpenAIMessageExtractor struct{}
OpenAIMessageExtractor extracts messages from OpenAI-format requests
func (*OpenAIMessageExtractor) ExtractMessages ¶
func (e *OpenAIMessageExtractor) ExtractMessages(r *http.Request, body []byte) ([]llms.MessageContent, error)
ExtractMessages wraps the existing ChatCompletionRequest.GetMessages() method
func (*OpenAIMessageExtractor) VendorName ¶
func (e *OpenAIMessageExtractor) VendorName() string
VendorName returns "openai"
type OpenAIMessageReconstructor ¶
type OpenAIMessageReconstructor struct{}
OpenAIMessageReconstructor rebuilds OpenAI-format requests
func (*OpenAIMessageReconstructor) Reconstruct ¶
func (r *OpenAIMessageReconstructor) Reconstruct(messages []map[string]interface{}, originalBody []byte) ([]byte, error)
func (*OpenAIMessageReconstructor) VendorName ¶
func (r *OpenAIMessageReconstructor) VendorName() string
type OpenAPICache ¶
type PluginContext ¶
type PluginContext struct {
RequestID string `json:"request_id"`
LLMSlug string `json:"llm_slug"`
LLMID uint `json:"llm_id"`
AppID uint `json:"app_id"`
UserID uint `json:"user_id"`
Metadata map[string]string `json:"metadata"`
}
PluginContext provides context information for hooks
type PostAuthCallback ¶
type Proxy ¶
type Proxy struct {
// contains filtered or unexported fields
}
func New ¶
func New(gatewayService services.ServiceInterface, budgetService services.BudgetServiceInterface, cfg *Config) *Proxy
New creates a new Proxy instance using the unified services interface. This is the new interface-based constructor that supports flexible backends.
func NewProxy ¶
func NewProxy(service *services.Service, cfg *Config, budgetService services.BudgetServiceInterface) *Proxy
NewProxy creates a new Proxy instance using the existing concrete services. This is the legacy constructor that maintains backward compatibility.
func (*Proxy) AddResponseHook ¶
func (p *Proxy) AddResponseHook(hook ResponseHook)
AddResponseHook adds a response hook to the proxy (for embeddable AI Gateway)
func (*Proxy) CreateChatCompletionHandler ¶
func (p *Proxy) CreateChatCompletionHandler(w http.ResponseWriter, r *http.Request)
func (*Proxy) CreateCompletionHandler ¶
func (p *Proxy) CreateCompletionHandler(w http.ResponseWriter, r *http.Request)
Handlers
func (*Proxy) GetDatasource ¶
func (p *Proxy) GetDatasource(name string) (*models.Datasource, bool)
func (*Proxy) GetLLMByID ¶
GetLLMByID looks a loaded LLM up by its database id. Waterfall rungs are stored by id, so this is the lookup the loop needs; the slug map is what everything else uses.
func (*Proxy) GetResponseHookManager ¶
func (p *Proxy) GetResponseHookManager() ResponseHookManager
GetResponseHookManager returns the response hook manager for external configuration
func (*Proxy) Handler ¶
Handler returns the HTTP handler for the proxy, allowing it to be used with existing HTTP servers instead of starting its own server
func (*Proxy) SetAuthHooks ¶
SetAuthHooks sets authentication lifecycle hooks
func (*Proxy) SetPostAuthCallback ¶
func (p *Proxy) SetPostAuthCallback(callback PostAuthCallback)
SetPostAuthCallback is deprecated, use SetAuthHooks instead Kept for backward compatibility
func (*Proxy) SetRouteResolver ¶
func (p *Proxy) SetRouteResolver(r RouteResolver)
SetRouteResolver installs the router implementation. Called once by the host before the proxy serves traffic; nil disables routers.
func (*Proxy) Stop ¶
Stop shuts down the server Start is running. Called before Start has begun listening, it makes Start return http.ErrServerClosed instead.
func (*Proxy) UnifiedRouterBasePath ¶
UnifiedRouterBasePath reports where the OpenRouter-style unified ingress is mounted on this proxy's handler ("/v1" by default), or "" when it is disabled. Hosts that mount Handler() behind their own router use this to forward exactly the prefix the proxy claims, instead of hardcoding "/v1".
func (*Proxy) WaitForAnalytics ¶
WaitForAnalytics waits until the analysis of every response written so far has finished, or ctx is done. It runs after the response is sent, so a host shutting down calls it after draining its HTTP server and before stopping whatever the analysis records into (the analytics handler, data collection plugins such as the edge's analytics pulse).
type ResponseFormat ¶
type ResponseHook ¶
type ResponseHook interface {
// OnBeforeWriteHeaders is called before response headers are written
OnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
// OnBeforeWrite is called before response body is written (REST-only)
OnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
// OnStreamComplete is called after a streaming response finishes (streaming-only)
OnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
// GetName returns the name of this response hook
GetName() string
}
ResponseHook defines the interface that response hook implementations must satisfy
type ResponseHookManager ¶
type ResponseHookManager interface {
// ExecuteOnBeforeWriteHeaders is called before response headers are written
ExecuteOnBeforeWriteHeaders(ctx context.Context, req *HeadersRequest) (*HeadersResponse, error)
// ExecuteOnBeforeWrite is called before response body is written (REST-only)
ExecuteOnBeforeWrite(ctx context.Context, req *ResponseWriteRequest) (*ResponseWriteResponse, error)
// ExecuteOnStreamComplete is called after a streaming response finishes (streaming-only)
ExecuteOnStreamComplete(ctx context.Context, req *StreamCompleteRequest) (*StreamCompleteResponse, error)
}
ResponseHookManager defines the interface for managing response hooks
type ResponseWriteRequest ¶
type ResponseWriteRequest struct {
Body []byte `json:"body"`
Headers map[string]string `json:"headers"`
Context *PluginContext `json:"context"`
}
ResponseWriteRequest represents a request to modify response body (REST-only)
type ResponseWriteResponse ¶
type ResponseWriteResponse struct {
Modified bool `json:"modified"`
Body []byte `json:"body"`
Headers map[string]string `json:"headers"`
}
ResponseWriteResponse represents the response from body modification hooks
type RouteDecision ¶
type RouteDecision struct {
Router RouterRef
LLMID uint
Model string
// Pool is the Model Router pool that matched.
Pool string
// Route is the route that was chosen, for routers that have named routes.
Route string
// Reason says why this target was chosen, from a small fixed set
// ("model_pattern", ...), for analytics and the X-Tyk-Route-Reason header.
Reason string
// Selection is how the target was picked among the candidates (a Model
// Router pool's algorithm: "round_robin", "weighted"), for analytics.
Selection string
// SourceModel is the model the caller asked the router for; Model is what
// the chosen LLM is asked for. Set by resolveRoute.
SourceModel string
// Score is the similarity that decided a Semantic Router's embedding
// match, 0 otherwise.
Score float64
// ShadowRoute is, for a Semantic Router in shadow mode, the route the
// classifier picked while the default route served the request.
ShadowRoute string
}
RouteDecision is a resolver's answer: the LLM and model that serve the request, and why.
type RouteRequest ¶
type RouteRequest struct {
Router RouterRef
App *models.App
// Model is the model the caller asked for, with the router prefix
// already stripped by the ingress.
Model string
// Body is the OpenAI-shaped request body as the caller sent it.
Body []byte
// Header is the caller's request header (session affinity and the like).
Header http.Header
// Allow restricts the LLMs the resolver may pick. Nil allows every LLM
// the router can reach.
Allow func(llmID uint) bool
}
RouteRequest is what a resolver is asked to route.
type RouteResolver ¶
type RouteResolver interface {
// Lookup reports whether slug names a router.
Lookup(slug string) (RouterRef, bool)
// Resolve picks the LLM and model for a request. Errors are
// ErrRouteNoMatch, ErrRouteNoCandidates or ErrRouteUnavailable, possibly
// wrapped.
Resolve(ctx context.Context, req RouteRequest) (*RouteDecision, error)
// Reaches reports whether the router can send a request to the LLM.
Reaches(ref RouterRef, llmID uint) bool
// Models lists the model names the router advertises (GET /v1/models).
Models(ref RouterRef) []string
}
RouteResolver resolves router slugs. Implementations must be safe for concurrent use.
type RouterKind ¶
type RouterKind string
RouterKind names the kind of router a slug resolves to.
const ( // RouterKindModel is the Enterprise Model Router: pools matched on the // requested model name, vendors picked by round robin or weight. RouterKindModel RouterKind = "model_router" // RouterKindSemantic is the Enterprise Semantic Router: named routes // picked by classifying the prompt (keywords, embeddings, an LLM judge). RouterKindSemantic RouterKind = "semantic_router" )
type RouterRef ¶
type RouterRef struct {
Kind RouterKind
ID uint
Slug string
}
RouterRef identifies a router.
type SearchQuery ¶
type SearchResults ¶
type SearchResults struct {
Documents []DatasourceDocument `json:"documents"`
}
type StreamCompleteRequest ¶
type StreamCompleteRequest struct {
AccumulatedResponse []byte `json:"accumulated_response"` // Full SSE response (all chunks concatenated)
Headers map[string]string `json:"headers"` // Response headers from upstream
StatusCode int `json:"status_code"` // HTTP status code
Context *PluginContext `json:"context"` // Plugin context with request metadata
ChunkCount int `json:"chunk_count"` // Number of chunks received
RequestBody []byte `json:"request_body"` // Original request body (for cache key generation)
}
StreamCompleteRequest represents a request to process a completed streaming response
type StreamCompleteResponse ¶
type StreamCompleteResponse struct {
Handled bool `json:"handled"` // Plugin processed the response
Cached bool `json:"cached"` // Response was cached (for metrics/logging)
ErrorMessage string `json:"error_message"` // Error description if any
}
StreamCompleteResponse represents the response from stream complete hooks
type StreamOptions ¶
type StreamOptions struct {
IncludeUsage bool `json:"include_usage"`
}
type Tool ¶
type Tool struct {
Type string `json:"type"`
Function FunctionConfig `json:"function"`
}
func (Tool) ToLangchainTool ¶
type ValidationError ¶
type ValidationError struct {
// contains filtered or unexported fields
}
func (*ValidationError) Error ¶
func (e *ValidationError) Error() string
type VectorSearchQuery ¶
type VectorSearchQuery struct {
Embedding []float32 `json:"embedding"`
N int `json:"n"`
SimilarityThreshold float64 `json:"similarity_threshold"`
}
VectorSearchQuery is the request body for vector-based similarity search.
type VertexMessageExtractor ¶
type VertexMessageExtractor struct {
GoogleAIMessageExtractor
}
VertexMessageExtractor is an alias for GoogleAIMessageExtractor (same format)
func (*VertexMessageExtractor) VendorName ¶
func (e *VertexMessageExtractor) VendorName() string
VendorName returns "vertex"
type VertexMessageReconstructor ¶
type VertexMessageReconstructor struct {
GoogleAIMessageReconstructor
}
VertexMessageReconstructor is an alias for GoogleAIMessageReconstructor
func (*VertexMessageReconstructor) VendorName ¶
func (r *VertexMessageReconstructor) VendorName() string
Source Files
¶
- analyze_utils.go
- anthropic_bedrock_discovery.go
- anthropic_bedrock_translator.go
- anthropic_messages_models.go
- auth_hooks.go
- auth_identity.go
- bedrock_streaming.go
- bedrock_translator.go
- buffered_response_capture.go
- credential_validator.go
- credential_validator_plugin.go
- failover.go
- internal_hop.go
- internal_transport.go
- message_extractor.go
- message_reconstructor.go
- model_validator.go
- oai_error.go
- oai_error_contract.go
- proxy.go
- response_capture.go
- response_filter_utils.go
- response_hooks.go
- response_hooks_example.go
- router.go
- server_timing.go
- stream_detector.go
- stream_timing.go
- tool_filter_utils.go
- tracing.go
- translator.go
- translator_models.go
- unified_router.go
- upstream_path.go
- validators.go