Documentation
¶
Overview ¶
Package handlers provides eval type handler implementations.
Index ¶
- type A2AEvalHandler
- type A2AEvalSessionHandler
- type AgentInvokedHandler
- type AgentNotInvokedHandler
- type AgentResponseContainsHandler
- type AnswerRelevancyHandler
- type AudioDurationHandler
- type AudioEmotionHandler
- type AudioFormatHandler
- type BiasHandler
- type CompositionBranchTakenHandler
- type CompositionOutputHandler
- type CompositionParallelCompleteHandler
- type CompositionStepOutputHandler
- type ContainsAnyHandler
- type ContainsHandler
- type ContentExcludesHandler
- func (h *ContentExcludesHandler) Eval(ctx context.Context, evalCtx *evals.EvalContext, params map[string]any) (_ *evals.EvalResult, _ error)
- func (h *ContentExcludesHandler) EvalPartial(_ context.Context, content string, params map[string]any) (*evals.EvalResult, error)
- func (h *ContentExcludesHandler) Type() string
- type ContextualPrecisionHandler
- type ContextualRecallHandler
- type ContextualRelevancyHandler
- type CosineSimilarityHandler
- type CostBudgetHandler
- type DirectionalHandler
- type ExecEvalConfig
- type ExecEvalHandler
- type ExternalEvalRequest
- type FaithfulnessHandler
- type FieldPresenceHandler
- type GuardrailTriggeredHandler
- type HallucinationHandler
- type ImageDimensionsHandler
- type ImageFormatHandler
- type ImageModerationHandler
- type InvariantFieldsPreservedHandler
- type JSONPathHandler
- type JSONSchemaHandler
- type JSONValidHandler
- type JudgeOpts
- type JudgeProvider
- type JudgeResult
- type LLMJudgeHandler
- type LLMJudgeSessionHandler
- type LLMJudgeToolCallsHandler
- type LatencyBudgetHandler
- type MaxLengthHandler
- func (h *MaxLengthHandler) Eval(_ context.Context, evalCtx *evals.EvalContext, params map[string]any) (*evals.EvalResult, error)
- func (h *MaxLengthHandler) EvalPartial(_ context.Context, content string, params map[string]any) (*evals.EvalResult, error)
- func (h *MaxLengthHandler) Type() string
- func (h *MaxLengthHandler) ValidateParams(params map[string]any) error
- type MinLengthHandler
- type NoToolErrorsHandler
- type OutcomeEquivalentHandler
- type PIILeakageHandler
- type RegexHandler
- type RestEvalHandler
- type RestEvalSessionHandler
- type RoleViolationHandler
- type SentenceCountHandler
- type SkillActivatedHandler
- type SkillActivationOrderHandler
- type SkillNotActivatedHandler
- type SpecJudgeProvider
- type TextSentimentHandler
- type TextToxicityHandler
- type ToolAntiPatternHandler
- type ToolArgsExcludedSessionHandler
- type ToolArgsHandler
- type ToolArgsSessionHandler
- type ToolCallChainHandler
- type ToolCallCountHandler
- type ToolCallSequenceHandler
- type ToolCallsWithArgsHandler
- type ToolEfficiencyHandler
- type ToolExecHandler
- type ToolNoRepeatHandler
- type ToolResultHasMediaHandler
- type ToolResultIncludesHandler
- type ToolResultMatchesHandler
- type ToolResultMediaTypeHandler
- type ToolsCalledHandler
- type ToolsCalledSessionHandler
- type ToolsNotCalledHandler
- type ToolsNotCalledSessionHandler
- type TopicPolicyHandler
- type ToxicityHandler
- type VideoDurationHandler
- type VideoResolutionHandler
- type WorkflowCompleteHandler
- type WorkflowSpokeInStateHandler
- type WorkflowStateIsHandler
- type WorkflowToolAccessHandler
- type WorkflowTransitionOrderHandler
- type WorkflowTransitionedToHandler
Constants ¶
This section is empty.
Variables ¶
This section is empty.
Functions ¶
This section is empty.
Types ¶
type A2AEvalHandler ¶
type A2AEvalHandler struct{}
A2AEvalHandler evaluates a single assistant turn by sending conversation context to an A2A agent and interpreting the agent's response as a structured eval result.
Params:
- agent_url (string, required): A2A agent endpoint URL
- auth_token (string, optional): auth token, supports ${ENV_VAR}
- timeout (string, optional): request timeout, default 60s
- criteria (string, optional): evaluation criteria
- include_messages (bool, optional): include conversation history, default true
- include_tool_calls (bool, optional): include tool call records, default false
- min_score (float64, optional): minimum score threshold
- extra (map[string]any, optional): arbitrary data forwarded in request
func (*A2AEvalHandler) Eval ¶
func (h *A2AEvalHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval sends the current assistant output to the configured A2A agent.
func (*A2AEvalHandler) Type ¶
func (h *A2AEvalHandler) Type() string
Type returns the eval type identifier.
type A2AEvalSessionHandler ¶
type A2AEvalSessionHandler struct{}
A2AEvalSessionHandler evaluates an entire conversation by sending all assistant messages to an A2A agent.
Params: same as A2AEvalHandler.
func (*A2AEvalSessionHandler) Eval ¶
func (h *A2AEvalSessionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval sends all assistant messages to the configured A2A agent.
func (*A2AEvalSessionHandler) Type ¶
func (h *A2AEvalSessionHandler) Type() string
Type returns the eval type identifier.
type AgentInvokedHandler ¶
type AgentInvokedHandler struct{}
AgentInvokedHandler checks that expected agents were invoked as tool calls. Params: agents []string — list of agent names that should have been called.
func (*AgentInvokedHandler) Eval ¶
func (h *AgentInvokedHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks if expected agents were invoked.
func (*AgentInvokedHandler) Type ¶
func (h *AgentInvokedHandler) Type() string
Type returns the eval type identifier.
type AgentNotInvokedHandler ¶
type AgentNotInvokedHandler struct{}
AgentNotInvokedHandler checks that forbidden agents were NOT called. Params: agents []string — agent names that should not have been called.
func (*AgentNotInvokedHandler) Eval ¶
func (h *AgentNotInvokedHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks if any forbidden agents were invoked.
func (*AgentNotInvokedHandler) Type ¶
func (h *AgentNotInvokedHandler) Type() string
Type returns the eval type identifier.
type AgentResponseContainsHandler ¶
type AgentResponseContainsHandler struct{}
AgentResponseContainsHandler checks that a specific agent's response contains expected text. Agent responses appear as tool-result messages where the tool name matches the agent name. Params: agent string, contains string.
func (*AgentResponseContainsHandler) Eval ¶
func (h *AgentResponseContainsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks if the specified agent's tool result contains the expected text.
func (*AgentResponseContainsHandler) Type ¶
func (h *AgentResponseContainsHandler) Type() string
Type returns the eval type identifier.
type AnswerRelevancyHandler ¶
type AnswerRelevancyHandler struct{}
AnswerRelevancyHandler scores how directly the assistant's answer addresses the user's question. Equivalent in name to DeepEval / Ragas `answer_relevancy`.
Default prompts adapted from the public DeepEval / Ragas reference implementations (Apache 2.0). Override per-call by passing system_prompt or criteria in params.
Params (all optional):
- question (string): override the auto-extracted last user turn
- rubric, model, system_prompt: standard llm_judge knobs
func (*AnswerRelevancyHandler) Eval ¶
func (h *AnswerRelevancyHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output against the question for on-topic focus and directness.
func (*AnswerRelevancyHandler) Type ¶
func (h *AnswerRelevancyHandler) Type() string
Type returns the eval type identifier.
type AudioDurationHandler ¶
type AudioDurationHandler struct{}
AudioDurationHandler checks that audio duration is within range. Params: min_seconds float64, max_seconds float64.
func (*AudioDurationHandler) Eval ¶
func (h *AudioDurationHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks audio duration constraints.
func (*AudioDurationHandler) Type ¶
func (h *AudioDurationHandler) Type() string
Type returns the eval type identifier.
type AudioEmotionHandler ¶
type AudioEmotionHandler struct{}
AudioEmotionHandler is a pure eval primitive: it calls the AudioClassifier resolved from the orchestrator's classify registry, picks the score for the chosen expected_label, and emits it as EvalResult.Score. Threshold judgment (min_score / max_score) lives on `type: assertion` wrappers — NOT on this handler.
Wrap with `type: assertion` to assert against a threshold:
- type: assertion params: eval_type: audio_emotion eval_params: { model: "...", expected_label: "ang", message_role: user } min_score: 0.5
Use directly in pack `evals:` to emit the raw signal at runtime (for metrics / observability).
Params:
- model string (required) — backend model id, e.g. "superb/wav2vec2-base-superb-er"
- expected_label string (required) — label whose score is emitted
- message_role string (optional, default "user") — which speaker's audio to score
- message_index int (optional, default -1 = latest match) — pick a specific audio message
- classifier_id string (optional) — explicit registry id; empty uses the configured default
Putting min_score / max_score on this handler is rejected — the assertion wrapper is the canonical home for thresholds.
func (*AudioEmotionHandler) Eval ¶
func (h *AudioEmotionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval pulls the AudioClassifier out of context, locates the target audio part in the conversation, runs classification, and grades the requested label against the configured threshold.
Skipped vs Error distinction:
- Skipped — the assertion couldn't run because preconditions weren't met. Used for infrastructure absence: no classify registry configured (e.g. CI runs without HF_TOKEN), no audio parts in the message log (e.g. a mock provider that doesn't emit audio), or HF returned ErrModelLoading. These are configuration / environment shapes, not assertion failures.
- Error — the assertion is misconfigured at the call site (missing required param) or the classifier call failed at runtime. The user should fix something.
The split lets a single arena config sit happily under both real-provider runs (assertion exercises a real model) and keyless CI runs (assertion skips cleanly) without needing per-environment scenario forks.
func (*AudioEmotionHandler) Type ¶
func (h *AudioEmotionHandler) Type() string
Type returns the eval type identifier.
type AudioFormatHandler ¶
type AudioFormatHandler struct{}
AudioFormatHandler checks that audio content has allowed formats. Params: formats []string.
func (*AudioFormatHandler) Eval ¶
func (h *AudioFormatHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks audio formats against the allowed list.
func (*AudioFormatHandler) Type ¶
func (h *AudioFormatHandler) Type() string
Type returns the eval type identifier.
type BiasHandler ¶
type BiasHandler struct{}
BiasHandler scores assistant output for demographic / stereotype bias via the LLM judge. DeepEval-equivalent name; default-wired as a guardrail. Params: standard llm_judge knobs.
func (*BiasHandler) Eval ¶
func (h *BiasHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output for demographic / stereotype bias.
func (*BiasHandler) Type ¶
func (h *BiasHandler) Type() string
Type returns the eval type identifier.
type CompositionBranchTakenHandler ¶
type CompositionBranchTakenHandler struct{}
CompositionBranchTakenHandler asserts a branch step took the expected target. Params: branch (string, required), expected (string, required).
func (*CompositionBranchTakenHandler) Eval ¶
func (h *CompositionBranchTakenHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks the recorded branch target equals expected.
func (*CompositionBranchTakenHandler) Type ¶
func (h *CompositionBranchTakenHandler) Type() string
Type returns the eval type identifier.
type CompositionOutputHandler ¶
type CompositionOutputHandler struct{}
CompositionOutputHandler asserts the composition's final output contains/equals a value. Params: one of contains|equals (string).
func (*CompositionOutputHandler) Eval ¶
func (h *CompositionOutputHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval matches the composition's final output (CurrentOutput) against contains/equals.
func (*CompositionOutputHandler) Type ¶
func (h *CompositionOutputHandler) Type() string
Type returns the eval type identifier.
type CompositionParallelCompleteHandler ¶
type CompositionParallelCompleteHandler struct{}
CompositionParallelCompleteHandler asserts a parallel step completed. Params: parallel (string, required).
func (*CompositionParallelCompleteHandler) Eval ¶
func (h *CompositionParallelCompleteHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks the recorded parallel status is "complete".
func (*CompositionParallelCompleteHandler) Type ¶
func (h *CompositionParallelCompleteHandler) Type() string
Type returns the eval type identifier.
type CompositionStepOutputHandler ¶
type CompositionStepOutputHandler struct{}
CompositionStepOutputHandler asserts a named step's output contains/equals a value. Params: step (string, required); one of contains|equals (string).
func (*CompositionStepOutputHandler) Eval ¶
func (h *CompositionStepOutputHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval matches the named step's recorded output against contains/equals.
func (*CompositionStepOutputHandler) Type ¶
func (h *CompositionStepOutputHandler) Type() string
Type returns the eval type identifier.
type ContainsAnyHandler ¶
type ContainsAnyHandler struct{}
ContainsAnyHandler checks that at least one assistant message contains at least one of the specified patterns. Params: patterns []string (case-insensitive matching).
func (*ContainsAnyHandler) Eval ¶
func (h *ContainsAnyHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (_ *evals.EvalResult, _ error)
Eval checks assistant messages for any matching pattern.
func (*ContainsAnyHandler) Type ¶
func (h *ContainsAnyHandler) Type() string
Type returns the eval type identifier.
type ContainsHandler ¶
type ContainsHandler struct{}
ContainsHandler checks if CurrentOutput contains all specified patterns (case-insensitive). Params: patterns []string.
func (*ContainsHandler) Eval ¶
func (h *ContainsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that all patterns appear in the current output.
func (*ContainsHandler) Type ¶
func (h *ContainsHandler) Type() string
Type returns the eval type identifier.
type ContentExcludesHandler ¶
type ContentExcludesHandler struct{}
ContentExcludesHandler checks that NONE of the assistant messages across the full conversation contain any of the forbidden patterns. Params: patterns []string (case-insensitive matching).
func (*ContentExcludesHandler) Eval ¶
func (h *ContentExcludesHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (_ *evals.EvalResult, _ error)
Eval checks all assistant messages for forbidden patterns.
func (*ContentExcludesHandler) EvalPartial ¶
func (h *ContentExcludesHandler) EvalPartial( _ context.Context, content string, params map[string]any, ) (*evals.EvalResult, error)
EvalPartial checks partial streaming content for forbidden patterns. Always uses substring mode to avoid false negatives on partial words that would occur with word_boundary mode mid-stream.
func (*ContentExcludesHandler) Type ¶
func (h *ContentExcludesHandler) Type() string
Type returns the eval type identifier.
type ContextualPrecisionHandler ¶
type ContextualPrecisionHandler struct{}
ContextualPrecisionHandler scores the fraction of retrieved chunks that are actually relevant to the question. Equivalent in name to DeepEval `contextual_precision`.
Default prompts adapted from the public DeepEval reference implementation (Apache 2.0). Override per-call by passing system_prompt or criteria in params.
Params (all optional):
- contexts ([]string) | context (string) | context_field (string): retrieved chunks
- question (string): override the auto-extracted last user turn
- rubric, model, system_prompt: standard llm_judge knobs
func (*ContextualPrecisionHandler) Eval ¶
func (h *ContextualPrecisionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the precision of retrieved chunks relative to the question.
func (*ContextualPrecisionHandler) Type ¶
func (h *ContextualPrecisionHandler) Type() string
Type returns the eval type identifier.
type ContextualRecallHandler ¶
type ContextualRecallHandler struct{}
ContextualRecallHandler scores how completely the retrieved chunks cover the information needed for the ground-truth answer. Equivalent in name to DeepEval / Ragas `contextual_recall`.
Default prompts adapted from the public DeepEval / Ragas reference implementations (Apache 2.0). Override per-call by passing system_prompt or criteria in params.
Params:
- contexts ([]string) | context (string) | context_field (string): retrieved chunks (required)
- reference (string) | expected_output (string): the ground-truth answer (required)
- question (string): override the auto-extracted last user turn
- rubric, model, system_prompt: standard llm_judge knobs
func (*ContextualRecallHandler) Eval ¶
func (h *ContextualRecallHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores how completely the retrieved chunks cover the information the ground-truth answer relies on.
func (*ContextualRecallHandler) Type ¶
func (h *ContextualRecallHandler) Type() string
Type returns the eval type identifier.
type ContextualRelevancyHandler ¶
type ContextualRelevancyHandler struct{}
ContextualRelevancyHandler scores the average per-chunk relevance of the retrieved chunks to the question. Equivalent in name to DeepEval `contextual_relevancy`.
Distinct from `contextual_precision`: precision is a binary relevant/not-relevant ratio; relevancy is the mean of graded scores.
Default prompts adapted from the public DeepEval reference implementation (Apache 2.0). Override per-call by passing system_prompt or criteria in params.
Params (all optional):
- contexts ([]string) | context (string) | context_field (string)
- question (string)
- rubric, model, system_prompt
func (*ContextualRelevancyHandler) Eval ¶
func (h *ContextualRelevancyHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the mean relevance of retrieved chunks to the question.
func (*ContextualRelevancyHandler) Type ¶
func (h *ContextualRelevancyHandler) Type() string
Type returns the eval type identifier.
type CosineSimilarityHandler ¶
type CosineSimilarityHandler struct{}
CosineSimilarityHandler computes cosine similarity between embeddings. Params: reference []float64, min_similarity float64. Target embedding comes from Metadata["embedding"].
func (*CosineSimilarityHandler) Eval ¶
func (h *CosineSimilarityHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval computes cosine similarity and checks against threshold.
func (*CosineSimilarityHandler) Type ¶
func (h *CosineSimilarityHandler) Type() string
Type returns the eval type identifier.
type CostBudgetHandler ¶
type CostBudgetHandler struct{}
CostBudgetHandler checks conversation-level cost and token limits. Params:
- max_cost_usd: float64 — maximum total cost in USD (optional)
- max_input_tokens: int — maximum input tokens (optional)
- max_output_tokens: int — maximum output tokens (optional)
- max_total_tokens: int — maximum total tokens (optional)
func (*CostBudgetHandler) Eval ¶
func (h *CostBudgetHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks cost and token limits.
func (*CostBudgetHandler) Type ¶
func (h *CostBudgetHandler) Type() string
Type returns the eval type identifier.
type DirectionalHandler ¶
type DirectionalHandler struct{}
DirectionalHandler compares a run's output against a baseline expectation. Used in behavioral testing (Phase 6) to verify perturbation-invariant behavior. Params:
- check (string, required): "same_tool_calls", "same_outcome", or "similar_content"
- baseline_tools ([]string, optional): expected tool names for same_tool_calls
- baseline_state (string, optional): expected workflow state for same_outcome
- baseline_content (string, optional): expected content substring for similar_content
- threshold (float64, optional): minimum overlap ratio for similar_content (default 0.5)
func (*DirectionalHandler) Eval ¶
func (h *DirectionalHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the run's output matches the baseline for the given check type.
func (*DirectionalHandler) Type ¶
func (h *DirectionalHandler) Type() string
Type returns the eval type identifier.
type ExecEvalConfig ¶
type ExecEvalConfig struct {
TypeName string
Command string
Args []string
Env []string
TimeoutMs int
}
ExecEvalConfig holds the configuration for creating an ExecEvalHandler.
type ExecEvalHandler ¶
type ExecEvalHandler struct {
// contains filtered or unexported fields
}
ExecEvalHandler evaluates content by spawning an external subprocess. The subprocess receives an ExecEvalRequest as JSON on stdin and must return an ExecEvalResponse as JSON on stdout.
ExecEvalHandler is registered dynamically from RuntimeConfig exec bindings, not via init(). Each binding creates a handler whose Type() matches the eval type name in the pack — making exec evals transparent to the pack.
func NewExecEvalHandler ¶
func NewExecEvalHandler(cfg *ExecEvalConfig) *ExecEvalHandler
NewExecEvalHandler creates a new ExecEvalHandler from a config. Exec handlers are automatically classified as long-running and external for well-known group filtering.
func (*ExecEvalHandler) Eval ¶
func (h *ExecEvalHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval runs the external eval subprocess and returns the result.
func (*ExecEvalHandler) Type ¶
func (h *ExecEvalHandler) Type() string
Type returns the eval type identifier (matches the pack eval type name).
type ExternalEvalRequest ¶
type ExternalEvalRequest struct {
CurrentOutput string `json:"current_output"`
Messages []messageView `json:"messages,omitempty"`
ToolCalls []toolCallView `json:"tool_calls,omitempty"`
Criteria string `json:"criteria,omitempty"`
Variables map[string]any `json:"variables,omitempty"`
Extra map[string]any `json:"extra,omitempty"`
}
ExternalEvalRequest is the standard request body sent to external eval endpoints (REST) and formatted as context for A2A eval agents.
type FaithfulnessHandler ¶
type FaithfulnessHandler struct{}
FaithfulnessHandler scores how well the answer is supported by the provided context. A faithful answer makes only claims backed by the context; an unfaithful answer asserts facts the context does not support. Equivalent in name to DeepEval / Ragas `faithfulness`.
Default prompts adapted from the public DeepEval / Ragas reference implementations (Apache 2.0). Override per-call by passing system_prompt or criteria in params.
Params (all optional):
- contexts ([]string) | context (string) | context_field (string): retrieved chunks the answer should be grounded in
- rubric, model, system_prompt: standard llm_judge knobs
func (*FaithfulnessHandler) Eval ¶
func (h *FaithfulnessHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output against the supplied context for factual consistency.
func (*FaithfulnessHandler) Type ¶
func (h *FaithfulnessHandler) Type() string
Type returns the eval type identifier.
type FieldPresenceHandler ¶
type FieldPresenceHandler struct{}
FieldPresenceHandler checks that required fields are present in the output. Params: fields []string (field names to look for, case-insensitive).
func (*FieldPresenceHandler) Eval ¶
func (h *FieldPresenceHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks if each field name appears in CurrentOutput (case-insensitive).
func (*FieldPresenceHandler) Type ¶
func (h *FieldPresenceHandler) Type() string
Type returns the eval type identifier.
type GuardrailTriggeredHandler ¶
type GuardrailTriggeredHandler struct{}
GuardrailTriggeredHandler checks if a specific eval or guardrail triggered (or didn't trigger) as expected.
It searches EvalContext.PriorResults for an eval whose Type or EvalID matches the validator_type parameter. Pipeline-level guardrail results from message.Validations are automatically seeded into PriorResults by BuildEvalContext, so all guardrail outcomes are available through a single lookup path.
Params: validator_type string, should_trigger bool (default true), direction string (optional).
direction narrows the search to firings the guardrail recorder stamped with that side: "input" (judged the user's message, before the call) or "output" (judged the assistant response). Omitted — or "both" — matches a firing in either direction, which is the historical behavior and stays byte-identical. Set it when an input and an output guardrail of the same eval type are both declared: without it only the last recorded firing is reachable, so neither can be targeted (#1718).
func (*GuardrailTriggeredHandler) Eval ¶
func (h *GuardrailTriggeredHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks PriorResults for a matching guardrail or eval outcome.
func (*GuardrailTriggeredHandler) Type ¶
func (h *GuardrailTriggeredHandler) Type() string
Type returns the eval type identifier.
func (*GuardrailTriggeredHandler) ValidateParams ¶
func (h *GuardrailTriggeredHandler) ValidateParams(params map[string]any) error
ValidateParams checks that the required 'validator_type' (or alias 'validator') param is set to a non-empty string, and that 'direction' — if present — names a side the guardrail recorder can actually stamp.
A bad direction is an error here rather than a warning-and-fallback (which is what the guardrail factory does with the same param) because the failure modes are opposite: dropping a guardrail leaves a conversation unprotected, whereas an assertion with a misspelled direction silently matches nothing and reports "the guardrail never fired" — a green test that checks nothing.
type HallucinationHandler ¶
type HallucinationHandler struct{}
HallucinationHandler scores how free the answer is of claims that are not supported by, or contradict, the provided context. The inverse framing of `faithfulness` — kept as a separate handler so users coming from DeepEval find the vocabulary they expect.
A score of 1.0 means no hallucination (matches faithfulness=1.0); a score of 0.0 means entirely hallucinated.
Default prompts adapted from the public DeepEval reference implementation (Apache 2.0). Override per-call by passing system_prompt or criteria in params.
Params (all optional):
- contexts ([]string) | context (string) | context_field (string)
- rubric, model, system_prompt: standard llm_judge knobs
func (*HallucinationHandler) Eval ¶
func (h *HallucinationHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output for hallucinations relative to the supplied context.
func (*HallucinationHandler) Type ¶
func (h *HallucinationHandler) Type() string
Type returns the eval type identifier.
type ImageDimensionsHandler ¶
type ImageDimensionsHandler struct{}
ImageDimensionsHandler checks that images meet dimension requirements. Params: min_width, max_width, min_height, max_height, width, height.
func (*ImageDimensionsHandler) Eval ¶
func (h *ImageDimensionsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks image dimensions against constraints.
func (*ImageDimensionsHandler) Type ¶
func (h *ImageDimensionsHandler) Type() string
Type returns the eval type identifier.
type ImageFormatHandler ¶
type ImageFormatHandler struct{}
ImageFormatHandler checks that images in assistant messages have allowed formats. Params: formats []string.
func (*ImageFormatHandler) Eval ¶
func (h *ImageFormatHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks image formats against the allowed list.
func (*ImageFormatHandler) Type ¶
func (h *ImageFormatHandler) Type() string
Type returns the eval type identifier.
type ImageModerationHandler ¶
type ImageModerationHandler struct{}
ImageModerationHandler is a pure eval primitive: it runs the latest image in the target role's turns through the configured ImageClassifier and emits the score for expected_label (e.g. "nsfw") as EvalResult.Score. Threshold judgment (min_score / max_score) lives on `type: assertion` / `type: guardrail` wrappers — NOT on this handler.
Params:
- model string (required) — classifier model id, e.g. "Falconsai/nsfw_image_detection"
- expected_label string (required) — label whose score is emitted
- message_role string (optional, default "assistant") — whose image to score
- message_index int (optional, default -1 = latest match)
- classifier_id string (optional) — explicit registry id; empty uses the configured default
func (*ImageModerationHandler) Eval ¶
func (h *ImageModerationHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval resolves the ImageClassifier from context, locates the target image part, classifies it, and emits the requested label's score. Skipped vs Error split mirrors audio_emotion: infrastructure absence (no registry, no image, model loading/unsupported) is Skipped; misconfiguration or a runtime classifier failure is Error.
func (*ImageModerationHandler) Type ¶
func (h *ImageModerationHandler) Type() string
Type returns the eval type identifier.
type InvariantFieldsPreservedHandler ¶
type InvariantFieldsPreservedHandler struct{}
InvariantFieldsPreservedHandler checks that field values in tool call arguments are not lost between calls to the same tool. If a field was present in an earlier call but disappears in a later call, that is a violation.
Params:
- tool (string, required): The tool name to track across calls.
- fields ([]string, required): JSON field names to track in tool call arguments.
func (*InvariantFieldsPreservedHandler) Eval ¶
func (h *InvariantFieldsPreservedHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that tracked fields are not lost between calls to the same tool.
func (*InvariantFieldsPreservedHandler) Type ¶
func (h *InvariantFieldsPreservedHandler) Type() string
Type returns the eval type identifier.
type JSONPathHandler ¶
type JSONPathHandler struct{}
JSONPathHandler validates assistant output as JSON using JMESPath expressions. Params:
- expression string (JMESPath expression)
- expected any (optional: exact match)
- contains []any (optional: array contains check)
- min_results int, max_results int (optional: array length bounds)
- min float64, max float64 (optional: numeric range)
- allow_wrapped bool (optional: extract JSON from code blocks)
- extract_json bool (optional: extract JSON from mixed text)
func (*JSONPathHandler) Eval ¶
func (h *JSONPathHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval executes a JMESPath expression on the assistant output and validates the result.
func (*JSONPathHandler) Type ¶
func (h *JSONPathHandler) Type() string
Type returns the eval type identifier.
type JSONSchemaHandler ¶
type JSONSchemaHandler struct{}
JSONSchemaHandler validates CurrentOutput against a JSON schema. Params: schema map[string]any.
func (*JSONSchemaHandler) Eval ¶
func (h *JSONSchemaHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval validates the current output against the provided JSON schema.
func (*JSONSchemaHandler) Type ¶
func (h *JSONSchemaHandler) Type() string
Type returns the eval type identifier.
type JSONValidHandler ¶
type JSONValidHandler struct{}
JSONValidHandler checks if CurrentOutput is valid JSON. Params: allow_wrapped bool, extract_json bool (both optional).
func (*JSONValidHandler) Eval ¶
func (h *JSONValidHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that the current output is parseable JSON.
func (*JSONValidHandler) Type ¶
func (h *JSONValidHandler) Type() string
Type returns the eval type identifier.
type JudgeOpts ¶
type JudgeOpts struct {
// Content is the text being evaluated (assistant response or full conversation).
Content string
// Criteria describes what the judge should evaluate (e.g. "Is the response helpful?").
Criteria string
// Rubric provides detailed scoring guidance (optional).
Rubric string
// Model specifies which model to use for judging (optional, provider decides default).
Model string
// SystemPrompt overrides the default judge system prompt (optional).
SystemPrompt string
// Extra holds additional parameters for provider-specific features.
Extra map[string]any
// Emitter is an optional event emitter for provider call telemetry.
Emitter *events.Emitter
}
JudgeOpts configures a judge evaluation request.
type JudgeProvider ¶
type JudgeProvider interface {
// Judge sends the evaluation prompt to an LLM and returns
// the parsed verdict. Implementations handle provider selection,
// prompt formatting, and response parsing.
Judge(ctx context.Context, opts JudgeOpts) (*JudgeResult, error)
}
JudgeProvider abstracts LLM access for judge-based evaluations. Arena, SDK, and eval workers each provide their own implementation wiring their respective provider infrastructure.
type JudgeResult ¶
type JudgeResult struct {
// Passed indicates whether the content met the evaluation criteria.
Passed bool
// Score is the numerical score assigned by the judge (typically 0.0-1.0).
Score float64
// Reasoning explains the judge's evaluation.
Reasoning string
// Raw is the unprocessed LLM response text.
Raw string
}
JudgeResult captures the output of an LLM judge evaluation.
type LLMJudgeHandler ¶
type LLMJudgeHandler struct{}
LLMJudgeHandler evaluates a single assistant turn using an LLM judge. Pure eval primitive (see docs/reference/checks.md for the `type: assertion` wrapper pattern that adds thresholds). The JudgeProvider must be supplied in evalCtx.Metadata["judge_provider"].
Params:
- criteria (string, required): what to evaluate
- rubric (string, optional): detailed scoring guidance
- model (string, optional): model override for the judge
- system_prompt (string, optional): override default system prompt
func (*LLMJudgeHandler) Eval ¶
func (h *LLMJudgeHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval runs the LLM judge on the current assistant output.
func (*LLMJudgeHandler) Type ¶
func (h *LLMJudgeHandler) Type() string
Type returns the eval type identifier.
type LLMJudgeSessionHandler ¶
type LLMJudgeSessionHandler struct{}
LLMJudgeSessionHandler is the session-level counterpart of LLMJudgeHandler. It runs the judge once over a full, role-labeled transcript of the conversation — user (and other) turns, assistant text, and every tool call with its arguments and result — so the judge sees what the agent actually did, not only its prose. Same params, same conventions. Registered under `llm_judge_conversation` too (see register.go).
func (*LLMJudgeSessionHandler) Eval ¶
func (h *LLMJudgeSessionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval runs the LLM judge over the full session transcript.
func (*LLMJudgeSessionHandler) Type ¶
func (h *LLMJudgeSessionHandler) Type() string
Type returns the eval type identifier.
type LLMJudgeToolCallsHandler ¶
type LLMJudgeToolCallsHandler struct{}
LLMJudgeToolCallsHandler is the tool-call counterpart of LLMJudgeHandler — feeds tool call data (names, args, results) instead of the assistant's text. Accepts the same base params plus `tools []string` to filter to specific tool names.
func (*LLMJudgeToolCallsHandler) Eval ¶
func (h *LLMJudgeToolCallsHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval runs the LLM judge on formatted tool call data.
func (*LLMJudgeToolCallsHandler) Type ¶
func (h *LLMJudgeToolCallsHandler) Type() string
Type returns the eval type identifier.
type LatencyBudgetHandler ¶
type LatencyBudgetHandler struct{}
LatencyBudgetHandler checks Metadata["latency_ms"] against a max. Params: max_ms float64.
func (*LatencyBudgetHandler) Eval ¶
func (h *LatencyBudgetHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that the latency is within budget.
func (*LatencyBudgetHandler) Type ¶
func (h *LatencyBudgetHandler) Type() string
Type returns the eval type identifier.
type MaxLengthHandler ¶
type MaxLengthHandler struct{}
MaxLengthHandler checks that CurrentOutput does not exceed the specified character count. Accepts params: max or max_characters (int).
func (*MaxLengthHandler) Eval ¶
func (h *MaxLengthHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the output does not exceed the maximum length.
func (*MaxLengthHandler) EvalPartial ¶
func (h *MaxLengthHandler) EvalPartial( _ context.Context, content string, params map[string]any, ) (*evals.EvalResult, error)
EvalPartial checks partial streaming content against max length limits. This enables early abort when streaming content exceeds the limit.
func (*MaxLengthHandler) Type ¶
func (h *MaxLengthHandler) Type() string
Type returns the eval type identifier.
func (*MaxLengthHandler) ValidateParams ¶
func (h *MaxLengthHandler) ValidateParams(params map[string]any) error
ValidateParams checks that at least one of the canonical or aliased max-length keys is set to a positive integer. Called at guardrail hook construction and at eval preflight so invalid pack validators fail loudly at load time instead of at request time.
type MinLengthHandler ¶
type MinLengthHandler struct{}
MinLengthHandler checks that CurrentOutput has at least the specified character count. Accepts params: min or min_characters (int).
func (*MinLengthHandler) Eval ¶
func (h *MinLengthHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the output meets the minimum length requirement.
func (*MinLengthHandler) Type ¶
func (h *MinLengthHandler) Type() string
Type returns the eval type identifier.
func (*MinLengthHandler) ValidateParams ¶
func (h *MinLengthHandler) ValidateParams(params map[string]any) error
ValidateParams checks that at least one of the canonical or aliased min-length keys is set to a positive integer.
type NoToolErrorsHandler ¶
type NoToolErrorsHandler struct{}
NoToolErrorsHandler checks that no tool calls returned errors. Params: tools []string (optional) — if set, only checks calls matching those tool names.
func (*NoToolErrorsHandler) Eval ¶
func (h *NoToolErrorsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks for tool errors in the eval context's tool calls.
func (*NoToolErrorsHandler) Type ¶
func (h *NoToolErrorsHandler) Type() string
Type returns the eval type identifier.
type OutcomeEquivalentHandler ¶
type OutcomeEquivalentHandler struct{}
OutcomeEquivalentHandler checks that a single run's outcome matches an expected value. Used in behavioral testing (Phase 6) to verify perturbation-invariant outcomes. Params:
- metric (string, required): "tool_calls", "final_state", or "content_hash"
- expected_tools ([]string, optional): for tool_calls metric
- expected_state (string, optional): for final_state metric
- expected_content (string, optional): for content_hash metric
func (*OutcomeEquivalentHandler) Eval ¶
func (h *OutcomeEquivalentHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the run's outcome matches the expected value for the given metric.
func (*OutcomeEquivalentHandler) Type ¶
func (h *OutcomeEquivalentHandler) Type() string
Type returns the eval type identifier.
type PIILeakageHandler ¶
type PIILeakageHandler struct{}
PIILeakageHandler scores whether the assistant output leaks personally-identifiable information. Equivalent in name to DeepEval `pii_leakage`. Default wiring in this codebase is as a guardrail (pack `validators:` block); scenarios observe firing via `guardrail_triggered`. The runtime guardrail enforces (blocks / replaces) the offending content; the assertion observes the firing.
Implementation runs a regex pre-pass for high-confidence patterns (emails, US SSN, 16-digit card-shape numbers) before the LLM-judged path. On regex match, the handler returns score 0 immediately without an LLM call — keeps the obvious cases cheap and deterministic. On miss, the LLM judge inspects the answer for ambiguous PII (names tied to other PII, less-strict patterns, etc.).
Default prompts adapted from the public DeepEval reference implementation (Apache 2.0).
Params (all optional):
- rubric, model, system_prompt, criteria: standard llm_judge knobs
func (*PIILeakageHandler) Eval ¶
func (h *PIILeakageHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output for PII leakage. Runs the regex pre-pass first; if no high-confidence pattern matched, falls through to llm_judge if a judge provider is configured. If no judge is available, returns a pass (score 1.0) — the regex pre-pass is the deterministic baseline; the LLM judge is an optional second layer for ambiguous patterns. The combination must not "fail closed" when the optional layer isn't wired, or wiring pii_leakage as a guardrail without an LLM key would block every output.
func (*PIILeakageHandler) Type ¶
func (h *PIILeakageHandler) Type() string
Type returns the eval type identifier.
type RegexHandler ¶
type RegexHandler struct{}
RegexHandler checks if CurrentOutput matches a regex pattern. Params: pattern string.
func (*RegexHandler) Eval ¶
func (h *RegexHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that the current output matches the regex pattern.
func (*RegexHandler) Type ¶
func (h *RegexHandler) Type() string
Type returns the eval type identifier.
type RestEvalHandler ¶
type RestEvalHandler struct{}
RestEvalHandler evaluates a single assistant turn by POSTing conversation context to an external HTTP endpoint and interpreting the structured JSON response.
Params:
- url (string, required): endpoint URL
- method (string, optional): HTTP method, default POST
- headers (map[string]string, optional): request headers, supports ${ENV_VAR}
- timeout (string, optional): request timeout, default 30s
- include_messages (bool, optional): include conversation history, default true
- include_tool_calls (bool, optional): include tool call records, default false
- criteria (string, optional): evaluation criteria forwarded in request
- min_score (float64, optional): minimum score threshold
- extra (map[string]any, optional): arbitrary data forwarded in request
func (*RestEvalHandler) Eval ¶
func (h *RestEvalHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval sends the current assistant output to the configured REST endpoint.
func (*RestEvalHandler) Type ¶
func (h *RestEvalHandler) Type() string
Type returns the eval type identifier.
type RestEvalSessionHandler ¶
type RestEvalSessionHandler struct{}
RestEvalSessionHandler evaluates an entire conversation by POSTing all assistant messages to an external HTTP endpoint.
Params: same as RestEvalHandler.
func (*RestEvalSessionHandler) Eval ¶
func (h *RestEvalSessionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval sends all assistant messages to the configured REST endpoint.
func (*RestEvalSessionHandler) Type ¶
func (h *RestEvalSessionHandler) Type() string
Type returns the eval type identifier.
type RoleViolationHandler ¶
type RoleViolationHandler struct{}
RoleViolationHandler scores whether the assistant output breaks its assigned role, persona, or instruction set. Equivalent in name to DeepEval `role_violation`. Default wiring is as a guardrail; scenarios observe firing via `guardrail_triggered`.
The handler injects the active system prompt (if available via evalCtx.Metadata["system_prompt"]) into the judge prompt so the judge can decide whether the answer deviates from it. If no system prompt is supplied via metadata or params, the judge falls back to generic role-consistency scoring.
Default prompts adapted from the public DeepEval reference implementation (Apache 2.0).
Params (all optional):
- system_prompt (string): the role / persona the answer should adhere to; overrides metadata. Distinct from the standard llm_judge `system_prompt` which controls the JUDGE's prompt.
- rubric, model, criteria: standard llm_judge knobs
func (*RoleViolationHandler) Eval ¶
func (h *RoleViolationHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output for adherence to its assigned role.
func (*RoleViolationHandler) Type ¶
func (h *RoleViolationHandler) Type() string
Type returns the eval type identifier.
type SentenceCountHandler ¶
type SentenceCountHandler struct{}
SentenceCountHandler counts sentences in the current output. This is a pure measurement handler — it always passes. Params: (none required — returns count as measurement)
func (*SentenceCountHandler) Eval ¶
func (h *SentenceCountHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, _ map[string]any, ) (*evals.EvalResult, error)
Eval counts sentences in the current output.
func (*SentenceCountHandler) Type ¶
func (h *SentenceCountHandler) Type() string
Type returns the eval type identifier.
type SkillActivatedHandler ¶
type SkillActivatedHandler struct{}
SkillActivatedHandler checks that specific skills were activated. Scans evalCtx.ToolCalls for "skill__activate" calls and extracts the "name" argument. Params: skill_names []string, min_calls int (optional, default 1).
func (*SkillActivatedHandler) Eval ¶
func (h *SkillActivatedHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that required skills were activated at least the minimum number of times.
func (*SkillActivatedHandler) Type ¶
func (h *SkillActivatedHandler) Type() string
Type returns the eval type identifier.
type SkillActivationOrderHandler ¶
type SkillActivationOrderHandler struct{}
SkillActivationOrderHandler checks that skills were activated in a specified subsequence order. Params: sequence []string — the expected skill names in order.
func (*SkillActivationOrderHandler) Eval ¶
func (h *SkillActivationOrderHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the expected sequence appears as a subsequence of the actual skill activations.
func (*SkillActivationOrderHandler) Type ¶
func (h *SkillActivationOrderHandler) Type() string
Type returns the eval type identifier.
type SkillNotActivatedHandler ¶
type SkillNotActivatedHandler struct{}
SkillNotActivatedHandler checks that specific skills were NOT activated. Params: skill_names []string.
func (*SkillNotActivatedHandler) Eval ¶
func (h *SkillNotActivatedHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval ensures forbidden skills were never activated.
func (*SkillNotActivatedHandler) Type ¶
func (h *SkillNotActivatedHandler) Type() string
Type returns the eval type identifier.
type SpecJudgeProvider ¶
type SpecJudgeProvider struct {
// contains filtered or unexported fields
}
SpecJudgeProvider implements JudgeProvider by creating a provider from a ProviderSpec. This is the standard implementation used by Arena and any caller that has judge targets as ProviderSpecs.
func NewSpecJudgeProvider ¶
func NewSpecJudgeProvider(spec *providers.ProviderSpec) *SpecJudgeProvider
NewSpecJudgeProvider creates a JudgeProvider from a provider spec.
func (*SpecJudgeProvider) Judge ¶
func (sp *SpecJudgeProvider) Judge(ctx context.Context, opts JudgeOpts) (*JudgeResult, error)
Judge creates a provider from the spec, sends the evaluation prompt, and parses the verdict. Emits ProviderCallStarted/Completed/Failed events if an emitter is set on opts.
type TextSentimentHandler ¶
type TextSentimentHandler struct{}
TextSentimentHandler is a pure eval primitive: it scores text against a sentiment classification model (e.g. `cardiffnlp/twitter-roberta-base-sentiment-latest`, `distilbert-base-uncased-finetuned-sst-2-english`) and emits the score for the chosen expected_label.
Threshold judgment lives on the `type: assertion` wrapper:
- type: assertion params: eval_type: text_sentiment eval_params: model: cardiffnlp/twitter-roberta-base-sentiment-latest expected_label: positive min_score: 0.7
Pack-level runtime eval (emits raw signal, no judgment):
evals:
- id: response-sentiment
type: text_sentiment
trigger: every_turn
params:
model: cardiffnlp/twitter-roberta-base-sentiment-latest
expected_label: positive
Params:
- model string (required)
- expected_label string (required)
- message_role string (optional, default "assistant")
- message_index int (optional, default -1)
- classifier_id string (optional)
func (*TextSentimentHandler) Eval ¶
func (h *TextSentimentHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval runs the shared text-classify pipeline.
func (*TextSentimentHandler) Type ¶
func (h *TextSentimentHandler) Type() string
Type returns the eval type identifier.
type TextToxicityHandler ¶
type TextToxicityHandler struct{}
TextToxicityHandler is a pure eval primitive: it scores text against a toxicity classification model (e.g. `unitary/toxic-bert`, `s-nlp/roberta_toxicity_classifier`) and emits the score for the chosen expected_label. Distinct from the legacy `toxicity` handler, which is an LLM-judge with the same name — `text_toxicity` is the deterministic classifier path that depends on `classify.TextClassifier` and an `inference` provider.
Threshold judgment lives on the `type: assertion` wrapper:
# "this output should NOT be toxic"
- type: assertion
params:
eval_type: text_toxicity
eval_params: { model: unitary/toxic-bert, expected_label: toxic }
max_score: 0.3
# "this output should sit in the neutral class"
- type: assertion
params:
eval_type: text_toxicity
eval_params: { model: s-nlp/roberta_toxicity_classifier, expected_label: neutral }
min_score: 0.7
Pack-level runtime eval (emits the raw signal, no judgment):
evals:
- id: response-toxicity
type: text_toxicity
trigger: every_turn
params: { model: unitary/toxic-bert, expected_label: toxic }
Params:
- model string (required) — backend model id
- expected_label string (required) — label whose score is emitted
- message_role string (optional, default "assistant")
- message_index int (optional, default -1)
- classifier_id string (optional)
func (*TextToxicityHandler) Eval ¶
func (h *TextToxicityHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval runs the shared text-classify pipeline.
func (*TextToxicityHandler) Type ¶
func (h *TextToxicityHandler) Type() string
Type returns the eval type identifier.
type ToolAntiPatternHandler ¶
type ToolAntiPatternHandler struct{}
ToolAntiPatternHandler checks that tool calls do NOT contain forbidden subsequences. Params: patterns []map[string]any — each with "sequence" ([]string) and optional "message" (string).
func (*ToolAntiPatternHandler) Eval ¶
func (h *ToolAntiPatternHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks for forbidden tool call subsequences.
func (*ToolAntiPatternHandler) Type ¶
func (h *ToolAntiPatternHandler) Type() string
Type returns the eval type identifier.
type ToolArgsExcludedSessionHandler ¶
type ToolArgsExcludedSessionHandler struct{}
ToolArgsExcludedSessionHandler checks that a tool was NOT called with specific argument values across the session. Params: tool_name string, excluded_args map[string]any. Also accepts legacy param forbidden_args map[string][]any (from the original tools_not_called_with_args validator).
func (*ToolArgsExcludedSessionHandler) Eval ¶
func (h *ToolArgsExcludedSessionHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (_ *evals.EvalResult, _ error)
Eval ensures the tool was never called with excluded args.
func (*ToolArgsExcludedSessionHandler) Type ¶
func (h *ToolArgsExcludedSessionHandler) Type() string
Type returns the eval type identifier.
type ToolArgsHandler ¶
type ToolArgsHandler struct{}
ToolArgsHandler checks that a tool was called with specific args. Params: tool_name string, expected_args map[string]any.
func (*ToolArgsHandler) Eval ¶
func (h *ToolArgsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that the specified tool was called with matching args.
func (*ToolArgsHandler) Type ¶
func (h *ToolArgsHandler) Type() string
Type returns the eval type identifier.
type ToolArgsSessionHandler ¶
type ToolArgsSessionHandler struct{}
ToolArgsSessionHandler checks that a tool was called with specific arguments across the session. Params: tool_name string, expected_args map[string]any.
func (*ToolArgsSessionHandler) Eval ¶
func (h *ToolArgsSessionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (_ *evals.EvalResult, _ error)
Eval checks tool calls for expected arguments.
func (*ToolArgsSessionHandler) Type ¶
func (h *ToolArgsSessionHandler) Type() string
Type returns the eval type identifier.
type ToolCallChainHandler ¶
type ToolCallChainHandler struct{}
ToolCallChainHandler checks a dependency chain of tool calls with per-step constraints. Params: steps []map — each with tool, result_includes, result_matches, args_match, no_error.
func (*ToolCallChainHandler) Eval ¶
func (h *ToolCallChainHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the chain of tool calls satisfies all step constraints in order.
func (*ToolCallChainHandler) Type ¶
func (h *ToolCallChainHandler) Type() string
Type returns the eval type identifier.
type ToolCallCountHandler ¶
type ToolCallCountHandler struct{}
ToolCallCountHandler checks the count of tool calls within bounds. Params: tool string (optional), min int (optional), max int (optional).
func (*ToolCallCountHandler) Eval ¶
func (h *ToolCallCountHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval counts matching tool calls and checks min/max bounds.
func (*ToolCallCountHandler) Type ¶
func (h *ToolCallCountHandler) Type() string
Type returns the eval type identifier.
type ToolCallSequenceHandler ¶
type ToolCallSequenceHandler struct{}
ToolCallSequenceHandler checks that tool calls appear in a specified subsequence order. Params: sequence []string — the expected tool names in order.
func (*ToolCallSequenceHandler) Eval ¶
func (h *ToolCallSequenceHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks subsequence ordering of tool calls.
func (*ToolCallSequenceHandler) Type ¶
func (h *ToolCallSequenceHandler) Type() string
Type returns the eval type identifier.
type ToolCallsWithArgsHandler ¶
type ToolCallsWithArgsHandler struct{}
ToolCallsWithArgsHandler checks that a tool was called with expected arguments. Supports exact value matching (expected_args), regex pattern matching (args_match), and result-level constraints (result_includes, result_matches, no_error).
func (*ToolCallsWithArgsHandler) Eval ¶
func (h *ToolCallsWithArgsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks tool calls for argument and result constraints.
func (*ToolCallsWithArgsHandler) Type ¶
func (h *ToolCallsWithArgsHandler) Type() string
Type returns the eval type identifier.
type ToolEfficiencyHandler ¶
type ToolEfficiencyHandler struct{}
ToolEfficiencyHandler checks tool usage efficiency metrics. Params:
- max_calls: int — maximum total tool calls allowed (optional)
- max_errors: int — maximum tool errors allowed (optional)
- max_error_rate: float64 — maximum error rate 0.0-1.0 (optional)
func (*ToolEfficiencyHandler) Eval ¶
func (h *ToolEfficiencyHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks tool efficiency metrics.
func (*ToolEfficiencyHandler) Type ¶
func (h *ToolEfficiencyHandler) Type() string
Type returns the eval type identifier.
type ToolExecHandler ¶
type ToolExecHandler struct{}
ToolExecHandler invokes a tool by name through the runtime tool registry and asserts the call succeeded. The pass condition is:
- tools.Registry.Execute returns no error, AND
- the resulting ToolResult has an empty Error field.
This makes it a generic "is this tool happy" gate that works with any registered tool: MCP-backed (e.g. a sandbox's run_tests), HTTP/local executors, custom client tools — whatever the host has wired up. The handler doesn't know or care about the transport.
Params:
- tool string (required) — registry name of the tool to invoke
- args map[string]any (optional) — arguments passed verbatim to Execute
- timeout_seconds int (optional) — bounds the call; default 120
Score is 1.0 on success, 0.0 on any failure (error returned, result has Error field set, missing tool, missing registry, timeout).
func (*ToolExecHandler) Eval ¶
func (h *ToolExecHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval invokes the named tool and returns pass/fail based on whether the tool reported success.
func (*ToolExecHandler) Type ¶
func (h *ToolExecHandler) Type() string
Type returns the eval type identifier.
type ToolNoRepeatHandler ¶
type ToolNoRepeatHandler struct{}
ToolNoRepeatHandler detects consecutive repeated calls to the same tool. Params:
- tools: []string — tool names to check (empty = all tools)
- max_repeats: int — maximum allowed consecutive calls to the same tool (default 1)
func (*ToolNoRepeatHandler) Eval ¶
func (h *ToolNoRepeatHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks for consecutive repeated tool calls.
func (*ToolNoRepeatHandler) Type ¶
func (h *ToolNoRepeatHandler) Type() string
Type returns the eval type identifier.
type ToolResultHasMediaHandler ¶
type ToolResultHasMediaHandler struct{}
ToolResultHasMediaHandler asserts that a named tool call returned media content of a given type. Params: tool string, media_type string ("image", "audio", "video", "document").
func (*ToolResultHasMediaHandler) Eval ¶
func (h *ToolResultHasMediaHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that a tool result contains ContentPart entries with the specified media type.
func (*ToolResultHasMediaHandler) Type ¶
func (h *ToolResultHasMediaHandler) Type() string
Type returns the eval type identifier.
type ToolResultIncludesHandler ¶
type ToolResultIncludesHandler struct{}
ToolResultIncludesHandler checks that tool results contain expected substrings. Params: tool string, patterns []string, occurrence int (optional, default 1).
func (*ToolResultIncludesHandler) Eval ¶
func (h *ToolResultIncludesHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks substring patterns in tool results.
func (*ToolResultIncludesHandler) Type ¶
func (h *ToolResultIncludesHandler) Type() string
Type returns the eval type identifier.
type ToolResultMatchesHandler ¶
type ToolResultMatchesHandler struct{}
ToolResultMatchesHandler checks that tool results match a regex pattern. Params: tool string, pattern string, occurrence int (optional, default 1).
func (*ToolResultMatchesHandler) Eval ¶
func (h *ToolResultMatchesHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks a regex pattern on tool results.
func (*ToolResultMatchesHandler) Type ¶
func (h *ToolResultMatchesHandler) Type() string
Type returns the eval type identifier.
type ToolResultMediaTypeHandler ¶
type ToolResultMediaTypeHandler struct{}
ToolResultMediaTypeHandler asserts the MIME type of media in a tool result. Params: tool string, mime_type string.
func (*ToolResultMediaTypeHandler) Eval ¶
func (h *ToolResultMediaTypeHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that a tool result contains a non-text ContentPart with the specified MIME type.
func (*ToolResultMediaTypeHandler) Type ¶
func (h *ToolResultMediaTypeHandler) Type() string
Type returns the eval type identifier.
type ToolsCalledHandler ¶
type ToolsCalledHandler struct{}
ToolsCalledHandler checks if specific tools were called successfully.
By default, only counts tool calls that completed without error. Use ignore_validation: true to also count calls that failed argument validation.
Params:
- tool_names/tools []string — required tool names
- min_calls int — minimum calls per tool (default 1)
- max_calls int — maximum calls per tool (default unbounded, use 0 to forbid). When any listed tool exceeds max_calls the assertion fails hard with score 0, regardless of min_calls satisfaction.
- ignore_validation bool — count validation failures as successful (default false)
- require_args bool — only count calls with non-empty arguments (default false)
func (*ToolsCalledHandler) Eval ¶
func (h *ToolsCalledHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that all expected tools were called.
func (*ToolsCalledHandler) Type ¶
func (h *ToolsCalledHandler) Type() string
Type returns the eval type identifier.
type ToolsCalledSessionHandler ¶
type ToolsCalledSessionHandler struct{}
ToolsCalledSessionHandler checks that specific tools were called across the full session. Params: tool_names []string, min_calls int (optional, default 1).
func (*ToolsCalledSessionHandler) Eval ¶
func (h *ToolsCalledSessionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (_ *evals.EvalResult, _ error)
Eval checks that all required tools were called at least min_calls times.
func (*ToolsCalledSessionHandler) Type ¶
func (h *ToolsCalledSessionHandler) Type() string
Type returns the eval type identifier.
type ToolsNotCalledHandler ¶
type ToolsNotCalledHandler struct{}
ToolsNotCalledHandler checks that specific tools were NOT called. Params: tool_names []string.
func (*ToolsNotCalledHandler) Eval ¶
func (h *ToolsNotCalledHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (result *evals.EvalResult, err error)
Eval checks that none of the forbidden tools were called.
func (*ToolsNotCalledHandler) Type ¶
func (h *ToolsNotCalledHandler) Type() string
Type returns the eval type identifier.
type ToolsNotCalledSessionHandler ¶
type ToolsNotCalledSessionHandler struct{}
ToolsNotCalledSessionHandler checks that specific tools were NOT called anywhere in the session. Params: tool_names []string.
func (*ToolsNotCalledSessionHandler) Eval ¶
func (h *ToolsNotCalledSessionHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (_ *evals.EvalResult, _ error)
Eval ensures forbidden tools were never called across the session.
func (*ToolsNotCalledSessionHandler) Type ¶
func (h *ToolsNotCalledSessionHandler) Type() string
Type returns the eval type identifier.
type TopicPolicyHandler ¶ added in v2.2.0
type TopicPolicyHandler struct {
// contains filtered or unexported fields
}
TopicPolicyHandler confines a conversation to a declared subject scope, decided by a classify.TopicClassifier rather than by the model being governed. Declared as a guardrail, it gates input by default and a denied turn never reaches the primary provider.
Pack validator (the normal declaration site):
validators:
- type: topic_policy
message: "I can help with questions about AltairaLabs products."
params:
description: Helps users evaluate and operate AltairaLabs products.
allowed: [Omnia and PromptKit, licensing and support]
disallowed: [politics]
Host side, an ordinary provider file:
id: topic-control role: inference type: nvidia-topic-control base_url: http://topic-control:8000/v1
Params: description (required), allowed (required, non-empty), disallowed, small_talk (allow|deny), examples.{allowed,disallowed}, on_deny (block|respond), on_unknown (deny|allow), on_error (deny|allow), recent_turns (>= 0), classifier_id.
func (*TopicPolicyHandler) Eval ¶ added in v2.2.0
func (h *TopicPolicyHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval classifies the current user message against the policy.
func (*TopicPolicyHandler) Type ¶ added in v2.2.0
func (h *TopicPolicyHandler) Type() string
Type returns the eval type identifier.
func (*TopicPolicyHandler) ValidateParams ¶ added in v2.2.0
func (h *TopicPolicyHandler) ValidateParams(params map[string]any) error
ValidateParams rejects a malformed policy at pack load rather than letting it degrade silently on every turn.
type ToxicityHandler ¶
type ToxicityHandler struct{}
ToxicityHandler scores the assistant output for insults, harassment, threats, hate speech via the LLM judge. DeepEval-equivalent name; distinct from the classifier-backed text_toxicity. Default-wired as a guardrail. Standard llm_judge params.
func (*ToxicityHandler) Eval ¶
func (h *ToxicityHandler) Eval( ctx context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval scores the current assistant output for toxic content.
func (*ToxicityHandler) Type ¶
func (h *ToxicityHandler) Type() string
Type returns the eval type identifier.
type VideoDurationHandler ¶
type VideoDurationHandler struct{}
VideoDurationHandler checks that video duration is within range. Params: min_seconds float64, max_seconds float64.
func (*VideoDurationHandler) Eval ¶
func (h *VideoDurationHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks video duration constraints.
func (*VideoDurationHandler) Type ¶
func (h *VideoDurationHandler) Type() string
Type returns the eval type identifier.
type VideoResolutionHandler ¶
type VideoResolutionHandler struct{}
VideoResolutionHandler checks that video resolution meets requirements. Params: min_width, max_width, min_height, max_height, presets []string.
func (*VideoResolutionHandler) Eval ¶
func (h *VideoResolutionHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks video resolution constraints.
func (*VideoResolutionHandler) Type ¶
func (h *VideoResolutionHandler) Type() string
Type returns the eval type identifier.
type WorkflowCompleteHandler ¶
type WorkflowCompleteHandler struct{}
WorkflowCompleteHandler checks that the workflow reached a terminal state. Reads evalCtx.Extras["workflow_complete"] (bool) and ["workflow_current_state"] (string).
func (*WorkflowCompleteHandler) Eval ¶
func (h *WorkflowCompleteHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, _ map[string]any, ) (*evals.EvalResult, error)
Eval checks whether the workflow is in a terminal state.
func (*WorkflowCompleteHandler) Type ¶
func (h *WorkflowCompleteHandler) Type() string
Type returns the eval type identifier.
type WorkflowSpokeInStateHandler ¶
type WorkflowSpokeInStateHandler struct{}
WorkflowSpokeInStateHandler checks that a workflow state actually produced assistant output, rather than merely being entered.
This closes the blind spot that let the in-turn handoff bug ship. state_is, transitioned_to, workflow_complete and workflow_transition_order all read the state machine's history, so a transition that generates no output at all satisfies every one of them — which is exactly what was broken: the machine advanced and the destination state never spoke.
Params: state string (required).
func (*WorkflowSpokeInStateHandler) Eval ¶
func (h *WorkflowSpokeInStateHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval reports whether any assistant message with non-empty text was produced while the workflow was in the named state.
func (*WorkflowSpokeInStateHandler) Type ¶
func (h *WorkflowSpokeInStateHandler) Type() string
Type returns the eval type identifier.
func (*WorkflowSpokeInStateHandler) ValidateParams ¶
func (h *WorkflowSpokeInStateHandler) ValidateParams(params map[string]any) error
ValidateParams checks that the required 'state' param is set to a non-empty string.
type WorkflowStateIsHandler ¶
type WorkflowStateIsHandler struct{}
WorkflowStateIsHandler checks that the current workflow state matches an expected value. Params: state string (required).
func (*WorkflowStateIsHandler) Eval ¶
func (h *WorkflowStateIsHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks if the current workflow state matches the expected value.
func (*WorkflowStateIsHandler) Type ¶
func (h *WorkflowStateIsHandler) Type() string
Type returns the eval type identifier.
func (*WorkflowStateIsHandler) ValidateParams ¶
func (h *WorkflowStateIsHandler) ValidateParams(params map[string]any) error
ValidateParams checks that the required 'state' param is set to a non-empty string.
type WorkflowToolAccessHandler ¶
type WorkflowToolAccessHandler struct{}
WorkflowToolAccessHandler enforces that tools are only called when the workflow is in a state that permits them. Params: rules []map[string]any — each with "state" (string) and "allowed" ([]string).
func (*WorkflowToolAccessHandler) Eval ¶
func (h *WorkflowToolAccessHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that every tool call occurred in a workflow state that allows it.
func (*WorkflowToolAccessHandler) Type ¶
func (h *WorkflowToolAccessHandler) Type() string
Type returns the eval type identifier.
type WorkflowTransitionOrderHandler ¶
type WorkflowTransitionOrderHandler struct{}
WorkflowTransitionOrderHandler checks that workflow state transitions happen in an expected order. Params: sequence []string (required) — expected ordered list of states that must appear as a subsequence.
func (*WorkflowTransitionOrderHandler) Eval ¶
func (h *WorkflowTransitionOrderHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks that the expected sequence appears as a subsequence of actual workflow transitions.
func (*WorkflowTransitionOrderHandler) Type ¶
func (h *WorkflowTransitionOrderHandler) Type() string
Type returns the eval type identifier.
type WorkflowTransitionedToHandler ¶
type WorkflowTransitionedToHandler struct{}
WorkflowTransitionedToHandler checks that a transition to a specific state occurred. Params: state string (required).
func (*WorkflowTransitionedToHandler) Eval ¶
func (h *WorkflowTransitionedToHandler) Eval( _ context.Context, evalCtx *evals.EvalContext, params map[string]any, ) (*evals.EvalResult, error)
Eval checks if the workflow transitioned to the specified state.
func (*WorkflowTransitionedToHandler) Type ¶
func (h *WorkflowTransitionedToHandler) Type() string
Type returns the eval type identifier.
Source Files
¶
- a2a_eval.go
- agent_invoked.go
- agent_not_invoked.go
- agent_response_contains.go
- answer_relevancy.go
- audio_duration.go
- audio_emotion.go
- audio_format.go
- behavioral_helpers.go
- bias.go
- classify_handler_base.go
- composition.go
- contains.go
- contains_any.go
- content_excludes.go
- contextual_precision.go
- contextual_recall.go
- contextual_relevancy.go
- cosine_similarity.go
- cost_budget.go
- directional.go
- exec_handler.go
- external_eval.go
- faithfulness.go
- field_presence.go
- guardrail_triggered.go
- hallucination.go
- helpers.go
- image_dimensions.go
- image_format.go
- image_moderation.go
- invariant_fields_preserved.go
- json_path.go
- json_schema.go
- json_valid.go
- judge_provider.go
- latency_budget.go
- length.go
- llm_judge.go
- llm_judge_session.go
- llm_judge_tool_calls.go
- media_helpers.go
- no_tool_errors.go
- outcome_equivalent.go
- pii_leakage.go
- rag_helpers.go
- regex.go
- register.go
- rest_eval.go
- role_violation.go
- safety_helpers.go
- score_helpers.go
- sentence_count.go
- skill_activated.go
- skill_activation_order.go
- skill_not_activated.go
- text_classify_helpers.go
- text_sentiment.go
- text_toxicity.go
- tool_anti_pattern.go
- tool_args.go
- tool_args_excluded_session.go
- tool_args_session.go
- tool_call_chain.go
- tool_call_count.go
- tool_call_sequence.go
- tool_call_views.go
- tool_calls_with_args.go
- tool_efficiency.go
- tool_exec.go
- tool_no_repeat.go
- tool_result_includes.go
- tool_result_matches.go
- tool_result_media.go
- tools_called.go
- tools_called_session.go
- tools_not_called.go
- tools_not_called_session.go
- topic_policy.go
- topic_policy_params.go
- toxicity.go
- video_duration.go
- video_resolution.go
- workflow_complete.go
- workflow_spoke_in_state.go
- workflow_state_is.go
- workflow_tool_access.go
- workflow_transition_order.go
- workflow_transitioned_to.go