Documentation
¶
Overview ¶
Package eval provides provider-neutral evaluation of agents.
Index ¶
- Constants
- func DatasetDigest(dataset Dataset) (string, error)
- func ValidateDataset(dataset *Dataset, evaluators Evaluators) error
- func WriteJSON(w io.Writer, report Report, options ExportOptions) error
- func WriteJSONFile(path string, report Report, options ExportOptions) error
- func WriteJUnit(w io.Writer, report Report) error
- func WriteJUnitFile(path string, report Report) error
- type Capabilities
- type Case
- type CaseResult
- type CaseStatus
- type Check
- type Coverage
- type Dataset
- type Durations
- type ErrorDetail
- type Evaluation
- type Evaluator
- type Evaluators
- type ExecutionSummary
- type ExportOptions
- type JudgeModel
- type MetricResult
- type MetricStatus
- type MetricSummary
- type ModelJudge
- type Observation
- type Recorder
- type RecorderOptions
- type Report
- type RunStatus
- type Runner
- type RunnerOptions
- type SpanEnd
- type SpanHandle
- type SpanStart
- type Subject
- type Summary
- type Target
- type TargetFactory
- type TokenUsage
- type ToolLayer
- type ToolSpan
- type ToolSpanStatus
- type Trace
Constants ¶
const ( EvaluatorOutputExact = "output_exact" EvaluatorOutputContains = "output_contains" EvaluatorOutputRegex = "output_regex" EvaluatorOutputJSONSchema = "output_json_schema" EvaluatorToolTrajectory = "tool_trajectory" EvaluatorMaxToolCalls = "max_tool_calls" EvaluatorMaxLatencyMS = "max_latency_ms" EvaluatorMaxTotalTokens = "max_total_tokens" )
const ( DefaultMaxSpans = 10_000 DefaultMaxContentBytes = 10 << 20 )
const ( // DatasetSchemaVersion is the dataset format understood by this package. DatasetSchemaVersion = "1" // ReportSchemaVersion is the report format emitted by this package. ReportSchemaVersion = "1" // RootAgentPath is the conventional path used for the evaluated root agent. RootAgentPath = "root" )
const DefaultMaxDatasetBytes int64 = 10 << 20
const (
// EvaluatorModelJudge is the optional LLM-backed rubric evaluator.
EvaluatorModelJudge = "model_judge"
)
Variables ¶
This section is empty.
Functions ¶
func DatasetDigest ¶
DatasetDigest returns a stable digest of the validated, resolved dataset. encoding/json sorts object keys; numeric spellings inside evaluator configs are preserved exactly.
func ValidateDataset ¶
func ValidateDataset(dataset *Dataset, evaluators Evaluators) error
ValidateDataset validates a dataset and resolves omitted threshold/required values in place. No agent execution occurs during validation.
func WriteJSON ¶
func WriteJSON(w io.Writer, report Report, options ExportOptions) error
WriteJSON writes a versioned report. It never mutates the in-memory report.
func WriteJSONFile ¶
func WriteJSONFile(path string, report Report, options ExportOptions) error
WriteJSONFile writes a report through a temporary file and atomically renames it into place on filesystems that support atomic rename.
func WriteJUnit ¶
WriteJUnit writes one testcase per evaluation case.
func WriteJUnitFile ¶
WriteJUnitFile writes a JUnit report atomically.
Types ¶
type Capabilities ¶
type Capabilities struct {
ToolCapture Coverage `json:"tool_capture"`
Usage Coverage `json:"usage"`
Output Coverage `json:"output"`
}
Capabilities describes which parts of an execution are trustworthy enough to grade.
type Case ¶
type Case struct {
ID string `json:"id"`
Input string `json:"input"`
Reference *string `json:"reference,omitempty"`
Tags []string `json:"tags,omitempty"`
Checks []Check `json:"checks"`
}
Case is one independent agent execution and its checks.
type CaseResult ¶
type CaseResult struct {
CaseID string `json:"case_id"`
AttemptID string `json:"attempt_id,omitempty"`
Status CaseStatus `json:"status"`
Observation Observation `json:"observation"`
Metrics []MetricResult `json:"metrics,omitempty"`
Errors []ErrorDetail `json:"errors,omitempty"`
Durations Durations `json:"durations"`
}
CaseResult combines an observation and its metric results.
func Grade ¶
func Grade(ctx context.Context, evalCase Case, observation Observation, evaluators Evaluators) (CaseResult, error)
Grade evaluates one saved observation without running an agent.
type CaseStatus ¶
type CaseStatus string
CaseStatus is the suite-level result for a case.
const ( CaseStatusPass CaseStatus = "pass" CaseStatusFail CaseStatus = "fail" CaseStatusError CaseStatus = "error" CaseStatusNotStarted CaseStatus = "not_started" )
type Check ¶
type Check struct {
ID string `json:"id"`
Type string `json:"type"`
Config json.RawMessage `json:"config"`
Threshold *float64 `json:"threshold,omitempty"`
Required *bool `json:"required,omitempty"`
}
Check configures one evaluator. Config is interpreted by the evaluator named in Type. Pointer fields preserve the distinction between omitted values and explicit false/zero values while loading a dataset.
func (Check) RequiredValue ¶
RequiredValue returns whether a check gates the case. The default is true.
func (Check) ThresholdValue ¶
ThresholdValue returns the resolved threshold. The default is 1.
type Coverage ¶
type Coverage string
Coverage describes how completely an observation captured a capability.
type Dataset ¶
type Dataset struct {
SchemaVersion string `json:"schema_version"`
ID string `json:"id"`
Cases []Case `json:"cases"`
}
Dataset is an ordered collection of evaluation cases.
func LoadDataset ¶
LoadDataset reads and validates a dataset using the built-in evaluators.
func LoadDatasetWithEvaluators ¶
func LoadDatasetWithEvaluators(r io.Reader, evaluators Evaluators) (Dataset, error)
LoadDatasetWithEvaluators reads a dataset and validates custom check types against the supplied evaluator registry.
func ParseDataset ¶
ParseDataset parses and validates a JSON dataset with built-in evaluators.
func ParseDatasetWithEvaluators ¶
func ParseDatasetWithEvaluators(data []byte, evaluators Evaluators) (Dataset, error)
ParseDatasetWithEvaluators parses a single strict JSON document. Duplicate object keys and unknown fields are rejected, including inside check configs.
type Durations ¶
type Durations struct {
SetupMS int64 `json:"setup_ms,omitempty"`
ExecutionMS int64 `json:"execution_ms,omitempty"`
GradingMS int64 `json:"grading_ms,omitempty"`
CleanupMS int64 `json:"cleanup_ms,omitempty"`
TotalMS int64 `json:"total_ms,omitempty"`
}
Durations separates agent execution from setup, grading, and cleanup.
type ErrorDetail ¶
type ErrorDetail struct {
Stage string `json:"stage"`
Type string `json:"type,omitempty"`
Message string `json:"message"`
}
ErrorDetail is a serializable error. Stage identifies where the failure happened without exposing implementation-specific error values.
type Evaluation ¶
type Evaluation struct {
Case Case
Check Check
Observation Observation
}
Evaluation is the input to one evaluator invocation.
type Evaluator ¶
type Evaluator interface {
Name() string
Validate(Check) error
Evaluate(context.Context, Evaluation) (MetricResult, error)
}
Evaluator validates and evaluates one check type.
type Evaluators ¶
Evaluators maps check types to their implementations.
func BuiltinEvaluators ¶
func BuiltinEvaluators() Evaluators
BuiltinEvaluators returns a fresh registry of deterministic evaluators.
type ExecutionSummary ¶
type ExecutionSummary struct {
LLMCalls int `json:"llm_calls"`
ToolCalls int `json:"tool_calls"`
SubAgentCalls int `json:"sub_agent_calls"`
ExecutionTimeMS int64 `json:"execution_time_ms"`
UsedTools []string `json:"used_tools,omitempty"`
UsedSubAgents []string `json:"used_sub_agents,omitempty"`
UsageByModel map[string]TokenUsage `json:"usage_by_model,omitempty"`
}
ExecutionSummary is the stable report representation of an agent execution summary. It intentionally excludes arbitrary response metadata.
type ExportOptions ¶
type ExportOptions struct {
IncludeObservations bool
MaxContentBytes int
Redact func(field, value string) string
}
ExportOptions controls how much execution content is written to JSON. Observations are omitted by default. Any redaction or truncation makes the affected observation ineligible for offline regrading.
type JudgeModel ¶
type JudgeModel interface {
GenerateDetailed(context.Context, string, ...interfaces.GenerateOption) (*interfaces.LLMResponse, error)
Name() string
}
JudgeModel is the minimum model API needed by ModelJudge. interfaces.LLM implementations satisfy it.
type MetricResult ¶
type MetricResult struct {
CheckID string `json:"check_id"`
Evaluator string `json:"evaluator"`
Status MetricStatus `json:"status"`
Score *float64 `json:"score,omitempty"`
Threshold float64 `json:"threshold"`
Required bool `json:"required"`
Message string `json:"message,omitempty"`
Evidence map[string]any `json:"evidence,omitempty"`
}
MetricResult is the result of one configured check.
type MetricStatus ¶
type MetricStatus string
MetricStatus is the outcome of evaluating one check.
const ( MetricStatusPass MetricStatus = "pass" MetricStatusFail MetricStatus = "fail" MetricStatusError MetricStatus = "error" )
type MetricSummary ¶
type MetricSummary struct {
Evaluator string `json:"evaluator"`
Scored int `json:"scored"`
Errors int `json:"errors"`
MeanScore float64 `json:"mean_score,omitempty"`
}
MetricSummary aggregates one evaluator without hiding unavailable results.
type ModelJudge ¶
type ModelJudge struct {
// contains filtered or unexported fields
}
ModelJudge grades an agent response against a natural-language rubric using an injected model. It has no access to the evaluated agent's tools or memory.
func NewModelJudge ¶
func NewModelJudge(model JudgeModel) *ModelJudge
NewModelJudge creates an opt-in evaluator backed by model. A nil model is valid for dataset validation, but evaluation returns a configuration error.
func (*ModelJudge) Evaluate ¶
func (j *ModelJudge) Evaluate(ctx context.Context, in Evaluation) (MetricResult, error)
func (*ModelJudge) Name ¶
func (*ModelJudge) Name() string
func (*ModelJudge) Validate ¶
func (*ModelJudge) Validate(check Check) error
type Observation ¶
type Observation struct {
CaseID string `json:"case_id"`
AttemptID string `json:"attempt_id"`
Status RunStatus `json:"status"`
Output *string `json:"output,omitempty"`
Error *ErrorDetail `json:"error,omitempty"`
AgentName string `json:"agent_name,omitempty"`
Model string `json:"model,omitempty"`
Usage *TokenUsage `json:"usage,omitempty"`
ExecutionSummary ExecutionSummary `json:"execution_summary"`
Trace Trace `json:"trace"`
Capabilities Capabilities `json:"capabilities"`
ExecutionDuration int64 `json:"execution_duration_ms"`
Regradable bool `json:"regradable"`
}
Observation is everything graders may inspect about one execution.
type Recorder ¶
type Recorder struct {
// contains filtered or unexported fields
}
Recorder captures correlated tool spans safely across concurrent calls.
func NewRecorder ¶
func NewRecorder(options RecorderOptions) *Recorder
NewRecorder creates an empty recorder with bounded retention.
func (*Recorder) Begin ¶
Begin records tool entry and returns a context carrying the new span as the parent for nested decorators and sub-agents.
func (*Recorder) EnablePairedLayers ¶
func (r *Recorder) EnablePairedLayers()
EnablePairedLayers tells Snapshot that attempt spans should have execution children. The SDK adapter enables this when it installs both decorators.
func (*Recorder) End ¶
func (r *Recorder) End(handle SpanHandle, end SpanEnd)
End completes exactly the span identified by handle.
func (*Recorder) MarkIncomplete ¶
MarkIncomplete marks the trace unusable for checks requiring complete tool capture. It is useful when an adapter detects a boundary it cannot observe.
type RecorderOptions ¶
RecorderOptions bounds trace retention. Now exists to make recorder tests deterministic; production callers normally leave it nil.
type Report ¶
type Report struct {
SchemaVersion string `json:"schema_version"`
DatasetID string `json:"dataset_id"`
DatasetDigest string `json:"dataset_digest"`
ConfigFingerprint string `json:"config_fingerprint,omitempty"`
BuildRevision string `json:"build_revision,omitempty"`
StartedAt time.Time `json:"started_at"`
EndedAt time.Time `json:"ended_at"`
ResolvedDataset *Dataset `json:"resolved_dataset,omitempty"`
Cases []CaseResult `json:"cases"`
Summary Summary `json:"summary"`
}
Report is the versioned result of a dataset run.
func ReadReport ¶
ReadReport reads one strict JSON report document.
type Runner ¶
type Runner struct {
Factory TargetFactory
Evaluators Evaluators
Options RunnerOptions
}
Runner executes validated cases through freshly constructed targets.
type RunnerOptions ¶
type RunnerOptions struct {
Concurrency int
CaseTimeout time.Duration
GradingTimeout time.Duration
CleanupTimeout time.Duration
Recorder RecorderOptions
ConfigFingerprint string
BuildRevision string
Now func() time.Time
AttemptID func() string
}
RunnerOptions configures suite execution. Durations of zero mean no extra deadline. Concurrency defaults to one.
type SpanHandle ¶
type SpanHandle struct {
// contains filtered or unexported fields
}
SpanHandle identifies a span allocated by Recorder.Begin.
func (SpanHandle) Valid ¶
func (h SpanHandle) Valid() bool
Valid reports whether the recorder retained this span.
type SpanStart ¶
type SpanStart struct {
AgentPath string
Layer ToolLayer
Tool string
Method string
Arguments string
}
SpanStart describes a tool invocation at decorator entry.
type Subject ¶
type Subject interface {
RunDetailed(context.Context, string) (*interfaces.AgentResponse, error)
}
Subject is the minimum execution API required by Runner.
type Summary ¶
type Summary struct {
Total int `json:"total"`
Passed int `json:"passed"`
Failed int `json:"failed"`
Errors int `json:"errors"`
NotStarted int `json:"not_started"`
PassRate float64 `json:"pass_rate"`
Metrics []MetricSummary `json:"metrics,omitempty"`
}
Summary aggregates the suite result.
type Target ¶
type Target struct {
Subject Subject
Close func(context.Context) error
Capabilities Capabilities
}
Target owns a freshly constructed subject and its cleanup callback.
type TargetFactory ¶
TargetFactory builds isolated state for one case attempt.
type TokenUsage ¶
type TokenUsage struct {
InputTokens int `json:"input_tokens"`
OutputTokens int `json:"output_tokens"`
TotalTokens int `json:"total_tokens"`
ReasoningTokens int `json:"reasoning_tokens,omitempty"`
CacheCreationInputTokens int `json:"cache_creation_input_tokens,omitempty"`
CacheReadInputTokens int `json:"cache_read_input_tokens,omitempty"`
}
TokenUsage is the stable report representation of interfaces.TokenUsage.
type ToolLayer ¶
type ToolLayer string
ToolLayer identifies whether a span represents a provider attempt or an actual execution after policy decorators.
type ToolSpan ¶
type ToolSpan struct {
ID string `json:"id"`
ParentID string `json:"parent_id,omitempty"`
AgentPath string `json:"agent_path"`
Layer ToolLayer `json:"layer"`
Sequence uint64 `json:"sequence"`
Tool string `json:"tool"`
Method string `json:"method"`
Arguments string `json:"arguments,omitempty"`
Result string `json:"result,omitempty"`
Error string `json:"error,omitempty"`
StartedAt time.Time `json:"started_at"`
EndedAt *time.Time `json:"ended_at,omitempty"`
Status ToolSpanStatus `json:"status"`
ContentTruncated bool `json:"content_truncated,omitempty"`
}
ToolSpan records one invocation observed at one decorator layer.
type ToolSpanStatus ¶
type ToolSpanStatus string
ToolSpanStatus is the terminal state of a tool span.
const ( ToolSpanCompleted ToolSpanStatus = "completed" ToolSpanError ToolSpanStatus = "error" ToolSpanPanicked ToolSpanStatus = "panicked" ToolSpanShortCircuited ToolSpanStatus = "short_circuited" ToolSpanIncomplete ToolSpanStatus = "incomplete" )