Documentation
¶
Overview ¶
Package eval stores eval task sets and runs comparison evals against a base agent's live engine under the no-side-effects execution policy (agent.ExecPolicy). Stages B and C of design/eval-subsystem.md: L2 task sets, the L3 runner and its objective metrics, then the blinded pairing, judging and decision rule layered on top.
Index ¶
- Constants
- Variables
- func BoundTraceLimit(n int) int
- func Categories() []string
- func Dimensions() []string
- func HistoryCategories() []string
- func IsTerminal(status string) bool
- func JudgeAgentIdent(baseAgent string) string
- func JudgeConvID(runID, pass int64) string
- func ProbeKinds() []string
- func SampleConvID(runID, taskID int64, k int, variantID int64) string
- func SourceKey(convID string, messageID int64) string
- func ValidCategory(c string) bool
- func ValidWinner(w string) bool
- func VariantFor(a Assignment, order, winner string) int64
- type Agreement
- type Assignment
- type BlindedItem
- type BlindedResponse
- type BlindedToolCall
- type Candidate
- type CategoryResult
- type Completeness
- type Config
- type ConvStats
- type Engine
- type EngineSource
- type Estimate
- type EstimateInput
- type EstimateVariant
- type Estimator
- type Gate
- type ImportError
- type JSONLTask
- type Judge
- func (j *Judge) Available() bool
- func (j *Judge) Config() JudgeConfig
- func (j *Judge) IsActive(runID int64) bool
- func (j *Judge) SetConfig(cfg JudgeConfig)
- func (j *Judge) Shutdown()
- func (j *Judge) Start(ctx context.Context, runID int64, opts JudgeOpts) (*JudgePass, error)
- func (j *Judge) Stop(runID int64) bool
- func (j *Judge) StopAll()
- type JudgeConfig
- type JudgeOpts
- type JudgePass
- type Judgment
- type JudgmentItem
- type ModelPrice
- type Overlay
- type Pair
- type PairDetail
- type PairItem
- type PairSide
- type PairVerdict
- type PairView
- type PendingItem
- type PrecedingMessage
- type PriceLookup
- type Probe
- type ProbeOpts
- type ProgressEvent
- type Run
- type Runner
- type Sample
- type SpecSource
- type StatsLookup
- type Store
- func (s *Store) AddJudgeCost(ctx context.Context, runID int64, cost float64) error
- func (s *Store) AddRunCost(ctx context.Context, runID int64, cost float64) error
- func (s *Store) AddSample(ctx context.Context, smp Sample) (*Sample, error)
- func (s *Store) AddTask(ctx context.Context, setID int64, t Task) (*Task, error)
- func (s *Store) Close() error
- func (s *Store) CountPairs(ctx context.Context, runID int64) (int, error)
- func (s *Store) CountSamples(ctx context.Context, runID int64) (int, error)
- func (s *Store) CountTraces(ctx context.Context, f TraceFilter) (int, error)
- func (s *Store) CreatePairs(ctx context.Context, runID int64) (int, error)
- func (s *Store) CreateRun(ctx context.Context, run Run, variants []Variant) (*Run, []Variant, error)
- func (s *Store) CreateTaskSet(ctx context.Context, name, description string) (*TaskSet, error)
- func (s *Store) DeleteTask(ctx context.Context, setID, taskID int64) error
- func (s *Store) DeleteTaskSet(ctx context.Context, name string) error
- func (s *Store) ExportJSONL(ctx context.Context, setID int64, w io.Writer) error
- func (s *Store) FinishRun(ctx context.Context, runID int64, status, errMsg string) error
- func (s *Store) GetBlindedItem(ctx context.Context, itemID int64) (*BlindedItem, error)
- func (s *Store) GetItem(ctx context.Context, itemID int64) (*JudgmentItem, error)
- func (s *Store) GetPair(ctx context.Context, pairID int64) (*Pair, error)
- func (s *Store) GetRun(ctx context.Context, id int64) (*Run, error)
- func (s *Store) GetSample(ctx context.Context, id int64) (*Sample, error)
- func (s *Store) GetTask(ctx context.Context, setID, taskID int64) (*Task, error)
- func (s *Store) GetTaskSet(ctx context.Context, name string) (*TaskSet, error)
- func (s *Store) GetTaskSetByID(ctx context.Context, id int64) (*TaskSet, error)
- func (s *Store) GetTrace(ctx context.Context, id int64) (*TraceRow, agent.TracePayload, error)
- func (s *Store) ImportJSONL(ctx context.Context, setID int64, r io.Reader) (int, error)
- func (s *Store) ListItems(ctx context.Context, runID int64) ([]JudgmentItem, error)
- func (s *Store) ListPairs(ctx context.Context, runID int64) ([]Pair, error)
- func (s *Store) ListPending(ctx context.Context, runID int64, limit, sampleN int) ([]PendingItem, error)
- func (s *Store) ListRuns(ctx context.Context, taskSetID int64, status string) ([]Run, error)
- func (s *Store) ListSamples(ctx context.Context, runID int64) ([]Sample, error)
- func (s *Store) ListTaskSamples(ctx context.Context, runID, taskID int64) ([]Sample, error)
- func (s *Store) ListTaskSets(ctx context.Context) ([]TaskSet, error)
- func (s *Store) ListTasks(ctx context.Context, setID int64) ([]Task, error)
- func (s *Store) ListTraces(ctx context.Context, f TraceFilter) ([]TraceRow, error)
- func (s *Store) ListVariants(ctx context.Context, runID int64) ([]Variant, error)
- func (s *Store) ListVerdicts(ctx context.Context, runID int64) ([]Verdict, error)
- func (s *Store) PairDetails(ctx context.Context, runID, taskID int64) (*PairView, error)
- func (s *Store) PruneTracesBefore(ctx context.Context, before time.Time) (int, error)
- func (s *Store) RecordVerdict(ctx context.Context, v Verdict) (*Verdict, error)
- func (s *Store) RunTasks(ctx context.Context, run *Run) ([]Task, error)
- func (s *Store) SaveTrace(ctx context.Context, t agent.TurnTrace) error
- func (s *Store) SavedTaskSources(ctx context.Context) (map[string]struct{}, error)
- func (s *Store) SetRunStatus(ctx context.Context, runID int64, status, errMsg string) error
- func (s *Store) Summarize(ctx context.Context, runID int64, opts SummaryOpts) (*Summary, error)
- func (s *Store) UpdateTask(ctx context.Context, setID, taskID int64, patch TaskPatch) (*Task, error)
- func (s *Store) UpdateTaskSet(ctx context.Context, name string, newName, description *string) (*TaskSet, error)
- type SuggestOpts
- type Summary
- type SummaryOpts
- type Task
- type TaskIDList
- type TaskMetrics
- type TaskPatch
- type TaskSet
- type TaskVariantMetrics
- type TraceFilter
- type TraceRow
- type Variant
- type VariantEstimate
- type VariantMetrics
- type VariantVerdict
- type Verdict
Constants ¶
const ( // BasisHistory: the task's source conversation has real telemetry, so the // per-turn figure is a measurement rather than a model. BasisHistory = "history" // BasisListPrice: no usable history, so the figure is the variant's list // price times a nominal token budget. BasisListPrice = "list_price" // BasisUnknown: neither was available. The caller shows the hard cap alone // rather than a fabricated number. BasisUnknown = "unknown" )
Estimate bases, weakest last. A figure is only ever labelled with a basis that actually contributed to it.
const ( // ProbeDenialCompliance: the operator refused something. Does the // candidate accept the refusal, or re-issue the call by another route? ProbeDenialCompliance = "denial_compliance" // ProbeTierBoundary: does the candidate act within the permission tier it // is actually on, and describe that tier honestly when asked? ProbeTierBoundary = "tier_boundary" // ProbeBudgetHint: the operator bounded the answer ("one sentence", "no // tools"). Does the candidate honour the bound? ProbeBudgetHint = "budget_hint" // ProbeApprovalPolicy: does the candidate treat a chat request as standing // consent for a tool the operator has not blessed? ProbeApprovalPolicy = "approval_policy" // ProbeSkillInstruction: does the candidate follow a configured skill's // written instructions, and leave them alone when they do not apply? ProbeSkillInstruction = "skill_instruction" // ProbePersonaFidelity: does the candidate hold the persona sections the // operator wrote? ProbePersonaFidelity = "persona_fidelity" )
Probe kinds — the behaviour family a probe exercises. Recorded on the task's tags, so a results table can be read by family, and named in the prose the judge sees.
const ( CategoryChat = "chat" CategorySkillCommand = "skill_command" CategoryScheduled = "scheduled" CategoryToolHeavy = "tool_heavy" CategoryProbe = "probe" )
Task categories. The curation axis from design/eval-subsystem.md §4.3 — validated in Go rather than by a SQL CHECK constraint, matching the house style (there are no CHECK constraints anywhere in the schema).
The first four are the bottom-up axis: what the agent has actually been asked to do. CategoryProbe is the top-down one (Stage E item 2) and is deliberately its own value rather than folded into chat or tool_heavy: a probe is generated from written intent, not sampled from history, so mixing the two would let a regression on specified behaviour hide inside a chat win rate — and the per-category breakdown exists to keep those questions apart.
const ( StatusPending = "pending" StatusRunning = "running" StatusDone = "done" StatusCapped = "capped" StatusStopped = "stopped" StatusFailed = "failed" )
Run statuses.
const ( SampleOK = "ok" SampleFailed = "failed" )
Sample statuses. A sample is the unit of failure tolerance: a provider hiccup fails the sample, never the run.
const ( ItemPending = "pending" ItemJudged = "judged" )
Judgment item statuses.
const ( OrderAB = "ab" OrderBA = "ba" )
Presentation orders. An item's order says which pair letter is shown to the judge first: "ab" shows the pair's A sample as Response A, "ba" swaps them. Two items per pair, one of each, is what makes position-bias control structural rather than a matter of judge discipline.
const ( WinnerA = "a" WinnerB = "b" WinnerTie = "tie" )
Verdict winners, in terms of the *presented* responses — the judge never learns the pair letters, only "Response A" and "Response B" as shown to it.
const ( DimTaskSuccess = "task_success" DimToolPath = "tool_path" DimPersonaFit = "persona_fit" DimLength = "length" )
Judgment dimensions, in the order the rubric lists them.
const ( // SignalToolFault: the reply had a rejected (bad args) or failed // (transport) tool call. The sharpest model-fault signal we record. SignalToolFault = "tool_fault" // SignalManyRounds: the reply took three or more tool rounds. SignalManyRounds = "many_rounds" // SignalHighCost: the reply's cost is in the candidate pool's top decile. SignalHighCost = "high_cost" // SignalCommandSkill: a command-triggered skill drove the turn. SignalCommandSkill = "command_skill" )
Signals a past turn can carry. A turn with none of them is not offered at all — "interesting" is the whole point, and an unremarkable turn teaches the eval set nothing.
const ( VerdictUpgrade = "upgrade" VerdictDowngrade = "downgrade" VerdictNoRegressions = "no_regressions" VerdictInconclusive = "inconclusive" )
Decision-rule outcomes.
The rule is asymmetric on purpose: the objective gates alone can declare a downgrade (a gate failed — no judge needed to reject a candidate) or report that nothing regressed, but they can never declare an upgrade. That needs the judge win-rate, so a candidate cannot be promoted on the strength of being cheap and quiet.
const ( GateRejectedRate = "rejected_rate" GateMeanRounds = "mean_rounds" GateCostPerTask = "mean_cost_per_task" )
Gate names, as they appear in the gate table.
const ( PairOutcomeWin = "win" PairOutcomeLoss = "loss" PairOutcomeTie = "tie" PairOutcomePending = "pending" )
Pair outcomes, from the candidate's point of view. They follow the aggregation rules exactly: a pair is only decided once both presentation orders carry a judge verdict, and orders that disagree are a tie.
const JudgeInternal = "judge_model"
JudgeInternal is the judge identity the internal judge records verdicts under. It is not JudgeOperator, so its calls count toward the win rate and flip items to judged, exactly like the MCP judge's; it is a fixed name rather than a key name because there is no key — the caller is the server.
const JudgeOperator = "operator"
JudgeOperator is the judge identity reserved for the operator's calibration marks. Stored as ordinary verdicts (no schema of their own) and excluded from the win rate; they only feed the operator–judge agreement figure.
const RubricVersion = "v1"
RubricVersion is the revision of the judging rubric the internal judge grades under, and what it stamps on every verdict it writes.
It must match the `Rubric version:` line of `.claude/skills/judge-eval/SKILL.md`: the two judges have to be comparable, and a results view naming one version for verdicts produced under two different rubrics would be a lie. TestRubricVersion_MatchesTheSkillFile pins the pair.
Variables ¶
var ( // ErrJudgeNotConfigured means [eval] judge_model is unset: the internal // judge is opt-in, and its absence is not a failure — the MCP judge path // is unaffected. ErrJudgeNotConfigured = errors.New("eval: internal judge not configured") // ErrRunNotTerminal means the run is still producing samples. Judging a // moving queue wastes money on pairs that do not exist yet. ErrRunNotTerminal = errors.New("eval: run is not terminal") // ErrJudgeActive means a judging pass is already working this run's queue. ErrJudgeActive = errors.New("eval: run is already being judged") // ErrJudgeModelSwapped means the completion came back from a model other // than the configured one — a router cost_limit fallback, most likely. ErrJudgeModelSwapped = errors.New("eval: judge completion served by a different model") )
Errors the judge endpoints map onto status codes.
var ErrNameTaken = errors.New("eval: task set name already exists")
ErrNameTaken is returned when a task-set name collides with an existing one.
var ErrNotFound = errors.New("eval: not found")
ErrNotFound is the sentinel every lookup wraps when the addressed row does not exist, so REST handlers can classify a 404 with errors.Is rather than by inspecting the message (the tool.ErrToolNotFound convention).
var ErrRunNotActive = errors.New("eval: run is not active")
ErrRunNotActive is returned by handlers that tried to stop a run that is already terminal.
var ErrTaskSetInUse = errors.New("eval: task set is referenced by runs")
ErrTaskSetInUse is returned by DeleteTaskSet when runs still reference the set. Deleting would either orphan those runs or cascade away results the operator may still be reading, so the delete is refused and the caller maps it to 409.
Functions ¶
func BoundTraceLimit ¶ added in v0.47.0
BoundTraceLimit resolves a requested page size to the one a listing will actually use. Exported so a caller can echo the effective limit rather than the one it asked for: a pager that trusts its own request walks past rows when the store clamped it.
func Categories ¶
func Categories() []string
Categories returns the five valid task categories, in the order the docs list them. Every stratified draw and per-category breakdown cycles this slice, so adding a value here is what widens the axis everywhere.
func Dimensions ¶
func Dimensions() []string
Dimensions returns the four judgment dimensions, in rubric order.
func HistoryCategories ¶ added in v0.47.0
func HistoryCategories() []string
HistoryCategories returns the categories a turn sampled from history can land in — Categories() minus CategoryProbe, which is generated from written intent and never inferred from a turn. Suggest stratifies across these rather than all of Categories(): a share reserved for a family the pass can never fill would just shrink the pass.
func IsTerminal ¶
IsTerminal reports whether a run status is final — nothing further will be dispatched and the row will not change again.
func JudgeAgentIdent ¶ added in v0.47.0
JudgeAgentIdent is the pseudo-identity judging spend and audit events are attributed to. "#" is rejected by the resource-name validator, so it can never collide with a real agent and never lands in one's totals.
func JudgeConvID ¶ added in v0.47.0
JudgeConvID is the cost tracker's session key for one judging pass, distinct from every eval:{run}:{task}:{k}:{variant} sample key so judge cost can never be mistaken for a sample's.
The pass number is part of it because the key is also what the router's own session guards read: a key shared across passes accumulates forever, so a [llm.fallbacks] cost_limit rule would eventually swap the judge model out from under the rubric, and a [costs] hard limit would refuse every later pass. One key per pass keeps both guards measuring the pass in front of them.
func ProbeKinds ¶ added in v0.47.0
func ProbeKinds() []string
ProbeKinds returns the probe families in the order the generator emits them: the three canned families first (they need no configuration and are the starter set), then the spec-derived ones.
func SampleConvID ¶
SampleConvID mints the in-flight identity for one sample. The variant is part of it, not just run/task/k: the identity doubles as the cost tracker's session key, and two variants of the same (task, k) sharing one key would bill the second for the first's spend — enough to trip the cap early and to report a per-sample cost that is really a running total. It is also what makes the audit log's session grouping genuinely per-sample.
func SourceKey ¶ added in v0.44.0
SourceKey identifies a turn by its source conversation and message, the pair eval_tasks records when a suggestion is accepted.
func ValidCategory ¶
ValidCategory reports whether c is one of the five curation categories.
func ValidWinner ¶
ValidWinner reports whether w is one of the three verdict outcomes.
func VariantFor ¶
func VariantFor(a Assignment, order, winner string) int64
VariantFor resolves a presented winner letter back to the variant that produced it, given the item's presentation order. Returns 0 for a tie.
Two indirections, deliberately: the judge names a presented letter, the item says whether that letter was the pair's own A or B, and only the assignment says which variant that was. Nothing short of the pair row can unblind a verdict.
Types ¶
type Agreement ¶
type Agreement struct {
Items int `json:"items"`
Agreed int `json:"agreed"`
Rate float64 `json:"rate"`
}
Agreement is the operator–judge calibration figure: how often the operator's own call on a calibration item matched the judge's. Below roughly 80 % the rubric wants fixing before headless judging is trusted, since a drifted rubric silently devalues every later run.
type Assignment ¶
Assignment is the decoded eval_pairs.assignment JSON: which variant each presented letter really was.
func DecodeAssignment ¶
func DecodeAssignment(raw string) (Assignment, error)
DecodeAssignment parses a pair's unblinding key.
type BlindedItem ¶
type BlindedItem struct {
ItemID int64 `json:"item_id"`
RunID int64 `json:"run_id"`
TaskID int64 `json:"task_id"`
Prompt string `json:"prompt"`
Category string `json:"category"`
// Notes is the task's free-text "what good looks like". Judge context, not
// an assertion — nothing parses it.
Notes string `json:"notes,omitempty"`
// PinnedHistory is the context the turn ran against, so a verdict can tell
// a non-sequitur from a correct follow-up.
PinnedHistory json.RawMessage `json:"pinned_history,omitempty"`
Status string `json:"status"`
ResponseA BlindedResponse `json:"response_a"`
ResponseB BlindedResponse `json:"response_b"`
}
BlindedItem is the judge-visible payload for one judgment item.
type BlindedResponse ¶
type BlindedResponse struct {
Response string `json:"response"`
Rounds int `json:"rounds"`
StopReason string `json:"stop_reason,omitempty"`
ToolCalls []BlindedToolCall `json:"tool_calls"`
}
BlindedResponse is one side of a pair. Deliberately absent: variant name, model, provider, token usage, cost, latency, and the sample's conversation id (which names the variant). Duration is dropped too — a consistently slower side is an identity hint.
type BlindedToolCall ¶
type BlindedToolCall struct {
Round int `json:"round"`
Name string `json:"tool_name"`
Server string `json:"server_name,omitempty"`
Outcome string `json:"outcome"`
Arguments string `json:"arguments,omitempty"`
Result string `json:"result,omitempty"`
Error string `json:"error,omitempty"`
}
BlindedToolCall is one tool call as the judge sees it. Built field by field from agent.ToolCallRecord rather than embedding it, so a field added to the record cannot leak into a judge payload by default.
type Candidate ¶ added in v0.44.0
type Candidate struct {
Prompt string `json:"prompt"`
Category string `json:"category"`
ConversationID string `json:"conversation_id"`
MessageID int64 `json:"message_id"`
CreatedAt time.Time `json:"created_at"`
Signals []string `json:"signals"`
Preceding []PrecedingMessage `json:"preceding"`
// Agent handled the turn; empty when its stats row was pruned.
Agent string `json:"agent"`
// Trigger is the skill or schedule name that fired a scheduled turn, whose
// prompt is a generated label rather than anything a person wrote. Empty
// for every other category.
Trigger string `json:"trigger"`
// ReplyPreview is the head of the answering reply.
ReplyPreview string `json:"reply_preview"`
// ToolCalls, MaxRound, Faults and CostUSD describe the answering reply:
// the same telemetry the signals are drawn from, in numbers.
ToolCalls int `json:"tool_calls"`
MaxRound int `json:"max_round"`
Faults int `json:"faults"`
CostUSD float64 `json:"cost_usd"`
}
Candidate is one past turn offered as a test case. Beyond what an accepted task needs, it carries enough of the source turn — who ran it, what set it off, what it cost, what came back — for the offer to be judged without opening the conversation it came from.
func Suggest ¶ added in v0.44.0
func Suggest(turns []agent.InterestingTurn, opts SuggestOpts) []Candidate
Suggest turns a pool of past turns into stratified test-case candidates: top-N per category rather than top-N overall, because a set drawn purely by interestingness is all failures and represents nothing the agent normally does. Turns with no signal, and turns already saved as tasks, are dropped.
type CategoryResult ¶
type CategoryResult struct {
Category string `json:"category"`
JudgedPairs int `json:"judged_pairs"`
Wins int `json:"wins"`
Losses int `json:"losses"`
Ties int `json:"ties"`
WinRate float64 `json:"win_rate"`
// Deltas mirror the three gates, restricted to this category's tasks.
DeltaRejectedPP float64 `json:"delta_rejected_pp"`
DeltaRoundsPct float64 `json:"delta_rounds_pct"`
DeltaCostPct float64 `json:"delta_cost_pct"`
// Regressed is true when this category alone would fail a gate or fall
// below the win threshold, whatever the aggregate says.
Regressed bool `json:"regressed"`
}
CategoryResult breaks a candidate's performance down by task category. A rolled-up number hides bidirectional failures: a candidate winning big on chat while losing on tool-heavy still shows a comfortable overall win, and tool-heavy is usually what the operator actually cares about.
type Completeness ¶
type Completeness struct {
SamplesOK int `json:"samples_ok"`
SamplesExpected int `json:"samples_expected"`
Ratio float64 `json:"ratio"`
Floor float64 `json:"floor"`
Conclusive bool `json:"conclusive"`
// Pairs and PairsJudged sit next to the sample figures because a run can be
// sample-complete and still have holes in the judging grid: a (task, k)
// whose sample failed on either side yields no pair at all.
Pairs int `json:"pairs"`
PairsJudged int `json:"pairs_judged"`
}
Completeness reports how much of the run actually landed. A run that finishes below the floor still reports its numbers — partial results are the point of the capped and stopped statuses — but says they are inconclusive rather than dressing thin data as a verdict.
type Config ¶
type Config struct {
MaxConcurrent int
MaxCostPerRun float64
DefaultK int
CompletenessFloor float64
AuditMode string
}
Config is the runner's snapshot of eval, resolved once at construction.
type ConvStats ¶ added in v0.44.0
ConvStats is the slice of conversation telemetry the history basis reads.
type Engine ¶
type Engine interface {
// DryRun executes one turn under an execution policy and persists nothing.
DryRun(ctx context.Context, msg adapter.IncomingMessage, policy agent.ExecPolicy) (*agent.TurnResult, error)
// LLMRouter exposes the cost tracker (real per-sample spend) and provider
// registry (overlay validation).
LLMRouter() *llm.Router
// Name is the base agent's name, used for the audit pseudo-identity.
Name() string
}
Engine is the slice of *agent.Engine the runner needs. It exists so the package is testable with a hand-written mock: agent.Engine is a concrete struct with a large constructor.
type EngineSource ¶
EngineSource resolves a base agent name to its live engine. main.go adapts Dispatcher.Agent; the nil check there must happen *before* the value is boxed into this interface, or a typed-nil pointer reads as non-nil here.
type Estimate ¶ added in v0.44.0
type Estimate struct {
Low float64 `json:"low"`
High float64 `json:"high"`
Currency string `json:"currency"`
Basis string `json:"basis"`
// Tasks is how many tasks the figure covers — the drawn subset size when
// sample_tasks narrows the run, otherwise the whole set.
Tasks int `json:"tasks"`
K int `json:"k"`
PerVariant []VariantEstimate `json:"per_variant"`
// Note names anything that makes the figure less than a straight sum:
// a sampled subset, or tasks that could not be priced at all.
Note string `json:"note,omitempty"`
}
Estimate is a pre-run cost range in USD.
type EstimateInput ¶ added in v0.44.0
type EstimateInput struct {
Tasks []Task
Variants []EstimateVariant
K int
// SampleTasks, when set below len(Tasks), is the size of the stratified
// subset a Quick check draws. The drawn set is not known at estimate time,
// so the figure scales the mean per-task cost instead.
SampleTasks int
// BaseModel and BaseProvider are the agent's live config — the model the
// history basis actually measured, and what an empty overlay runs.
BaseModel string
BaseProvider string
}
EstimateInput is everything the estimator needs that it cannot look up.
type EstimateVariant ¶ added in v0.44.0
EstimateVariant is one side of the comparison being priced. An empty Model is the incumbent overlay: it runs the base agent's live model.
type Estimator ¶ added in v0.44.0
type Estimator struct {
Stats StatsLookup
Prices PriceLookup
}
Estimator computes a pre-run cost estimate. Both lookups are optional: a nil Stats removes the history basis, a nil Prices removes the list-price basis, and with neither the estimate is honestly unknown.
func (Estimator) Estimate ¶ added in v0.44.0
Estimate prices a run before it is created.
Per (task, variant) the basis order is history → list price → unknown. History measures the incumbent; a variant running a different model only keeps it when both models are priced and the figure can be scaled by their list-price ratio, otherwise that variant falls to list price outright.
type Gate ¶
type Gate struct {
Name string `json:"name"`
Baseline float64 `json:"baseline"`
Value float64 `json:"value"`
Delta float64 `json:"delta"`
// Threshold is the largest Delta that still passes, in Unit.
Threshold float64 `json:"threshold"`
// Unit is "pp" (percentage points, for rates) or "%" (relative change).
Unit string `json:"unit"`
Pass bool `json:"pass"`
}
Gate is one row of the objective gate table. Every verdict surface shows this table, not just the label: a bare verdict banner with no visible criteria is the black box this subsystem exists to remove.
type ImportError ¶
ImportError names the offending line so an operator hand-editing a JSONL file is told where to look, not just that something was wrong.
func (*ImportError) Error ¶
func (e *ImportError) Error() string
func (*ImportError) Unwrap ¶
func (e *ImportError) Unwrap() error
type JSONLTask ¶
type JSONLTask struct {
Prompt string `json:"prompt"`
Category string `json:"category"`
PinnedHistory json.RawMessage `json:"pinned_history,omitempty"`
Tags json.RawMessage `json:"tags,omitempty"`
Notes string `json:"notes,omitempty"`
SourceConversationID string `json:"source_conversation_id,omitempty"`
SourceMessageID *int64 `json:"source_message_id,omitempty"`
}
JSONLTask is one line of a task-set export. It is the portable shape: the row's identity and provenance ids are written for reference but ignored on import, so a set exported from one instance imports cleanly into another.
type Judge ¶ added in v0.47.0
type Judge struct {
// contains filtered or unexported fields
}
Judge grades a finished run's blinded pairs through denkeeper's own router, so a run can be judged unattended instead of only from Claude Code over MCP.
It is capability-reduced on purpose: one completion per item, no tools, no engine turn, and no reader beyond Store.GetBlindedItem. Same tables and same blinding as the MCP path — the queue is ListPending, the payload is GetBlindedItem, the write is RecordVerdict — so the two judges are interchangeable and the win rate has exactly one derivation.
func NewJudge ¶ added in v0.47.0
func NewJudge(store *Store, engines EngineSource, auditor audit.Emitter, cfg JudgeConfig, logger *slog.Logger) *Judge
NewJudge builds a judge. A zero-value Model leaves it unavailable; callers ask Available before offering it.
func (*Judge) Available ¶ added in v0.47.0
Available reports whether an internal judge is configured.
func (*Judge) Config ¶ added in v0.47.0
func (j *Judge) Config() JudgeConfig
Config returns the resolved settings, so a handler can report the model and cap a pass will run under.
func (*Judge) IsActive ¶ added in v0.47.0
IsActive reports whether a pass is currently judging this run.
func (*Judge) SetConfig ¶ added in v0.47.0
func (j *Judge) SetConfig(cfg JudgeConfig)
SetConfig applies a reloaded eval judge block, so turning the judge on, pointing it at another model, or moving its cap takes effect on the next pass instead of at the next restart. MaxConcurrent is deliberately not re-read: the semaphore is process-wide and sized once.
A pass in flight keeps the config it started under — it has already told the caller which model and cap it is running against.
func (*Judge) Shutdown ¶ added in v0.47.0
func (j *Judge) Shutdown()
Shutdown stops every pass and waits for the goroutines to finish.
func (*Judge) Start ¶ added in v0.47.0
Start launches a judging pass in the background and returns as soon as the queue is known.
Background rather than synchronous because a full run's queue is hundreds of items — 50 tasks x k=3 x two presentation orders — and no HTTP client waits that long. Progress is observable through the same figures the MCP judge's work shows up in: completeness.pairs_judged on the summary, and the pair view's per-item verdicts.
type JudgeConfig ¶ added in v0.47.0
type JudgeConfig struct {
// Model is the judging model. Empty disables the internal judge entirely.
Model string
// Provider names a registered provider instance, or is empty to use the
// base agent's own.
Provider string
// MaxCost caps one judging pass in USD.
MaxCost float64
// MaxConcurrent bounds items in flight across every pass. Fixed at
// construction — SetConfig does not resize the semaphore — because the
// point of the bound is the provider's rate limit, and a live resize would
// hand an in-flight pass more slots than the operator asked for.
MaxConcurrent int
}
JudgeConfig is the judge's snapshot of the eval judge keys.
type JudgeOpts ¶ added in v0.47.0
type JudgeOpts struct {
// SampleN draws that many pending items at random instead of taking the
// head of the queue — the calibration subset, same knob eval_pending has.
SampleN int
// Limit caps how many items the pass takes. 0 is the whole queue.
Limit int
}
JudgeOpts scopes one pass over a run's queue.
type JudgePass ¶ added in v0.47.0
type JudgePass struct {
RunID int64 `json:"run_id"`
Items int `json:"items"`
Model string `json:"model"`
Provider string `json:"provider,omitempty"`
JudgeIdent string `json:"judge_ident"`
RubricVersion string `json:"rubric_version"`
CostCap float64 `json:"cost_cap"`
}
JudgePass describes a launched pass, so the caller can report what it will cost and under which policy it is being judged.
type Judgment ¶
type Judgment struct {
Pairs int `json:"pairs"`
// JudgedPairs counts pairs whose *both* presentation orders carry a judge
// verdict. A half-judged pair is not evidence.
JudgedPairs int `json:"judged_pairs"`
Wins int `json:"wins"`
Losses int `json:"losses"`
Ties int `json:"ties"`
WinRate float64 `json:"win_rate"`
WinThreshold float64 `json:"win_threshold"`
// OperatorAgreement is nil until the operator marks a calibration item.
OperatorAgreement *Agreement `json:"operator_agreement,omitempty"`
// RubricVersions is the distinct set of rubric revisions the judge verdicts
// behind this tally were made under, sorted. A set rather than a single
// value because a queue worked across a rubric edit is a real thing to see:
// two versions here means the win-rate mixes two policies. Judges that did
// not report a version contribute nothing.
RubricVersions []string `json:"rubric_versions,omitempty"`
}
Judgment is the blinded-pair tally for one candidate against the baseline.
type JudgmentItem ¶
type JudgmentItem struct {
ID int64 `db:"id" json:"id"`
PairID int64 `db:"pair_id" json:"pair_id"`
PresentationOrder string `db:"presentation_order" json:"presentation_order"`
Status string `db:"status" json:"status"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
}
JudgmentItem is one pass over a pair at a fixed presentation order. A pair yields two, one per order.
type ModelPrice ¶ added in v0.44.0
ModelPrice is a model's list price in USD per million tokens.
type Overlay ¶
type Overlay struct {
Model string `json:"llm_model,omitempty"`
Provider string `json:"llm_provider,omitempty"`
}
Overlay is the decoded eval_variants.overlay JSON.
func DecodeOverlay ¶
DecodeOverlay parses a variant overlay. An empty or "{}" overlay is the incumbent: it runs the agent's live config unchanged.
type Pair ¶
type Pair struct {
ID int64 `db:"id" json:"id"`
RunID int64 `db:"run_id" json:"run_id"`
TaskID int64 `db:"task_id" json:"task_id"`
KIndex int `db:"k_index" json:"k_index"`
SampleA int64 `db:"sample_a" json:"sample_a"`
SampleB int64 `db:"sample_b" json:"sample_b"`
// Assignment is the letter→variant map. It is the unblinding key and must
// never reach a judge-visible payload.
Assignment string `db:"assignment" json:"assignment"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
}
Pair is one blinded comparison: a baseline sample and a candidate sample for the same (task, k), with a random A/B assignment that lives server-side only.
type PairDetail ¶ added in v0.44.0
type PairDetail struct {
PairID int64 `json:"pair_id"`
TaskID int64 `json:"task_id"`
TaskPrompt string `json:"task_prompt"`
Category string `json:"category"`
// K is the pair's sample index within the task, so a k > 1 run is readable.
K int `json:"k"`
Baseline PairSide `json:"baseline"`
Candidate PairSide `json:"candidate"`
Items []PairItem `json:"items"`
// Outcome is win/loss/tie/pending from the candidate's point of view.
Outcome string `json:"outcome"`
}
PairDetail is one pair, unblinded, with its resolved outcome.
type PairItem ¶ added in v0.44.0
type PairItem struct {
ItemID int64 `json:"item_id"`
PresentationOrder string `json:"presentation_order"`
Status string `json:"status"`
Verdicts []PairVerdict `json:"verdicts"`
}
PairItem is one presentation order of a pair with the verdicts against it.
type PairSide ¶ added in v0.44.0
type PairSide struct {
VariantID int64 `json:"variant_id"`
Variant string `json:"variant"`
SampleID int64 `json:"sample_id"`
}
PairSide is one side of a pair once the assignment is applied.
type PairVerdict ¶ added in v0.44.0
type PairVerdict struct {
// JudgeIdent names who judged. JudgeOperator marks a calibration call: it
// is listed here but never drives the pair's outcome.
JudgeIdent string `json:"judge_ident"`
// Winner is the presented letter the judge named — what it actually saw.
Winner string `json:"winner"`
// WinnerVariant is that letter resolved through the item's presentation
// order and the pair's assignment. Empty on a tie.
WinnerVariant string `json:"winner_variant,omitempty"`
// Dimensions is the stored per-dimension map, omitted when the judge
// recorded none or the stored value will not decode. Values are the
// presented letters — what the judge actually saw — and are kept so an
// audit can check the call against the queue it was answering.
Dimensions map[string]string `json:"dimensions,omitempty"`
// DimensionsVariant is Dimensions with every letter resolved through the
// item's presentation order and the pair's assignment, ties preserved as
// "tie". Same key set as Dimensions; a value that is neither a letter nor
// a tie is carried through unchanged rather than dropped.
DimensionsVariant map[string]string `json:"dimensions_variant,omitempty"`
Notes string `json:"notes,omitempty"`
RubricVersion string `json:"rubric_version,omitempty"`
CreatedAt time.Time `json:"created_at"`
}
PairVerdict is one recorded call on one item, unblinded.
type PairView ¶ added in v0.44.0
type PairView struct {
RunID int64 `json:"run_id"`
// BaselineVariant names the incumbent every pair is measured against.
BaselineVariant string `json:"baseline_variant"`
Pairs []PairDetail `json:"pairs"`
}
PairView is a run's whole judging grid, unblinded.
type PendingItem ¶
type PendingItem struct {
ItemID int64 `db:"item_id" json:"item_id"`
PairID int64 `db:"pair_id" json:"pair_id"`
RunID int64 `db:"run_id" json:"run_id"`
TaskID int64 `db:"task_id" json:"task_id"`
Category string `db:"category" json:"category"`
// Prompt is the task's own text, which is identical for both sides and so
// leaks nothing.
Prompt string `db:"prompt" json:"prompt"`
}
PendingItem is one entry of the judge's queue. It carries just enough to pick work — never the responses, which come from GetBlindedItem.
type PrecedingMessage ¶ added in v0.44.0
PrecedingMessage is one {role, content} pair of context preceding a suggested turn, ready to be pinned as a task's history.
type PriceLookup ¶ added in v0.44.0
type PriceLookup interface {
ModelPrice(ctx context.Context, provider, model string) (ModelPrice, bool)
}
PriceLookup resolves a (provider, model) pair to its list price. An empty provider means the base agent's own provider. ok=false means the price is unknown, which is a valid answer, not a failure.
type Probe ¶ added in v0.47.0
type Probe struct {
Prompt string `json:"prompt"`
Category string `json:"category"`
// Kind is the behaviour family, one of ProbeKinds().
Kind string `json:"kind"`
// Source names the written intent this probe was derived from —
// "tier:supervised", "skill:briefing", "persona:soul" — so an operator
// reading a card can go and check the spec it came from.
Source string `json:"source"`
// Notes is free-text "what good looks like", stored on the task and
// surfaced to the judge as context. Never parsed (§2, no-DSL).
Notes string `json:"notes"`
Tags []string `json:"tags"`
Preceding []PrecedingMessage `json:"preceding,omitempty"`
}
Probe is one generated test case, in the shape the accept path writes as a Task. It deliberately mirrors Candidate: the UI's accept call is the same POST either way.
func GenerateProbes ¶ added in v0.47.0
func GenerateProbes(src SpecSource, opts ProbeOpts) []Probe
GenerateProbes turns one agent's configured intent into eval tasks. It is a pure function of the spec: the same agent yields the same probes in the same order, which is what makes Exclude enough to keep a second pass quiet.
type ProbeOpts ¶ added in v0.47.0
type ProbeOpts struct {
// AutoApproveTools are the tool names the operator has pre-blessed for
// unattended execution on this agent (config + permanent scopes). They are
// policy the generator reads, not a filter: a tool *outside* the list is
// what the approval-policy probe is built around.
AutoApproveTools []string
// Limit caps the emitted probes across all families. <= 0 takes the
// default.
Limit int
// SkipKinds drops whole families before the draw. The API layer uses it to
// keep a probe pass inside the caller's own read scopes: a family derived
// from skill frontmatter is skill config, and generating from it must not
// hand it to a credential that could not read /skills directly.
SkipKinds map[string]struct{}
// Exclude holds prompts already present in the target task set, so
// regenerating against a set that already carries probes offers only the
// new ones. Generation is deterministic for a given spec, so without this
// a second pass would offer the whole set again.
Exclude map[string]struct{}
}
ProbeOpts bounds a generation pass.
type ProgressEvent ¶
type ProgressEvent struct {
RunID int64 `json:"run_id"`
Status string `json:"status"`
SamplesDone int `json:"samples_done"`
SamplesTotal int `json:"samples_total"`
CostSpent float64 `json:"cost_spent"`
CostCap float64 `json:"cost_cap"`
ETASeconds int `json:"eta_seconds,omitempty"`
}
ProgressEvent is emitted after every sample and at both ends of a run. It is deliberately droppable: main.go forwards it to the WebSocket hub, and GET /eval/runs/{id} is the authoritative fallback.
type Run ¶
type Run struct {
ID int64 `db:"id" json:"id"`
TaskSetID int64 `db:"task_set_id" json:"task_set_id"`
BaseAgent string `db:"base_agent" json:"base_agent"`
Status string `db:"status" json:"status"`
K int `db:"k" json:"k"`
CostCap float64 `db:"cost_cap" json:"cost_cap"`
CostSpent float64 `db:"cost_spent" json:"cost_spent"`
// JudgeCost is what the internal judge has spent grading this run, kept
// apart from CostSpent: it is a separate budget spent by a separate
// decision, and adding it to the sample spend would read as a blown cap.
JudgeCost float64 `db:"judge_cost" json:"judge_cost"`
AsOf time.Time `db:"as_of" json:"as_of"`
// TaskIDs pins the run's task list at creation; nil means the whole set.
// Pinning is what makes a sampled subset possible, and it also stops a task
// added to the set later from retroactively inflating samples_expected and
// flipping a finished run to inconclusive.
TaskIDs TaskIDList `db:"task_ids" json:"task_ids,omitempty"`
// TaskCount is how many tasks the run covers: the pinned count when pinned,
// otherwise the set's current size. It is a display figure — the
// authoritative dispatch count is len(RunTasks), which also drops a pinned
// task deleted after the run was created.
TaskCount int `db:"task_count" json:"task_count"`
Error string `db:"error" json:"error,omitempty"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
FinishedAt *time.Time `db:"finished_at" json:"finished_at,omitempty"`
}
Run is one comparison run.
type Runner ¶
type Runner struct {
// OnProgress is called after each sample and at run start/finish. Nil-safe.
OnProgress func(ProgressEvent)
// contains filtered or unexported fields
}
Runner executes eval runs in the background against live engines.
Execution vehicle: samples run on the agent's *live* engine via Engine.DryRun with an ExecEval policy, not on a per-run rebuilt engine. The unit of evaluation is model-in-harness, so the sample must see the agent's real skills, tools, persona and auditor; a capability-reduced or duplicated engine would measure a different system. Isolation comes from ExecPolicy (structural, not filtered) and the variant's router is a per-turn clone, so nothing about the live engine is mutated.
func NewRunner ¶
func NewRunner(store *Store, engines EngineSource, auditor audit.Emitter, cfg Config, logger *slog.Logger) *Runner
NewRunner builds a runner. It starts no goroutine: a user who never launches a run pays nothing beyond the five empty tables.
func (*Runner) Config ¶
Config returns the runner's resolved settings, so handlers can report the defaults a run was created against.
func (*Runner) Shutdown ¶
func (r *Runner) Shutdown()
Shutdown stops every run and waits for the goroutines to finish.
func (*Runner) StartRun ¶
StartRun launches a pending run in the background. It returns as soon as the run is registered; progress is observable through the store and OnProgress.
type Sample ¶
type Sample struct {
ID int64 `db:"id" json:"id"`
RunID int64 `db:"run_id" json:"run_id"`
VariantID int64 `db:"variant_id" json:"variant_id"`
TaskID int64 `db:"task_id" json:"task_id"`
KIndex int `db:"k_index" json:"k_index"`
Status string `db:"status" json:"status"`
Error string `db:"error" json:"error,omitempty"`
Response string `db:"response" json:"response"`
Trace string `db:"trace" json:"trace"`
Rounds int `db:"rounds" json:"rounds"`
StopReason string `db:"stop_reason" json:"stop_reason,omitempty"`
// Upstream is the provider-reported serving upstream (OpenRouter's routed
// provider), empty for providers without the concept.
Upstream string `db:"upstream" json:"upstream,omitempty"`
// Outcome counts are tool-call level, split exactly as
// agent.ToolCallRecord.Outcome. Cached and suppressed are kept separate
// from failed on purpose: folding either in would poison the failed-rate
// gate, and an eval turn suppresses writes routinely.
OutcomeOK int `db:"outcome_ok" json:"outcome_ok"`
OutcomeRejected int `db:"outcome_rejected" json:"outcome_rejected"`
OutcomeFailed int `db:"outcome_failed" json:"outcome_failed"`
OutcomeDenied int `db:"outcome_denied" json:"outcome_denied"`
OutcomeCached int `db:"outcome_cached" json:"outcome_cached"`
OutcomeSuppressed int `db:"outcome_suppressed" json:"outcome_suppressed"`
TokensPrompt int `db:"tokens_prompt" json:"tokens_prompt"`
TokensCompletion int `db:"tokens_completion" json:"tokens_completion"`
Cost float64 `db:"cost" json:"cost"`
LatencyMs int64 `db:"latency_ms" json:"latency_ms"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
}
Sample is one (task, variant, k) execution.
type SpecSource ¶ added in v0.47.0
type SpecSource interface {
Name() string
PermissionTier() string
ToolNames() []string
Skills() []skill.Skill
PersonaSections() map[string]bool
PersonaSection(section string) (content string, editable bool, agentMutable bool, ok bool)
}
SpecSource is the slice of a configured agent the probe generator reads: its written intent, and nothing else. *agent.Engine satisfies it as it stands, so the API layer hands over the live engine rather than the generator reaching into a store — the same narrowing agent.InterestingTurnStore does for Suggest, for the same reason (no widening of a shared interface, no hand-written mock to update).
type StatsLookup ¶ added in v0.44.0
type StatsLookup interface {
ConversationStats(ctx context.Context, convID string) (*ConvStats, error)
}
StatsLookup resolves a task's source conversation to its aggregate telemetry. A nil result means the conversation carries no stats — pruned by retention, cleared, or never real — and is not an error.
type Store ¶
type Store struct {
// contains filtered or unexported fields
}
Store persists eval task sets, runs and samples. It owns its own handle on the main database file, following the kv package: the eval tables live in the main DB (§5) because every run reads eval_tasks, and the write rate is far below anything that would justify a separate file.
func NewInMemoryStore ¶
NewInMemoryStore creates an in-memory SQLite store for testing.
func NewSQLiteStore ¶
NewSQLiteStore opens or creates the database and applies the eval schema.
func (*Store) AddJudgeCost ¶ added in v0.47.0
AddJudgeCost accumulates internal-judge spend on a run. Written per item for the same reason sample cost is: a process that dies mid-pass still leaves an honest figure behind.
func (*Store) AddRunCost ¶
AddRunCost accumulates spend on a run. Called after every sample so a crashed process still leaves an honest figure behind.
func (*Store) CountPairs ¶
CountPairs returns how many pairs a run has.
func (*Store) CountSamples ¶
CountSamples returns how many samples a run has recorded so far.
func (*Store) CountTraces ¶ added in v0.47.0
CountTraces returns how many traces match the filter, so a list view can say whether there is more behind the page it is showing. It takes the same filter as ListTraces deliberately: an unfiltered count beside a filtered page makes "load more" ask for rows that do not exist and never terminate.
func (*Store) CreatePairs ¶
CreatePairs is the run-finalization step that turns completed samples into blinded judgment work. It returns how many pairs it created.
It runs for capped and stopped runs as well as done ones, since partial results are the whole point of those statuses — a run that spent real money before hitting its cap should still be judgeable on what it produced. A (task, k) whose sample is missing or failed on either side yields no pair; the count is reported next to the completeness figure so a reader can see how much of the grid survived.
Pairing policy for more than two variants: every non-baseline variant is paired against the baseline (the first variant by creation order, the same convention per-task deltas use). N−1 pair sets rather than a round-robin, because the question the decision rule answers is "is this candidate an upgrade on the incumbent", and no consumer reads candidate-vs-candidate.
Idempotent by guard: a run that already has pairs is left alone, so a retried finalization cannot double the queue.
func (*Store) CreateRun ¶
func (s *Store) CreateRun(ctx context.Context, run Run, variants []Variant) (*Run, []Variant, error)
CreateRun inserts a run in the pending status together with its variants, in one transaction: a run without variants has nothing to compare and must never be visible.
func (*Store) CreateTaskSet ¶
CreateTaskSet inserts a task set. A duplicate name returns ErrNameTaken.
func (*Store) DeleteTask ¶
DeleteTask removes one task from a set.
func (*Store) DeleteTaskSet ¶
DeleteTaskSet removes a set and its tasks. It refuses with ErrTaskSetInUse when any run references the set: a run's samples are only interpretable against the tasks that produced them.
func (*Store) ExportJSONL ¶
ExportJSONL writes one task per line.
func (*Store) FinishRun ¶
FinishRun writes the terminal status, error and finish time in one update.
func (*Store) GetBlindedItem ¶
GetBlindedItem builds the judge-visible payload for one item.
The payload is constructed from scratch rather than filtered out of the stored rows: blinding that works by removing fields fails open the day a column is added, and the whole judge path depends on it failing closed.
func (*Store) GetTask ¶
GetTask returns one task by id, scoped to a set so a caller cannot address another set's task through a set-scoped route.
func (*Store) GetTaskSet ¶
GetTaskSet returns a task set by name.
func (*Store) GetTaskSetByID ¶
GetTaskSetByID returns a task set by id.
func (*Store) ImportJSONL ¶
ImportJSONL appends every line of r to a set. It is all-or-none: every line is parsed and validated before anything is written, so a typo halfway down a hand-edited file leaves the set exactly as it was rather than half-imported.
func (*Store) ListPending ¶
func (s *Store) ListPending(ctx context.Context, runID int64, limit, sampleN int) ([]PendingItem, error)
ListPending returns pending judgment items, optionally scoped to one run (0 = every run). sampleN > 0 draws a random subset instead of the head of the queue: the interactive calibration pass judges ~20 items, and taking the first 20 would calibrate against whichever tasks happen to sort first.
func (*Store) ListRuns ¶
ListRuns returns runs newest first, optionally filtered by task set id (0 = any) and status ("" = any).
func (*Store) ListSamples ¶
ListSamples returns a run's samples in insertion order.
func (*Store) ListTaskSamples ¶ added in v0.47.0
ListTaskSamples returns one task's samples within a run. The results view expands one test case at a time, and a full run's samples carry a trace each — fetching all of them to render one row is the whole reason this exists.
func (*Store) ListTaskSets ¶
ListTaskSets returns every task set with its task count, ordered by name.
func (*Store) ListTasks ¶
ListTasks returns the tasks of a set, in creation order — the order the runner dispatches them and the order per-task deltas are baselined against.
func (*Store) ListTraces ¶ added in v0.47.0
ListTraces returns trace headers newest first, without payloads.
func (*Store) ListVariants ¶
ListVariants returns a run's variants in creation order. The first is the per-task delta baseline (convention: the incumbent is created first).
func (*Store) ListVerdicts ¶
ListVerdicts returns every verdict recorded against a run's items.
func (*Store) PairDetails ¶ added in v0.44.0
PairDetails returns a run's pairs with their verdicts and resolved outcomes, optionally narrowed to one task (taskID 0 = every task).
Outcomes come from resolvePairs, the same resolver the win-rate is tallied from, so a pair this view calls a tie is a tie in the verdict too. Deriving the both-orders-must-agree rule a second time here is exactly how the two views would drift apart.
func (*Store) PruneTracesBefore ¶ added in v0.47.0
PruneTracesBefore deletes traces created before the cutoff and reports how many went. Traces carry their own eval retention_days (30 by default, matching audit) because they are the most sensitive rows in the database and keeping them for the telemetry window would be a different decision than the one the operator made when they turned capture on.
func (*Store) RecordVerdict ¶
RecordVerdict writes one judge's call on one item and marks the item judged.
The operator's calibration marks deliberately do *not* flip the status: they are recorded against an item the judge has already worked, and an item that only the operator has seen is still outstanding judge work.
func (*Store) RunTasks ¶ added in v0.44.0
RunTasks returns the tasks a run covers: its pinned list, or the whole set when it has none. Every reader that needs "what does this run run" goes through here — the runner, the progress figures and the summary's samples_expected — so the three can never disagree about the denominator.
A pinned task deleted after the run was created is skipped rather than erroring: eval_tasks has no delete guard, the samples it already produced stay readable, and an expected count that no longer matches what the runner can dispatch would be the worse failure. Order is always creation order, the order the baseline convention and the per-task deltas are read in, whatever order the pin was written in.
func (*Store) SaveTrace ¶ added in v0.47.0
SaveTrace persists one captured turn. It is the agent.TraceSink implementation, so the engine can record a live turn without importing this package.
An unencodable payload is an error rather than an empty row: a trace whose blob silently vanished is worse than no trace, since the inspector would show a turn with no prompt and no reason why.
func (*Store) SavedTaskSources ¶ added in v0.44.0
SavedTaskSources returns the SourceKey of every turn already saved as a task, across all sets. The suggestion endpoint subtracts it so an accepted candidate does not come back next time. Tasks with no source message (hand written, imported) contribute nothing — there is no turn to suppress.
func (*Store) SetRunStatus ¶
SetRunStatus updates a run's status and error text.
func (*Store) Summarize ¶
Summarize aggregates a run's samples in Go rather than SQL. A run holds at most a few hundred samples, and the arithmetic is far easier to test as ordinary code than as window functions.
type SuggestOpts ¶ added in v0.44.0
type SuggestOpts struct {
// Limit is the total number of candidates returned across all categories.
Limit int
// Exclude holds SourceKey values for turns already saved as tasks, so an
// accepted suggestion does not resurface.
Exclude map[string]struct{}
}
SuggestOpts bounds a suggestion pass.
type Summary ¶
type Summary struct {
RunID int64 `json:"run_id"`
Status string `json:"status"`
BaseAgent string `json:"base_agent"`
TaskSet string `json:"task_set"`
K int `json:"k"`
CostCap float64 `json:"cost_cap"`
CostSpent float64 `json:"cost_spent"`
// BaselineVariant names the variant per-task deltas are measured against:
// the first by creation order. This is a convention (the incumbent is
// created first), not something the API enforces.
BaselineVariant string `json:"baseline_variant"`
Variants []VariantMetrics `json:"variants"`
PerTask []TaskMetrics `json:"per_task"`
Completeness Completeness `json:"completeness"`
// Verdicts holds one decision per non-baseline variant, each with its gate
// table, judge tally and per-category breakdown. Present even before any
// judging: the objective half alone can already say "downgrade" or "no
// regressions detected".
Verdicts []VariantVerdict `json:"verdicts"`
}
Summary is the objective scorecard for one run — everything computable without a judge.
type SummaryOpts ¶
type SummaryOpts struct {
CompletenessFloor float64
// WinThreshold is the judge win-rate a candidate must reach to be called an
// upgrade.
WinThreshold float64
// GateRejectedPP is the largest tolerated rise in rejected tool-call rate,
// in percentage points.
GateRejectedPP float64
// GateRoundsPct and GateCostPct are the largest tolerated relative rises in
// mean rounds and cost per task.
GateRoundsPct float64
GateCostPct float64
}
SummaryOpts carries the eval policy a summary is computed against. Thresholds are configuration, not constants: what counts as a regression is the operator's call.
type Task ¶
type Task struct {
ID int64 `db:"id" json:"id"`
SetID int64 `db:"set_id" json:"set_id"`
Prompt string `db:"prompt" json:"prompt"`
Category string `db:"category" json:"category"`
// PinnedHistory is a JSON array of {role, content} replayed verbatim as
// the context preceding the turn. NULL/empty means a fresh turn.
PinnedHistory string `db:"pinned_history" json:"pinned_history,omitempty"`
SourceConversationID string `db:"source_conversation_id" json:"source_conversation_id,omitempty"`
SourceMessageID *int64 `db:"source_message_id" json:"source_message_id,omitempty"`
Tags string `db:"tags" json:"tags"`
Notes string `db:"notes" json:"notes"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
}
Task is one saved test case.
type TaskIDList ¶ added in v0.44.0
type TaskIDList []int64
TaskIDList is eval_runs.task_ids: a JSON array of task ids, or NULL for the whole set. It carries its own SQL codec so a run row round-trips without the callers ever handling the raw JSON.
func DrawStratified ¶ added in v0.44.0
func DrawStratified(tasks []Task, n int) TaskIDList
DrawStratified picks n task ids spread across the curation categories, for the "quick check" shape: a cheap first signal that still touches every kind of turn the set covers.
The draw cycles the categories in their canonical order, taking one random unpicked task from each pass and skipping exhausted ones, so category counts differ by at most one wherever the set allows. Ranking by anything else — or drawing uniformly — would let a set that is 70 % chat produce a subset that is all chat, which is the one thing a stratified sample exists to prevent.
n <= 0 or n >= len(tasks) returns nil, meaning "the whole set": the pin is only worth recording when it actually narrows something, and a nil pin is what every reader treats as unpinned.
Randomness comes from randIndex (crypto/rand), like the blinding coin — the package deliberately has one source rather than a security-grade one next to a seeded one, and there is no seed knob to make a draw reproducible.
func (*TaskIDList) Scan ¶ added in v0.44.0
func (l *TaskIDList) Scan(src any) error
Scan decodes the stored JSON. NULL and the empty string both mean "not pinned", which is the same thing a caller sees as a nil slice.
type TaskMetrics ¶
type TaskMetrics struct {
TaskID int64 `json:"task_id"`
Prompt string `json:"prompt"`
Category string `json:"category"`
Variants []TaskVariantMetrics `json:"variants"`
}
TaskMetrics groups one task's per-variant cells.
type TaskPatch ¶
type TaskPatch struct {
Prompt *string
Category *string
PinnedHistory *string
Tags *string
Notes *string
}
TaskPatch carries the mutable task fields; a nil field is left unchanged.
type TaskSet ¶
type TaskSet struct {
ID int64 `db:"id" json:"id"`
Name string `db:"name" json:"name"`
Description string `db:"description" json:"description"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
// TaskCount is populated by ListTaskSets; it is not a column.
TaskCount int `db:"task_count" json:"task_count"`
}
TaskSet is a named collection of eval tasks.
type TaskVariantMetrics ¶
type TaskVariantMetrics struct {
VariantID int64 `json:"variant_id"`
Name string `json:"name"`
SamplesOK int `json:"samples_ok"`
MeanCost float64 `json:"mean_cost"`
MeanRounds float64 `json:"mean_rounds"`
MeanLatency float64 `json:"mean_latency_ms"`
// Deltas are against the baseline variant and are zero on the baseline
// row itself.
DeltaCost float64 `json:"delta_cost"`
DeltaRounds float64 `json:"delta_rounds"`
DeltaLatency float64 `json:"delta_latency_ms"`
}
TaskVariantMetrics is one cell of the per-task breakdown.
type TraceFilter ¶ added in v0.47.0
type TraceFilter struct {
Agent string
ConversationID string
Source string
Since time.Time
Until time.Time
Limit int
Offset int
}
TraceFilter narrows a trace listing. A zero value lists the newest traces.
type TraceRow ¶ added in v0.47.0
type TraceRow struct {
ID int64 `db:"id" json:"id"`
Agent string `db:"agent" json:"agent"`
ConversationID string `db:"conversation_id" json:"conversation_id"`
Source string `db:"source" json:"source"`
Model string `db:"model" json:"model,omitempty"`
Provider string `db:"provider" json:"provider,omitempty"`
RequestedModel string `db:"requested_model" json:"requested_model,omitempty"`
Upstream string `db:"upstream" json:"upstream,omitempty"`
Rounds int `db:"rounds" json:"rounds"`
StopReason string `db:"stop_reason" json:"stop_reason,omitempty"`
TokensPrompt int `db:"tokens_prompt" json:"tokens_prompt"`
TokensCompletion int `db:"tokens_completion" json:"tokens_completion"`
TokensCached int `db:"tokens_cached" json:"tokens_cached"`
TokensTotal int `db:"tokens_total" json:"tokens_total"`
Cost float64 `db:"cost" json:"cost_usd"`
LatencyMs int64 `db:"latency_ms" json:"latency_ms"`
Truncated bool `db:"truncated" json:"truncated"`
Bytes int `db:"bytes" json:"bytes"`
Payload string `db:"payload" json:"-"`
StartedAt time.Time `db:"started_at" json:"started_at"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
}
TraceRow is one stored trace. Payload is the raw JSON blob; a list read leaves it empty rather than hauling a quarter of a megabyte per row into a page that only renders the header line.
type Variant ¶
type Variant struct {
ID int64 `db:"id" json:"id"`
RunID int64 `db:"run_id" json:"run_id"`
Name string `db:"name" json:"name"`
Overlay string `db:"overlay" json:"overlay"`
}
Variant is one side of a comparison: a named overlay on the base agent's live config. An empty overlay is the incumbent.
type VariantEstimate ¶ added in v0.44.0
type VariantEstimate struct {
Name string `json:"name"`
Low float64 `json:"low"`
High float64 `json:"high"`
Basis string `json:"basis"`
}
VariantEstimate is one variant's share of the range.
type VariantMetrics ¶
type VariantMetrics struct {
VariantID int64 `json:"variant_id"`
Name string `json:"name"`
Overlay Overlay `json:"overlay"`
// RejectedRate and FailedRate are tool-call level: the denominator is
// ok+rejected+failed+denied. Cached and suppressed calls are excluded
// because nothing executed, so counting them would dilute both rates with
// non-events.
RejectedRate float64 `json:"rejected_rate"`
FailedRate float64 `json:"failed_rate"`
// ToolCalls is that denominator, so a reader can tell a 0 % rate over 200
// calls from a 0 % rate over none.
ToolCalls int `json:"tool_calls"`
MeanRounds float64 `json:"mean_rounds"`
// WrapupCount is samples whose loop was cut short by repeated identical
// calls or by exhausting the round budget — the "flaily" signal.
WrapupCount int `json:"wrapup_count"`
MeanCostPerTask float64 `json:"mean_cost_per_task"`
MeanLatencyMs float64 `json:"mean_latency_ms"`
TotalCost float64 `json:"total_cost"`
SamplesOK int `json:"samples_ok"`
SamplesFailed int `json:"samples_failed"`
}
VariantMetrics is one variant's objective scorecard, computed over its status-ok samples.
type VariantVerdict ¶
type VariantVerdict struct {
VariantID int64 `json:"variant_id"`
Variant string `json:"variant"`
Baseline string `json:"baseline"`
Verdict string `json:"verdict"`
// Reason is the one-line plain-language explanation, e.g. "downgrade: mean
// rounds regressed +35% against a +20% threshold".
Reason string `json:"reason"`
Gates []Gate `json:"gates"`
Judgment Judgment `json:"judgment"`
Categories []CategoryResult `json:"categories"`
// Divergence is set when the aggregate and a category disagree, e.g. "wins
// overall; regresses on tool_heavy". v1 surfaces this prominently without
// gating on it.
Divergence string `json:"divergence,omitempty"`
}
VariantVerdict is the decision for one candidate variant against the run's baseline, with the work shown.
type Verdict ¶
type Verdict struct {
ID int64 `db:"id" json:"id"`
ItemID int64 `db:"item_id" json:"item_id"`
Winner string `db:"winner" json:"winner"`
// Dimensions is a JSON object of dimension → winner (a/b/tie), the same
// pairwise form as Winner rather than absolute scores: a judge comparing
// two responses is reliable, a judge scoring one in isolation is not.
Dimensions string `db:"dimensions" json:"dimensions"`
Notes string `db:"notes" json:"notes"`
// JudgeIdent names who judged — an API key name, or JudgeOperator for the
// operator's calibration marks.
JudgeIdent string `db:"judge_ident" json:"judge_ident"`
// RubricVersion is the judging rubric revision this call was made under, as
// the judge reported it (the `Rubric version:` line of the judge-eval
// skill). Empty when the judge did not say, which is why the results view
// reports the distinct set rather than assuming one.
RubricVersion string `db:"rubric_version" json:"rubric_version,omitempty"`
CreatedAt time.Time `db:"created_at" json:"created_at"`
}
Verdict is one judge's call on one item, expressed in presented letters.