Documentation
¶
Index ¶
- Constants
- func LearnedSkillsDir(workDir string) string
- func ParseFeedbackRating(s string) (string, bool)
- func UCBScore(avgReward float64, totalSessions, timesUsed int, explorationC float64) float64
- func ValidSkillStatus(s string) bool
- type BackgroundState
- type Breakdown
- type ComponentScore
- type ContextProfile
- type ContextTrimmer
- type DailyMetric
- type DiagnoseOptions
- type DoctorBackground
- type DoctorError
- type DoctorJudge
- type DoctorSkills
- type DoctorVariants
- type EvaluateOptions
- type EvaluatorService
- func (s *EvaluatorService) BackgroundState() BackgroundState
- func (s *EvaluatorService) ClassifyTask(text string) string
- func (s *EvaluatorService) Diagnose(ctx context.Context) (*Report, error)
- func (s *EvaluatorService) EvaluateNow(ctx context.Context, sessionID string, opts EvaluateOptions) (*Result, error)
- func (s *EvaluatorService) EvaluateSession(ctx context.Context, sessionID string) error
- func (s *EvaluatorService) Flush(ctx context.Context) error
- func (s *EvaluatorService) GetActiveSkills(ctx context.Context, taskType string) ([]Skill, error)
- func (s *EvaluatorService) GetStats(ctx context.Context) (*Stats, error)
- func (s *EvaluatorService) IsEnabled() bool
- func (s *EvaluatorService) LastEvaluationError() *JudgeError
- func (s *EvaluatorService) LastJudgeError() *JudgeError
- func (s *EvaluatorService) ListSkills(ctx context.Context, status, taskType string) ([]Skill, error)
- func (s *EvaluatorService) MarkCompleted(ctx context.Context, sessionID, reason string) error
- func (s *EvaluatorService) NewContextTrimmer() *ContextTrimmer
- func (s *EvaluatorService) RecordFeedback(ctx context.Context, sessionID, rating, note string) (*Result, error)
- func (s *EvaluatorService) ReviewSkill(ctx context.Context, id, status string) (*Skill, error)
- func (s *EvaluatorService) RunBackground(ctx context.Context, isPrimary func() bool)
- func (s *EvaluatorService) SelectVariant(ctx context.Context, sessionID, section string, candidates []string) (string, error)
- func (s *EvaluatorService) SessionSkills(ctx context.Context, sessionID, taskType string) ([]Skill, error)
- func (s *EvaluatorService) SetFirstPassHook(fn func(ctx context.Context))
- func (s *EvaluatorService) SetWorkDir(dir string)
- func (s *EvaluatorService) Sweep(ctx context.Context, opts SweepOptions) (int, error)
- func (s *EvaluatorService) SyncLearnedSkills(ctx context.Context) error
- func (s *EvaluatorService) TemplatesEnabled() bool
- type Feedback
- type Judge
- type JudgeError
- type JudgeMeta
- type JudgeOutput
- type JudgeResult
- type PatternHit
- type PatternIssue
- type Report
- type Result
- type RewardResult
- type SectionVariants
- type Service
- type SessionDetail
- type SessionSkillRef
- type SessionVariant
- type Skill
- type SkillFile
- type Stats
- type SweepOptions
- type TemplateStats
- type ToolInfo
Constants ¶
const ( PatternIssueCompile = "compile" PatternIssueDoubleBackslash = "double_backslash" )
Pattern lint kinds reported by Diagnose.
const ( SkillStatusPending = "pending" SkillStatusApproved = "approved" SkillStatusRejected = "rejected" )
Skill review states.
const ( ComponentSuccess = "success" ComponentTokens = "tokens" ComponentToolErrors = "toolErrors" ComponentCancels = "cancels" ComponentRepetition = "repetition" ComponentTurns = "turns" ComponentEndState = "endState" // ComponentFeedback is informational: explicit feedback overrides the total. ComponentFeedback = "feedback" )
Reward components. Every component is a score in [0,1] where 1 is best.
const ( FeedbackGood = "good" FeedbackBad = "bad" )
Explicit feedback bounds: bad feedback forces total < 0.3, good > 0.8.
const ContextProfileKey contextKey = "context_profile"
ContextProfileKey is used to store a ContextProfile in the request context.
Variables ¶
This section is empty.
Functions ¶
func LearnedSkillsDir ¶ added in v1.2.4
LearnedSkillsDir returns the directory holding the learned-skill files.
func ParseFeedbackRating ¶ added in v1.2.4
ParseFeedbackRating normalises user input ("good", "bad", "+", "up", ...).
func UCBScore ¶
UCBScore computes the UCB1 value for a template. UCB_i = avg_reward_i + c * sqrt(ln(N) / n_i) where N = total sessions evaluated, n_i = times template i was used.
func ValidSkillStatus ¶ added in v1.2.4
ValidSkillStatus reports whether s is a skill review state.
Types ¶
type BackgroundState ¶ added in v1.2.4
type BackgroundState struct {
// Started is true once RunBackground was called (the goroutine exists).
Started bool `json:"started"`
// Primary is whether this instance owned the DB writer at the last tick.
Primary bool `json:"primary"`
// BackfillDone is true once the one-shot startup backfill ran.
BackfillDone bool `json:"backfill_done"`
// BackfillEvaluated is how many sessions the startup backfill scored.
BackfillEvaluated int `json:"backfill_evaluated"`
// LastSweepAt is the unix time of the last idle sweep (0 = none yet).
LastSweepAt int64 `json:"last_sweep_at"`
// LastSweepEvaluated is how many sessions the last idle sweep scored.
LastSweepEvaluated int `json:"last_sweep_evaluated"`
}
BackgroundState describes the idle sweeper / startup backfill of this process.
type Breakdown ¶ added in v1.2.4
type Breakdown struct {
// Components holds the score of every component that was available.
Components map[string]float64 `json:"components"`
// Weights holds the weight applied to each available component.
Weights map[string]float64 `json:"weights"`
// PatternHits are the user turns flagged as corrections.
PatternHits []PatternHit `json:"patternHits,omitempty"`
// Feedback is the explicit user rating ("good"/"bad") when present.
Feedback string `json:"feedback,omitempty"`
FeedbackNote string `json:"feedbackNote,omitempty"`
// Raw counters behind the components.
UserTurns int `json:"userTurns"`
ToolCalls int `json:"toolCalls"`
ToolErrors int `json:"toolErrors"`
Cancels int `json:"cancels"`
Repeats int `json:"repeats"`
// Baseline is the mean token count of recent sessions (0 when unknown).
Baseline float64 `json:"baseline,omitempty"`
// WeightedTotal is the total before any explicit-feedback override.
WeightedTotal float64 `json:"weightedTotal"`
}
Breakdown is the persisted, explainable decomposition of a reward. It is stored as JSON in session_scores.components.
type ComponentScore ¶ added in v1.2.4
ComponentScore is one reward component of a session.
type ContextProfile ¶ added in v0.407.0
type ContextProfile struct {
// TaskType is the classified task type (e.g. "code", "debug", "refactor").
TaskType string
// RelevantToolNames lists the tool names to keep for this task.
// An empty slice means keep all tools (no filtering).
RelevantToolNames []string
// SkipSections lists prompt section names that are not needed for this task.
SkipSections []string
// Confidence is a 0.0–1.0 measure of the trimmer's certainty.
// Profiles with Confidence < 0.5 should be ignored and defaults used.
Confidence float64
}
ContextProfile is produced by the ContextTrimmer at session start. It guides the PromptBuilder and agent tool assembly to include only what is relevant for the current task.
type ContextTrimmer ¶ added in v0.407.0
type ContextTrimmer struct {
// contains filtered or unexported fields
}
ContextTrimmer uses a cheap LLM to produce a ContextProfile at session start. It complements the post-session Judge: instead of "what went wrong?", it asks "what is needed for this task?".
func NewContextTrimmer ¶ added in v0.407.0
func NewContextTrimmer(cfg config.EvaluatorConfig, j *Judge) *ContextTrimmer
NewContextTrimmer creates a ContextTrimmer that reuses the evaluator's judge infrastructure. Returns nil if the evaluator has no judge configured.
func (*ContextTrimmer) MinConfidence ¶ added in v1.2.4
func (ct *ContextTrimmer) MinConfidence() float64
MinConfidence returns the confidence below which callers must ignore a profile.
func (*ContextTrimmer) ProfileTask ¶ added in v0.407.0
func (ct *ContextTrimmer) ProfileTask( ctx context.Context, firstMessage string, availableTools []ToolInfo, ) (*ContextProfile, error)
ProfileTask analyzes the user's first message and returns a ContextProfile describing which tools and prompt sections are relevant for the task.
The LLM call is bounded by a 3-second timeout; on timeout or error the method returns nil so callers can fall back to defaults. Results are cached by a hash of (firstMessage + toolNames) to avoid redundant LLM calls.
type DailyMetric ¶ added in v1.2.4
type DailyMetric struct {
Day string `json:"day"`
Evaluations int64 `json:"evaluations"`
AvgReward float64 `json:"avg_reward"`
JudgeCalls int64 `json:"judge_calls"`
JudgePromptTokens int64 `json:"judge_prompt_tokens"`
JudgeCompletionTokens int64 `json:"judge_completion_tokens"`
}
DailyMetric is one local day of evaluator activity.
func DailyMetrics ¶ added in v1.2.4
func DailyMetrics(ctx context.Context, q db.Querier, days int, now time.Time) ([]DailyMetric, error)
DailyMetrics returns one entry per local day for the last days days (today included), zero-filled so a chart has a continuous axis. Task type is not stored with a score (only the judge output carries one), so the mean reward is per day, not per task type.
type DiagnoseOptions ¶ added in v1.2.4
type DiagnoseOptions struct {
// Config is the evaluator config as loaded (not defaulted).
Config config.EvaluatorConfig
// DB reads sessions and scores; nil skips the database checks.
DB db.Querier
// Service is the running evaluator, when there is one in this process. It
// supplies the in-memory failures and the background worker state.
Service *EvaluatorService
// WorkDir is the project directory (variants, learned skills).
WorkDir string
// Now overrides the clock (tests).
Now time.Time
}
DiagnoseOptions are the inputs of Diagnose.
type DoctorBackground ¶ added in v1.2.4
type DoctorBackground struct {
IdleTimeout string `json:"idle_timeout"`
BackfillLimit int `json:"backfill_limit"`
IncludeSubagents bool `json:"include_subagents"`
// Known is true when the report came from a process running the workers.
Known bool `json:"known"`
BackgroundState
}
DoctorBackground describes the idle sweeper and backfill.
type DoctorError ¶ added in v1.2.4
DoctorError is a timestamped failure recorded in memory by the service.
type DoctorJudge ¶ added in v1.2.4
type DoctorJudge struct {
// Configured is true when a judge model is set and the judge initialised
// (or, without a service, when a model is set).
Configured bool `json:"configured"`
HighReward float64 `json:"high_reward"`
LowReward float64 `json:"low_reward"`
MinTurns int `json:"min_turns"`
DailyCalls int `json:"daily_calls"`
DailyTokens int64 `json:"daily_tokens"`
CallsToday int64 `json:"calls_today"`
TokensToday int64 `json:"tokens_today"`
BudgetExhausted bool `json:"budget_exhausted"`
LastError *DoctorError `json:"last_error,omitempty"`
BackfillJudge bool `json:"backfill_judge"`
MaxTranscriptTokens int `json:"max_transcript_tokens"`
}
DoctorJudge is the judge part of the report.
type DoctorSkills ¶ added in v1.2.4
type DoctorSkills struct {
Directory string `json:"directory"`
Pending int `json:"pending"`
Approved int `json:"approved"`
Rejected int `json:"rejected"`
}
DoctorSkills counts the learned skills by review state.
type DoctorVariants ¶ added in v1.2.4
type DoctorVariants struct {
Enabled bool `json:"enabled"`
// Directories are the variant roots that exist on disk.
Directories []string `json:"directories"`
Sections []SectionVariants `json:"sections"`
// Competing counts sections with two or more candidates.
Competing int `json:"competing"`
}
DoctorVariants is the template-variant part of the report.
type EvaluateOptions ¶ added in v1.2.4
type EvaluateOptions struct {
// Force bypasses the completion guards (minimum user turns, subagent
// sessions). Idempotency is never bypassed.
Force bool
// SkipJudge disables the LLM judge for this evaluation.
SkipJudge bool
// Rescore replaces an existing session score in place instead of skipping
// the session as already evaluated (used after explicit feedback). The
// judge never runs on a re-score.
Rescore bool
}
EvaluateOptions tunes a single evaluation.
type EvaluatorService ¶
type EvaluatorService struct {
// contains filtered or unexported fields
}
EvaluatorService is the concrete implementation of Service.
func New ¶
func New(cfg config.EvaluatorConfig, q db.Querier, msgs message.Service) (*EvaluatorService, error)
New creates a new EvaluatorService. Returns nil if disabled.
func (*EvaluatorService) BackgroundState ¶ added in v1.2.4
func (s *EvaluatorService) BackgroundState() BackgroundState
BackgroundState returns a snapshot of the background worker state.
func (*EvaluatorService) ClassifyTask ¶ added in v0.407.0
func (s *EvaluatorService) ClassifyTask(text string) string
ClassifyTask returns a task type label from the user's first message. It uses compiled patterns from config, evaluated in order; first match wins. Returns "general" if no pattern matches or the text is empty.
func (*EvaluatorService) Diagnose ¶ added in v1.2.4
func (s *EvaluatorService) Diagnose(ctx context.Context) (*Report, error)
Diagnose runs the doctor for this service (its config, DB and background state).
func (*EvaluatorService) EvaluateNow ¶ added in v1.2.4
func (s *EvaluatorService) EvaluateNow(ctx context.Context, sessionID string, opts EvaluateOptions) (*Result, error)
EvaluateNow evaluates a session synchronously and returns the reward decomposition (or the reason it was skipped). Unlike EvaluateSession it ignores cfg.Async.
func (*EvaluatorService) EvaluateSession ¶
func (s *EvaluatorService) EvaluateSession(ctx context.Context, sessionID string) error
EvaluateSession triggers evaluation of a session on explicit request. It does not apply the completion guards (minimum user turns, subagent sessions).
func (*EvaluatorService) Flush ¶ added in v1.2.4
func (s *EvaluatorService) Flush(ctx context.Context) error
Flush waits for in-flight async evaluations until ctx is done. Call it on shutdown with a short deadline so pending evaluations are not lost.
func (*EvaluatorService) GetActiveSkills ¶
GetActiveSkills returns the approved skills for a task type, best ranked first (success_rate, then usage). It has no side effects: injection accounting happens once per session in SessionSkills.
func (*EvaluatorService) GetStats ¶
func (s *EvaluatorService) GetStats(ctx context.Context) (*Stats, error)
GetStats returns system statistics for TUI display.
func (*EvaluatorService) IsEnabled ¶
func (s *EvaluatorService) IsEnabled() bool
IsEnabled returns whether the evaluator is active.
func (*EvaluatorService) LastEvaluationError ¶ added in v1.2.4
func (s *EvaluatorService) LastEvaluationError() *JudgeError
LastEvaluationError returns the most recent evaluation failure since the process started, or nil when there was none.
func (*EvaluatorService) LastJudgeError ¶ added in v1.2.4
func (s *EvaluatorService) LastJudgeError() *JudgeError
LastJudgeError returns the most recent judge failure (initialisation or call) since the process started, or nil when there was none.
func (*EvaluatorService) ListSkills ¶ added in v1.2.4
func (s *EvaluatorService) ListSkills(ctx context.Context, status, taskType string) ([]Skill, error)
ListSkills returns the learned skills (files joined with their statistics), optionally filtered by status ("" = all) and task type ("" = all; general skills always match a task type filter).
func (*EvaluatorService) MarkCompleted ¶ added in v1.2.4
func (s *EvaluatorService) MarkCompleted(ctx context.Context, sessionID, reason string) error
MarkCompleted evaluates a session that a surface considers completed. See Service.MarkCompleted.
func (*EvaluatorService) NewContextTrimmer ¶ added in v0.407.0
func (s *EvaluatorService) NewContextTrimmer() *ContextTrimmer
NewContextTrimmer creates a ContextTrimmer backed by this service's judge infrastructure. Returns nil if the evaluator has no judge configured, if the evaluator itself is nil, or unless evaluator.contextTrimmer.enabled is set (opt-in, default off).
func (*EvaluatorService) RecordFeedback ¶ added in v1.2.4
func (s *EvaluatorService) RecordFeedback(ctx context.Context, sessionID, rating, note string) (*Result, error)
RecordFeedback stores explicit feedback for a session as an event and re-scores the session so the feedback dominates: bad gives a total below 0.3, good above 0.8. An already scored session has its score row replaced in place; an unscored one is scored now. Later feedback supersedes earlier.
func (*EvaluatorService) ReviewSkill ¶ added in v1.2.4
ReviewSkill approves or rejects a learned skill: it updates the file (source of truth) and the mirror. Approving beyond MaxSkills evicts the lowest ranked approved skill (marked rejected so it is not proposed again).
func (*EvaluatorService) RunBackground ¶ added in v1.2.4
func (s *EvaluatorService) RunBackground(ctx context.Context, isPrimary func() bool)
RunBackground starts the idle sweeper and the one-shot startup backfill. It blocks until ctx is done, so call it in a goroutine. isPrimary is consulted before every action so the work only runs on the instance that owns the database writer, including after a failover promotion.
func (*EvaluatorService) SelectVariant ¶ added in v1.2.4
func (s *EvaluatorService) SelectVariant(ctx context.Context, sessionID, section string, candidates []string) (string, error)
SelectVariant returns the variant id to use for a section of a session. candidates[0] is the default variant (the embedded template); the rest are variant files. The first call for a (session, section) chooses and persists the variant in session_template_selections; later calls, also from another process after a restart, return the persisted choice so the prompt bytes and the reward attribution stay stable for the whole session.
func (*EvaluatorService) SessionSkills ¶ added in v1.2.4
func (s *EvaluatorService) SessionSkills(ctx context.Context, sessionID, taskType string) ([]Skill, error)
SessionSkills returns the learned skills injected into a session's system prompt. The set is chosen once, on the session's first prompt build, and persisted in session_skill_injections: later turns and later processes get the same skills in the same order, even when a skill is approved mid-session (it is picked up by the next session). usage_count is incremented once per skill when the set is persisted. Without a session id the current approved skills are returned unfrozen and without accounting.
func (*EvaluatorService) SetFirstPassHook ¶ added in v1.2.4
func (s *EvaluatorService) SetFirstPassHook(fn func(ctx context.Context))
SetFirstPassHook registers fn to run once, after the first background pass on the primary instance (startup backfill + idle sweep). Call before RunBackground.
func (*EvaluatorService) SetWorkDir ¶ added in v1.2.4
func (s *EvaluatorService) SetWorkDir(dir string)
SetWorkDir overrides the project directory used for learned-skill files.
func (*EvaluatorService) Sweep ¶ added in v1.2.4
func (s *EvaluatorService) Sweep(ctx context.Context, opts SweepOptions) (int, error)
Sweep evaluates sessions that have no score, at least two user messages and have been idle for opts.IdleFor, oldest first. It returns how many sessions were scored. Sessions that fail or are skipped are not retried by later sweeps of this process, so a bad session cannot starve the queue.
func (*EvaluatorService) SyncLearnedSkills ¶ added in v1.2.4
func (s *EvaluatorService) SyncLearnedSkills(ctx context.Context) error
SyncLearnedSkills mirrors the skill files into skill_library: new or changed files are upserted (status decides is_active), mirror rows whose file is gone are deactivated, and pre-review rows are exported as pending files. Only rows that differ are written, so calling it on every new session is cheap.
func (*EvaluatorService) TemplatesEnabled ¶ added in v1.2.4
func (s *EvaluatorService) TemplatesEnabled() bool
TemplatesEnabled reports whether prompt variant selection is active: the evaluator is enabled and evaluator.templates.enabled is not switched off.
type Judge ¶ added in v0.244.0
type Judge struct {
// contains filtered or unexported fields
}
Judge calls an LLM model to evaluate session quality.
func (*Judge) Evaluate ¶ added in v0.244.0
func (j *Judge) Evaluate(ctx context.Context, meta JudgeMeta, customPromptTemplate string) (*JudgeOutput, error)
Evaluate calls the judge model with the session transcript and returns structured output.
func (*Judge) EvaluateWithUsage ¶ added in v1.2.4
func (j *Judge) EvaluateWithUsage(ctx context.Context, meta JudgeMeta, customPromptTemplate string) (*JudgeResult, error)
EvaluateWithUsage is Evaluate plus the model name and token usage of the call. When the provider reports no usage the tokens are estimated.
type JudgeError ¶ added in v1.2.4
JudgeError is the most recent judge failure, kept in memory for diagnostics.
type JudgeMeta ¶
type JudgeMeta struct {
TemplateName string
TemplateVersion int
Corrections int
Tokens int64
Transcript string
}
JudgeMeta holds metadata passed to the judge prompt.
type JudgeOutput ¶
type JudgeOutput struct {
Reasoning string `json:"reasoning"`
KeyPoints []string `json:"key_points"`
NewSkill string `json:"new_skill"`
TaskType string `json:"task_type"`
Confidence float64 `json:"confidence"`
}
JudgeOutput is the structured response from the LLM judge model.
type JudgeResult ¶ added in v1.2.4
type JudgeResult struct {
Output *JudgeOutput
Model string
PromptTokens int64
CompletionTokens int64
}
JudgeResult is a parsed judge answer plus the accounting of the call.
type PatternHit ¶ added in v1.2.4
type PatternHit struct {
// Index is the position of the message in the session (0-based).
Index int `json:"index"`
Pattern string `json:"pattern"`
Weight float64 `json:"weight"`
// AfterAssistant is true when the turn directly follows an assistant turn.
AfterAssistant bool `json:"afterAssistant"`
Snippet string `json:"snippet"`
}
PatternHit records one user turn flagged as a correction.
type PatternIssue ¶ added in v1.2.4
type PatternIssue struct {
// Kind is PatternIssueCompile or PatternIssueDoubleBackslash.
Kind string `json:"kind"`
// Source is the config key holding the pattern.
Source string `json:"source"`
Pattern string `json:"pattern"`
Message string `json:"message"`
// Hint tells how to fix it.
Hint string `json:"hint"`
}
PatternIssue is a problem found in a correction or task pattern.
func LintPatterns ¶ added in v1.2.4
func LintPatterns(cfg config.EvaluatorConfig) []PatternIssue
LintPatterns checks the correction and task patterns. It flags patterns that do not compile and patterns containing a literal double backslash, which almost always come from a TOML single-quoted string ('\\b') or an escaped pattern in a double-quoted one written with four backslashes; the regex then matches a backslash followed by "b" instead of a word boundary.
type Report ¶ added in v1.2.4
type Report struct {
Enabled bool `json:"enabled"`
Model string `json:"model,omitempty"`
Provider string `json:"provider,omitempty"`
// DisabledReasons explain why the loop is off or cannot run.
DisabledReasons []string `json:"disabled_reasons,omitempty"`
// Warnings are problems that do not stop the loop but need attention.
Warnings []string `json:"warnings,omitempty"`
EligibleSessions int64 `json:"eligible_sessions"`
EvaluatedSessions int64 `json:"evaluated_sessions"`
// NeverEvaluated counts eligible sessions without a score.
NeverEvaluated int64 `json:"never_evaluated"`
// RecentWindow / RecentEvaluated: of the last RecentWindow idle eligible
// sessions, how many have a score.
RecentWindow int64 `json:"recent_window"`
RecentEvaluated int64 `json:"recent_evaluated"`
// LastEvaluatedAt is the unix time of the newest score (0 = none).
LastEvaluatedAt int64 `json:"last_evaluated_at"`
LastEvaluationError *DoctorError `json:"last_evaluation_error,omitempty"`
Judge DoctorJudge `json:"judge"`
Variants DoctorVariants `json:"variants"`
Skills DoctorSkills `json:"skills"`
ContextTrimmer bool `json:"context_trimmer"`
Background DoctorBackground `json:"background"`
PatternIssues []PatternIssue `json:"pattern_issues,omitempty"`
CollectionErrors []string `json:"collection_errors,omitempty"`
GeneratedAt int64 `json:"generated_at"`
SessionsAvailable bool `json:"sessions_available"`
}
Report is the outcome of Diagnose: everything needed to tell whether the self-improvement loop is working and, if not, why.
func Diagnose ¶ added in v1.2.4
func Diagnose(ctx context.Context, opts DiagnoseOptions) (*Report, error)
Diagnose inspects the self-improvement loop and returns a Report. It only reads: it never writes to the database or to disk.
func (*Report) HasProblem ¶ added in v1.2.4
HasProblem reports whether a banner should be shown: the loop is enabled but nothing was evaluated among the recent sessions, or a pattern is broken.
func (*Report) Healthy ¶ added in v1.2.4
Healthy reports whether the report has no disabled reasons and no warnings.
func (*Report) ProblemLine ¶ added in v1.2.4
ProblemLine is the one-line warning shown in the TUI and the WebUI, or "".
type Result ¶ added in v1.2.4
type Result struct {
SessionID string
// Skipped is non-empty when no score was written; it holds the reason.
Skipped string
// Reward is the decomposition of the persisted score (zero when Skipped).
Reward RewardResult
// Judged reports whether the LLM judge ran.
Judged bool
}
Result is the outcome of one evaluation.
type RewardResult ¶
type RewardResult struct {
Total float64
SuccessScore float64
EfficiencyScore float64
PromptTokens int64
CompletionTokens int64
MessageCount int64
UserCorrections int
// Breakdown is the explainable decomposition persisted with the score.
Breakdown Breakdown
}
RewardResult holds the decomposed reward calculation for a session.
type SectionVariants ¶ added in v1.2.4
type SectionVariants struct {
Section string `json:"section"`
// Files is the number of variant files (the embedded default is extra).
Files int `json:"files"`
// Competing is true when the section has the default plus at least one file,
// i.e. two or more candidates, so A/B selection actually happens.
Competing bool `json:"competing"`
}
SectionVariants is one prompt section that has variant files.
type Service ¶
type Service interface {
// EvaluateSession triggers evaluation of a session on explicit request
// (async if configured). It bypasses the completion guards of MarkCompleted.
EvaluateSession(ctx context.Context, sessionID string) error
// MarkCompleted is the entry point for session-completion triggers (session
// switch, idle sweep, shutdown). It applies the evaluation guards (minimum
// user turns, subagent sessions, idempotency, in-flight dedupe) and
// evaluates asynchronously when configured. reason is only used for logging.
MarkCompleted(ctx context.Context, sessionID, reason string) error
// SelectVariant returns the prompt variant id chosen for a section of a
// session among candidates (candidates[0] is the default variant). The choice
// is made once per (session, section), persisted and frozen; it is the
// default when the feature is off or there is nothing to choose from.
SelectVariant(ctx context.Context, sessionID, section string, candidates []string) (string, error)
// GetActiveSkills returns the approved skills for a task type.
GetActiveSkills(ctx context.Context, taskType string) ([]Skill, error)
// ListSkills returns the learned skills (files + statistics), optionally
// filtered by status (pending|approved|rejected) and task type.
ListSkills(ctx context.Context, status, taskType string) ([]Skill, error)
// ReviewSkill approves or rejects a learned skill by id.
ReviewSkill(ctx context.Context, id, status string) (*Skill, error)
// GetStats returns current UCB rankings and skill library summary.
GetStats(ctx context.Context) (*Stats, error)
// IsEnabled returns whether the evaluator is active.
IsEnabled() bool
// ClassifyTask returns a task type label from the user's first message.
// Returns "general" if no pattern matches.
ClassifyTask(text string) string
}
Service defines the evaluator interface used by other packages.
type SessionDetail ¶ added in v1.2.4
type SessionDetail struct {
SessionID string `json:"session_id"`
Title string `json:"title"`
Reward float64 `json:"reward"`
SuccessScore float64 `json:"success_score"`
EfficiencyScore float64 `json:"efficiency_score"`
MessageCount int64 `json:"message_count"`
UserCorrections int64 `json:"user_corrections"`
EvaluatedAt int64 `json:"evaluated_at"`
// Breakdown is the parsed components JSON (zero for legacy rows).
Breakdown Breakdown `json:"breakdown"`
// Components is the persisted components JSON as stored.
Components json.RawMessage `json:"components"`
// JudgeAnalysis is the stored judge output, JSON null when the judge did not run.
JudgeAnalysis json.RawMessage `json:"judge_analysis"`
JudgeModel string `json:"judge_model,omitempty"`
JudgePromptTokens int64 `json:"judge_prompt_tokens"`
JudgeCompletionTokens int64 `json:"judge_completion_tokens"`
Variants []SessionVariant `json:"variants"`
Skills []SessionSkillRef `json:"skills"`
}
SessionDetail is one evaluated session with everything the observability surfaces show: the reward decomposition, the corrections it counted, the variants and skills it ran with, and the judge output.
func RecentSessionDetails ¶ added in v1.2.4
RecentSessionDetails lists the most recently evaluated sessions with their components, variants and skills.
func (SessionDetail) TopComponents ¶ added in v1.2.4
func (d SessionDetail) TopComponents(n int) []ComponentScore
TopComponents returns up to n "name value" pairs of the session's reward components, highest weight first (ties by name).
type SessionSkillRef ¶ added in v1.2.4
SessionSkillRef is a learned skill injected into an evaluated session.
type SessionVariant ¶ added in v1.2.4
SessionVariant is a prompt variant served to an evaluated session.
type Skill ¶
type Skill struct {
ID string
Title string
Content string
TaskType string
// SuccessRate is the mean reward of the evaluated sessions the skill was
// injected in (EvalCount of them); UsageCount counts sessions it was injected in.
SuccessRate float64
UsageCount int
EvalCount int
// Status is the review state: pending, approved or rejected.
Status string
Confidence float64
JudgeModel string
SourceSession string
Created time.Time
}
Skill represents a learned optimization rule from the Skill Library.
type SkillFile ¶ added in v1.2.4
type SkillFile struct {
ID string
Title string
Status string
TaskType string
Confidence float64
SourceSession string
JudgeModel string
Created time.Time
// Content is the rule text (the file body).
Content string
// Path is the absolute file path.
Path string
}
SkillFile is a parsed learned-skill file.
func ReadLearnedSkills ¶ added in v1.2.4
ReadLearnedSkills reads every skill file of workDir, newest first. Files that cannot be parsed are skipped with a warning.
func SetSkillFileStatus ¶ added in v1.2.4
SetSkillFileStatus sets the status of a learned-skill file (approve/reject). It touches the file only; the database mirror is refreshed by the next sync.
type Stats ¶
type Stats struct {
TotalEvaluations int
Templates []TemplateStats
SkillCount int
TopSkills []Skill
AvgReward float64
LastEvaluation time.Time
IsEnabled bool
// RecentSessions are the latest evaluated sessions with their reward
// components, variants and skills.
RecentSessions []SessionDetail
// Daily is evaluations and mean reward per day for the last 14 days.
Daily []DailyMetric
// Problem is the one-line doctor warning ("" when healthy).
Problem string
}
Stats is the overall self-improvement system statistics.
type SweepOptions ¶ added in v1.2.4
type SweepOptions struct {
// IdleFor is how long a session must have been idle to be picked up.
IdleFor time.Duration
// Limit bounds the number of sessions evaluated in this sweep.
Limit int
// SkipJudge disables the LLM judge for the sweep.
SkipJudge bool
// Pause is slept between evaluations to rate-limit the sweep.
Pause time.Duration
// OnResult, when set, receives every successfully scored session.
OnResult func(*Result)
}
SweepOptions controls one background sweep over unevaluated sessions.
type TemplateStats ¶
type TemplateStats struct {
// VariantID is "<section>#<variant>"; "<section>#default" is the embedded template.
VariantID string
Section string
Variant string
TimesUsed int
AvgReward float64
UCBScore float64
Rank int
}
TemplateStats holds the UCB statistics of one prompt variant (for UI display).
func VariantStatsFromRows ¶ added in v1.2.4
func VariantStatsFromRows(rows []db.PromptVariantStat, explorationC float64) []TemplateStats
VariantStatsFromRows converts persisted per-variant stats into ranked TemplateStats: UCB1 within each section, sections in alphabetical order.