Files
Maven/internal/router/eval/eval.go
T
claude 2597a7b34a Score the destination apart from the intent (V-659)
The fixture measured the first half of a route and stopped. V-655 split a
routing decision in two, and the second half arrived with no fixture, so
Decision.Source had no accuracy number at all.

want_source is a pointer because the destination has three states and a
bare string has two. Absent is every intent but query, which never reaches
queryWalk. Present and empty is the SourceUnknown contract: name nothing
and let the daemon walk the chain, which is right whenever two destinations
can both answer and the utterance does not choose. Present and named is a
destination the route must produce.

A destination miss does not fail the case. It goes in SourceReason, never
in Reasons, so Accuracy and IntentAccuracy stay the numbers they were and
69/91 still means what it meant. SourceAccuracy is the second number, over
the labelled cases only, because a percentage of the whole fixture would be
a percentage of turns that never ask a query source.

A clarified or mis-routed case still counts in the denominator. It named no
destination and that is a miss, not a case to skip, or the denominator drops
every turn the route already lost.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_013ptwopxyo3Z2kwFckHkLvN
2026-08-08 18:00:37 +04:00

422 lines
15 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Package eval is the held-out routing contract — the fixture Vikunja #319
// measures against before #320 flips the default route decider.
//
// Why it is a separate package from router: the fixture must be scorable by
// BOTH paths (today's classifier cascade and the resident model's LLM router)
// from outside the router package, including from cmd/mavend where the real
// llama-server client lives. A _test.go file in router can't be imported, and
// testdata isn't reachable from another package's working directory — so the
// fixture is embedded here and the scorer takes a Router interface.
//
// The fixture is HELD OUT from models/seeds/*.txt on purpose: a classifier
// scored on its own seed phrases measures memorisation of frozen centroids,
// which is exactly the illusion that hid the weak RU query handling. See
// TestFixtureIsHeldOut, which enforces it.
package eval
import (
"context"
_ "embed"
"encoding/json"
"fmt"
"sort"
"strings"
"time"
"github.com/kami/maven/internal/router"
)
//go:embed ru_routing_v1.json
var fixtureJSON []byte
// Case — one utterance and the route it must produce. Slot expectations are
// deliberately coarse (see the fixture's notes): want_fn is a boolean because
// the fn allowlist lives in deploy config, and want_fact_key names the loop's
// rule keys because a fact under the wrong key starves its predicate.
//
// Intent is empty exactly when WantClarify is set: the contract there is that
// the router refuses instead of guessing.
//
// WantSource is a pointer because the destination has three states and a bare
// string only has two (V-659). Absent means the case does not score a
// destination at all, which is every intent but query: a fact, a reminder, a
// note, an act, a chat or a system turn never reaches queryWalk. Present and
// empty is the SourceUnknown contract — the decider must name nothing and let
// the daemon walk the whole chain, which is the right answer whenever two
// destinations can both answer and the utterance does not choose. Present and
// named is a destination the route must produce.
type Case struct {
ID string `json:"id"`
Utterance string `json:"utterance"`
Lang string `json:"lang"`
Intent router.Intent `json:"intent"`
WantTime bool `json:"want_time"`
WantFn bool `json:"want_fn"`
WantFactKey string `json:"want_fact_key"`
WantClarify bool `json:"want_clarify"`
WantSource *router.Source `json:"want_source,omitempty"`
Tags []string `json:"tags"`
Note string `json:"note"`
}
// Fixture — the versioned envelope, same shape as
// cmd/mavend/testdata/system_safety_scenarios.json. SchemaVersion gates the
// loader so an older binary refuses a fixture it would misread rather than
// scoring it wrong and reporting a number.
type Fixture struct {
SchemaVersion int `json:"schema_version"`
Name string `json:"name"`
ReferenceNow string `json:"reference_now"`
Notes []string `json:"notes"`
Cases []Case `json:"cases"`
}
// SchemaVersion — the version this package understands.
const SchemaVersion = 1
// Load returns the embedded fixture.
func Load() (Fixture, error) {
var f Fixture
if err := json.Unmarshal(fixtureJSON, &f); err != nil {
return Fixture{}, fmt.Errorf("parse fixture: %w", err)
}
if f.SchemaVersion != SchemaVersion {
return Fixture{}, fmt.Errorf("fixture schema_version %d, want %d", f.SchemaVersion, SchemaVersion)
}
if len(f.Cases) == 0 {
return Fixture{}, fmt.Errorf("fixture has no cases")
}
return f, nil
}
// Now — the fixture's reference clock. Relative reminder slots ("через
// полчаса") resolve against it, so a scoring run is reproducible regardless of
// when it runs.
func (f Fixture) Now() (time.Time, error) {
t, err := time.Parse(time.RFC3339, f.ReferenceNow)
if err != nil {
return time.Time{}, fmt.Errorf("parse reference_now %q: %w", f.ReferenceNow, err)
}
return t, nil
}
// Router — the one thing a route decider must do to be scorable. *router.Router
// satisfies it directly; an LLM-only path wraps its Route in RouterFunc.
type Router interface {
Route(ctx context.Context, utterance string, now time.Time) (router.Decision, error)
}
// RouterFunc adapts a bare function to Router — for scoring a single stage
// (e.g. *router.LLMRouter, whose Route returns an extra ok bool) without
// standing up the whole cascade.
type RouterFunc func(ctx context.Context, utterance string, now time.Time) (router.Decision, error)
// Route implements Router.
func (f RouterFunc) Route(ctx context.Context, utterance string, now time.Time) (router.Decision, error) {
return f(ctx, utterance, now)
}
// Outcome — one scored case. Reasons is empty exactly when Pass is true.
type Outcome struct {
Case Case
Decision router.Decision
Err error
Latency time.Duration
Pass bool
// IntentOK is tracked separately from Pass: a case can land the right
// intent and still fail on a slot, and #319 needs those two numbers apart
// (a slot gap is a parser fix; a wrong intent is a router fix).
IntentOK bool
Reasons []string
// SourceReason is set when the case labelled a destination and the route
// named a different one. It is kept out of Reasons on purpose: the
// destination is the second half of a route and it is scored separately,
// so a wrong destination must not move the intent number (V-659).
SourceReason string
}
// Report — the aggregate. Accuracy is the headline; the rest exists so a
// regression names itself instead of just moving a percentage.
type Report struct {
Name string
Total int
Passed int
IntentHit int
// FalseClarify — the router asked when the fixture expected a decision.
// A gap, recoverable by asking again.
FalseClarify int
// MissedClarify — the router decided confidently where the fixture
// expected a refusal. The dangerous direction: "сделай это" routed to an
// act is a confident destructive guess.
MissedClarify int
Errors int
// SlotsDeferred — stage-0 hits whose slot the daemon fills downstream
// (reminder grammar → applyAction's time parser). Not a miss, but not a
// full router-level win either; tracked so the two aren't conflated.
SlotsDeferred int
// SourceTotal counts the cases carrying a want_source, and SourceHit the
// ones whose route named it. Reported apart from Passed because intent and
// destination are two decisions, and one number hides which one moved.
SourceTotal int
SourceHit int
// SourceConfusion counts want→got destination pairs. "" reads as the
// SourceUnknown floor on either side.
SourceConfusion map[string]int
Outcomes []Outcome
// Confusion counts want→got intent pairs, decided cases only.
Confusion map[string]int
// ByTag accuracy for the fixture's tags ("hard", "homelab", …).
ByTag map[string]TagStat
// ByLang accuracy — the RU/EN split is the whole reason this fixture
// exists.
ByLang map[string]TagStat
P50 time.Duration
P95 time.Duration
Max time.Duration
}
// TagStat — passed/total for one slice of the fixture.
type TagStat struct{ Passed, Total int }
// Accuracy — fraction of cases fully satisfied (intent AND slots AND the
// clarify contract).
func (r Report) Accuracy() float64 {
if r.Total == 0 {
return 0
}
return float64(r.Passed) / float64(r.Total)
}
// IntentAccuracy — fraction with the right intent, ignoring slot fills.
func (r Report) IntentAccuracy() float64 {
if r.Total == 0 {
return 0
}
return float64(r.IntentHit) / float64(r.Total)
}
// SourceAccuracy — fraction of the labelled cases whose route named the right
// destination. Denominator is SourceTotal and not Total, because most of the
// fixture never reaches a query source and scoring those would report a
// percentage of nothing.
func (r Report) SourceAccuracy() float64 {
if r.SourceTotal == 0 {
return 0
}
return float64(r.SourceHit) / float64(r.SourceTotal)
}
// Score runs every case through r and aggregates. It never fails the run on a
// route error — an erroring case scores as a miss and is counted in Errors,
// because "the model was down" and "the model was wrong" are different numbers
// and #319 needs to tell them apart.
//
// Latency is wall-clock per Route call, including any llama-server round trip.
// That is the point on the CPU-only target: a correctness win the resident
// model pays for with seconds per turn is not a win.
func Score(ctx context.Context, name string, r Router, f Fixture) (Report, error) {
now, err := f.Now()
if err != nil {
return Report{}, err
}
rep := Report{
Name: name,
Total: len(f.Cases),
Confusion: map[string]int{},
SourceConfusion: map[string]int{},
ByTag: map[string]TagStat{},
ByLang: map[string]TagStat{},
}
lat := make([]time.Duration, 0, len(f.Cases))
for _, c := range f.Cases {
start := time.Now()
d, err := r.Route(ctx, c.Utterance, now)
o := Outcome{Case: c, Decision: d, Err: err, Latency: time.Since(start)}
lat = append(lat, o.Latency)
switch {
case err != nil:
rep.Errors++
o.Reasons = append(o.Reasons, fmt.Sprintf("route error: %v", err))
case c.WantClarify:
// Only the refusal matters here; whatever intent the cascade
// guessed underneath is irrelevant if it gated.
o.IntentOK = d.Clarify
if !d.Clarify {
rep.MissedClarify++
o.Reasons = append(o.Reasons, fmt.Sprintf("decided %q confidently (%.3f), want clarify", d.Intent, d.Confidence))
}
default:
o.IntentOK = d.Intent == c.Intent && !d.Clarify
if d.Clarify {
rep.FalseClarify++
o.Reasons = append(o.Reasons, fmt.Sprintf("clarified (%.3f), want intent %q", d.Confidence, c.Intent))
} else if d.Intent != c.Intent {
rep.Confusion[string(c.Intent)+"→"+string(d.Intent)]++
o.Reasons = append(o.Reasons, fmt.Sprintf("intent %q, want %q (%.3f)", d.Intent, c.Intent, d.Confidence))
}
if c.WantTime && !d.Slots.HasTime {
// Stage 0 skips the extractor by design: ReminderGrammar
// captures the text after "напомни"/"remind me" and the
// daemon's applyAction runs the time parser on it (see
// stage0.go). Charging the router for a slot it was never
// asked to fill would make an exact-match win look like a
// miss — so it is counted, not failed.
if d.Stage == 0 {
rep.SlotsDeferred++
} else {
o.Reasons = append(o.Reasons, "no time slot, want one")
}
}
if c.WantFn && !d.Slots.HasFn {
o.Reasons = append(o.Reasons, "no fn slot, want an allowlist match")
}
if c.WantFactKey != "" && d.Slots.Key != c.WantFactKey {
o.Reasons = append(o.Reasons, fmt.Sprintf("fact key %q, want %q", d.Slots.Key, c.WantFactKey))
}
}
// The destination is scored outside the switch and outside Pass. A case
// that clarified or landed the wrong intent named no destination, and
// that is a real miss rather than a case to skip — otherwise the
// denominator quietly drops every turn the route already lost. Only a
// route error is skipped, because "the model was down" is the Errors
// number and not a destination result.
if c.WantSource != nil && err == nil {
rep.SourceTotal++
switch {
case d.Source == *c.WantSource:
rep.SourceHit++
default:
rep.SourceConfusion[string(*c.WantSource)+"→"+string(d.Source)]++
o.SourceReason = fmt.Sprintf("source %q, want %q", d.Source, *c.WantSource)
}
}
o.Pass = len(o.Reasons) == 0
if o.Pass {
rep.Passed++
}
if o.IntentOK {
rep.IntentHit++
}
bump(rep.ByLang, c.Lang, o.Pass)
for _, tag := range c.Tags {
bump(rep.ByTag, tag, o.Pass)
}
rep.Outcomes = append(rep.Outcomes, o)
}
sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] })
rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95)
if len(lat) > 0 {
rep.Max = lat[len(lat)-1]
}
return rep, nil
}
func bump(m map[string]TagStat, key string, pass bool) {
if key == "" {
return
}
s := m[key]
s.Total++
if pass {
s.Passed++
}
m[key] = s
}
// percentile — nearest-rank on a pre-sorted slice. No interpolation: with ~80
// samples an interpolated p95 invents a latency no turn actually took.
func percentile(sorted []time.Duration, p float64) time.Duration {
if len(sorted) == 0 {
return 0
}
i := int(p * float64(len(sorted)))
if i >= len(sorted) {
i = len(sorted) - 1
}
return sorted[i]
}
// String renders the report as the comparison table #319 pastes into the task —
// headline accuracy, the two clarify directions apart, latency, and the slices
// that name where a path is weak.
func (r Report) String() string {
var b strings.Builder
fmt.Fprintf(&b, "%s: %d/%d cases (%.1f%% full, %.1f%% intent-only)\n",
r.Name, r.Passed, r.Total, 100*r.Accuracy(), 100*r.IntentAccuracy())
fmt.Fprintf(&b, " clarify: %d false (asked, shouldn't) / %d missed (guessed, shouldn't) | errors: %d | slots deferred to daemon: %d\n",
r.FalseClarify, r.MissedClarify, r.Errors, r.SlotsDeferred)
if r.SourceTotal > 0 {
fmt.Fprintf(&b, " destination: %d/%d labelled cases (%.1f%%)\n",
r.SourceHit, r.SourceTotal, 100*r.SourceAccuracy())
}
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang))
fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag))
if len(r.Confusion) > 0 {
fmt.Fprintf(&b, " confusion: %s\n", renderCounts(r.Confusion))
}
if len(r.SourceConfusion) > 0 {
fmt.Fprintf(&b, " destination confusion: %s\n", renderCounts(r.SourceConfusion))
}
return b.String()
}
// Failures — the per-case detail, sorted by ID so two runs diff cleanly. A case
// that landed its intent and missed its destination is listed too, marked, so
// the half that moved is readable without diffing two percentages.
func (r Report) Failures() string {
var b strings.Builder
out := append([]Outcome(nil), r.Outcomes...)
sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID })
for _, o := range out {
switch {
case !o.Pass:
reasons := o.Reasons
if o.SourceReason != "" {
reasons = append(append([]string(nil), reasons...), o.SourceReason)
}
fmt.Fprintf(&b, " %s %q: %s\n", o.Case.ID, o.Case.Utterance, strings.Join(reasons, "; "))
case o.SourceReason != "":
fmt.Fprintf(&b, " %s %q: route ok, %s\n", o.Case.ID, o.Case.Utterance, o.SourceReason)
}
}
return b.String()
}
func renderStats(m map[string]TagStat) string {
keys := make([]string, 0, len(m))
for k := range m {
keys = append(keys, k)
}
sort.Strings(keys)
parts := make([]string, 0, len(keys))
for _, k := range keys {
s := m[k]
parts = append(parts, fmt.Sprintf("%s %d/%d", k, s.Passed, s.Total))
}
return strings.Join(parts, " ")
}
func renderCounts(m map[string]int) string {
keys := make([]string, 0, len(m))
for k := range m {
keys = append(keys, k)
}
sort.Slice(keys, func(i, j int) bool {
if m[keys[i]] != m[keys[j]] {
return m[keys[i]] > m[keys[j]]
}
return keys[i] < keys[j]
})
parts := make([]string, 0, len(keys))
for _, k := range keys {
parts = append(parts, fmt.Sprintf("%s ×%d", k, m[k]))
}
return strings.Join(parts, " ")
}