d34fdf40aa
The onnxruntime .so was already vendored at deps/onnxruntime-linux-x64-1.26.0 — nothing to download. make eval-router now defaults MAVEN_ONNX_LIB there, so both baselines run by default and only a fresh clone without deps/ falls back to the hash ratchet alone. Prod-representative result, deployed 0.55 gate: 28/76 (36.8%), RU 25/61, EN 3/15, hard 0/11 → 4/11, p50 31ms / p95 71ms. Versus the hash floor's 13/76 at p50 9µs. The finding is not the accuracy, it's the refusal lane: missed clarifies went 0 → 5 of 6. Better embeddings raise cosine everywhere, so the 0.55 threshold that used to hold ambiguous utterances back stops holding — "сделай это" routes to act at 0.847, "бэкап" to chat at 0.755. The gate was implicitly tuned to the hash floor's low similarities. That is an argument about the threshold, not about the embedder, and it lands before #320 rather than after. Also fixes a fixture-model mismatch: ReminderGrammar deliberately skips the extractor at stage 0 and the daemon's applyAction parses the time downstream (stage0.go says so). Charging the router for that slot made 4 exact-match wins read as misses; they are now counted as SlotsDeferred instead. Hash baseline moves 13/76, ratchet to 0.15. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01X5JApcrCRVGmqrxnhynSik
355 lines
12 KiB
Go
355 lines
12 KiB
Go
// Package eval is the held-out routing contract — the fixture Vikunja #319
|
||
// measures against before #320 flips the default route decider.
|
||
//
|
||
// Why it is a separate package from router: the fixture must be scorable by
|
||
// BOTH paths (today's classifier cascade and the resident model's LLM router)
|
||
// from outside the router package, including from cmd/mavend where the real
|
||
// llama-server client lives. A _test.go file in router can't be imported, and
|
||
// testdata isn't reachable from another package's working directory — so the
|
||
// fixture is embedded here and the scorer takes a Router interface.
|
||
//
|
||
// The fixture is HELD OUT from models/seeds/*.txt on purpose: a classifier
|
||
// scored on its own seed phrases measures memorisation of frozen centroids,
|
||
// which is exactly the illusion that hid the weak RU query handling. See
|
||
// TestFixtureIsHeldOut, which enforces it.
|
||
package eval
|
||
|
||
import (
|
||
"context"
|
||
_ "embed"
|
||
"encoding/json"
|
||
"fmt"
|
||
"sort"
|
||
"strings"
|
||
"time"
|
||
|
||
"github.com/kami/maven/internal/router"
|
||
)
|
||
|
||
//go:embed ru_routing_v1.json
|
||
var fixtureJSON []byte
|
||
|
||
// Case — one utterance and the route it must produce. Slot expectations are
|
||
// deliberately coarse (see the fixture's notes): want_fn is a boolean because
|
||
// the fn allowlist lives in deploy config, and want_fact_key names the loop's
|
||
// rule keys because a fact under the wrong key starves its predicate.
|
||
//
|
||
// Intent is empty exactly when WantClarify is set: the contract there is that
|
||
// the router refuses instead of guessing.
|
||
type Case struct {
|
||
ID string `json:"id"`
|
||
Utterance string `json:"utterance"`
|
||
Lang string `json:"lang"`
|
||
Intent router.Intent `json:"intent"`
|
||
WantTime bool `json:"want_time"`
|
||
WantFn bool `json:"want_fn"`
|
||
WantFactKey string `json:"want_fact_key"`
|
||
WantClarify bool `json:"want_clarify"`
|
||
Tags []string `json:"tags"`
|
||
Note string `json:"note"`
|
||
}
|
||
|
||
// Fixture — the versioned envelope, same shape as
|
||
// cmd/mavend/testdata/system_safety_scenarios.json. SchemaVersion gates the
|
||
// loader so an older binary refuses a fixture it would misread rather than
|
||
// scoring it wrong and reporting a number.
|
||
type Fixture struct {
|
||
SchemaVersion int `json:"schema_version"`
|
||
Name string `json:"name"`
|
||
ReferenceNow string `json:"reference_now"`
|
||
Notes []string `json:"notes"`
|
||
Cases []Case `json:"cases"`
|
||
}
|
||
|
||
// SchemaVersion — the version this package understands.
|
||
const SchemaVersion = 1
|
||
|
||
// Load returns the embedded fixture.
|
||
func Load() (Fixture, error) {
|
||
var f Fixture
|
||
if err := json.Unmarshal(fixtureJSON, &f); err != nil {
|
||
return Fixture{}, fmt.Errorf("parse fixture: %w", err)
|
||
}
|
||
if f.SchemaVersion != SchemaVersion {
|
||
return Fixture{}, fmt.Errorf("fixture schema_version %d, want %d", f.SchemaVersion, SchemaVersion)
|
||
}
|
||
if len(f.Cases) == 0 {
|
||
return Fixture{}, fmt.Errorf("fixture has no cases")
|
||
}
|
||
return f, nil
|
||
}
|
||
|
||
// Now — the fixture's reference clock. Relative reminder slots ("через
|
||
// полчаса") resolve against it, so a scoring run is reproducible regardless of
|
||
// when it runs.
|
||
func (f Fixture) Now() (time.Time, error) {
|
||
t, err := time.Parse(time.RFC3339, f.ReferenceNow)
|
||
if err != nil {
|
||
return time.Time{}, fmt.Errorf("parse reference_now %q: %w", f.ReferenceNow, err)
|
||
}
|
||
return t, nil
|
||
}
|
||
|
||
// Router — the one thing a route decider must do to be scorable. *router.Router
|
||
// satisfies it directly; an LLM-only path wraps its Route in RouterFunc.
|
||
type Router interface {
|
||
Route(ctx context.Context, utterance string, now time.Time) (router.Decision, error)
|
||
}
|
||
|
||
// RouterFunc adapts a bare function to Router — for scoring a single stage
|
||
// (e.g. *router.LLMRouter, whose Route returns an extra ok bool) without
|
||
// standing up the whole cascade.
|
||
type RouterFunc func(ctx context.Context, utterance string, now time.Time) (router.Decision, error)
|
||
|
||
// Route implements Router.
|
||
func (f RouterFunc) Route(ctx context.Context, utterance string, now time.Time) (router.Decision, error) {
|
||
return f(ctx, utterance, now)
|
||
}
|
||
|
||
// Outcome — one scored case. Reasons is empty exactly when Pass is true.
|
||
type Outcome struct {
|
||
Case Case
|
||
Decision router.Decision
|
||
Err error
|
||
Latency time.Duration
|
||
Pass bool
|
||
// IntentOK is tracked separately from Pass: a case can land the right
|
||
// intent and still fail on a slot, and #319 needs those two numbers apart
|
||
// (a slot gap is a parser fix; a wrong intent is a router fix).
|
||
IntentOK bool
|
||
Reasons []string
|
||
}
|
||
|
||
// Report — the aggregate. Accuracy is the headline; the rest exists so a
|
||
// regression names itself instead of just moving a percentage.
|
||
type Report struct {
|
||
Name string
|
||
Total int
|
||
Passed int
|
||
IntentHit int
|
||
// FalseClarify — the router asked when the fixture expected a decision.
|
||
// A gap, recoverable by asking again.
|
||
FalseClarify int
|
||
// MissedClarify — the router decided confidently where the fixture
|
||
// expected a refusal. The dangerous direction: "сделай это" routed to an
|
||
// act is a confident destructive guess.
|
||
MissedClarify int
|
||
Errors int
|
||
// SlotsDeferred — stage-0 hits whose slot the daemon fills downstream
|
||
// (reminder grammar → applyAction's time parser). Not a miss, but not a
|
||
// full router-level win either; tracked so the two aren't conflated.
|
||
SlotsDeferred int
|
||
Outcomes []Outcome
|
||
// Confusion counts want→got intent pairs, decided cases only.
|
||
Confusion map[string]int
|
||
// ByTag accuracy for the fixture's tags ("hard", "homelab", …).
|
||
ByTag map[string]TagStat
|
||
// ByLang accuracy — the RU/EN split is the whole reason this fixture
|
||
// exists.
|
||
ByLang map[string]TagStat
|
||
P50 time.Duration
|
||
P95 time.Duration
|
||
Max time.Duration
|
||
}
|
||
|
||
// TagStat — passed/total for one slice of the fixture.
|
||
type TagStat struct{ Passed, Total int }
|
||
|
||
// Accuracy — fraction of cases fully satisfied (intent AND slots AND the
|
||
// clarify contract).
|
||
func (r Report) Accuracy() float64 {
|
||
if r.Total == 0 {
|
||
return 0
|
||
}
|
||
return float64(r.Passed) / float64(r.Total)
|
||
}
|
||
|
||
// IntentAccuracy — fraction with the right intent, ignoring slot fills.
|
||
func (r Report) IntentAccuracy() float64 {
|
||
if r.Total == 0 {
|
||
return 0
|
||
}
|
||
return float64(r.IntentHit) / float64(r.Total)
|
||
}
|
||
|
||
// Score runs every case through r and aggregates. It never fails the run on a
|
||
// route error — an erroring case scores as a miss and is counted in Errors,
|
||
// because "the model was down" and "the model was wrong" are different numbers
|
||
// and #319 needs to tell them apart.
|
||
//
|
||
// Latency is wall-clock per Route call, including any llama-server round trip.
|
||
// That is the point on the CPU-only target: a correctness win the resident
|
||
// model pays for with seconds per turn is not a win.
|
||
func Score(ctx context.Context, name string, r Router, f Fixture) (Report, error) {
|
||
now, err := f.Now()
|
||
if err != nil {
|
||
return Report{}, err
|
||
}
|
||
rep := Report{
|
||
Name: name,
|
||
Total: len(f.Cases),
|
||
Confusion: map[string]int{},
|
||
ByTag: map[string]TagStat{},
|
||
ByLang: map[string]TagStat{},
|
||
}
|
||
lat := make([]time.Duration, 0, len(f.Cases))
|
||
|
||
for _, c := range f.Cases {
|
||
start := time.Now()
|
||
d, err := r.Route(ctx, c.Utterance, now)
|
||
o := Outcome{Case: c, Decision: d, Err: err, Latency: time.Since(start)}
|
||
lat = append(lat, o.Latency)
|
||
|
||
switch {
|
||
case err != nil:
|
||
rep.Errors++
|
||
o.Reasons = append(o.Reasons, fmt.Sprintf("route error: %v", err))
|
||
case c.WantClarify:
|
||
// Only the refusal matters here; whatever intent the cascade
|
||
// guessed underneath is irrelevant if it gated.
|
||
o.IntentOK = d.Clarify
|
||
if !d.Clarify {
|
||
rep.MissedClarify++
|
||
o.Reasons = append(o.Reasons, fmt.Sprintf("decided %q confidently (%.3f), want clarify", d.Intent, d.Confidence))
|
||
}
|
||
default:
|
||
o.IntentOK = d.Intent == c.Intent && !d.Clarify
|
||
if d.Clarify {
|
||
rep.FalseClarify++
|
||
o.Reasons = append(o.Reasons, fmt.Sprintf("clarified (%.3f), want intent %q", d.Confidence, c.Intent))
|
||
} else if d.Intent != c.Intent {
|
||
rep.Confusion[string(c.Intent)+"→"+string(d.Intent)]++
|
||
o.Reasons = append(o.Reasons, fmt.Sprintf("intent %q, want %q (%.3f)", d.Intent, c.Intent, d.Confidence))
|
||
}
|
||
if c.WantTime && !d.Slots.HasTime {
|
||
// Stage 0 skips the extractor by design: ReminderGrammar
|
||
// captures the text after "напомни"/"remind me" and the
|
||
// daemon's applyAction runs the time parser on it (see
|
||
// stage0.go). Charging the router for a slot it was never
|
||
// asked to fill would make an exact-match win look like a
|
||
// miss — so it is counted, not failed.
|
||
if d.Stage == 0 {
|
||
rep.SlotsDeferred++
|
||
} else {
|
||
o.Reasons = append(o.Reasons, "no time slot, want one")
|
||
}
|
||
}
|
||
if c.WantFn && !d.Slots.HasFn {
|
||
o.Reasons = append(o.Reasons, "no fn slot, want an allowlist match")
|
||
}
|
||
if c.WantFactKey != "" && d.Slots.Key != c.WantFactKey {
|
||
o.Reasons = append(o.Reasons, fmt.Sprintf("fact key %q, want %q", d.Slots.Key, c.WantFactKey))
|
||
}
|
||
}
|
||
|
||
o.Pass = len(o.Reasons) == 0
|
||
if o.Pass {
|
||
rep.Passed++
|
||
}
|
||
if o.IntentOK {
|
||
rep.IntentHit++
|
||
}
|
||
bump(rep.ByLang, c.Lang, o.Pass)
|
||
for _, tag := range c.Tags {
|
||
bump(rep.ByTag, tag, o.Pass)
|
||
}
|
||
rep.Outcomes = append(rep.Outcomes, o)
|
||
}
|
||
|
||
sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] })
|
||
rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95)
|
||
if len(lat) > 0 {
|
||
rep.Max = lat[len(lat)-1]
|
||
}
|
||
return rep, nil
|
||
}
|
||
|
||
func bump(m map[string]TagStat, key string, pass bool) {
|
||
if key == "" {
|
||
return
|
||
}
|
||
s := m[key]
|
||
s.Total++
|
||
if pass {
|
||
s.Passed++
|
||
}
|
||
m[key] = s
|
||
}
|
||
|
||
// percentile — nearest-rank on a pre-sorted slice. No interpolation: with ~80
|
||
// samples an interpolated p95 invents a latency no turn actually took.
|
||
func percentile(sorted []time.Duration, p float64) time.Duration {
|
||
if len(sorted) == 0 {
|
||
return 0
|
||
}
|
||
i := int(p * float64(len(sorted)))
|
||
if i >= len(sorted) {
|
||
i = len(sorted) - 1
|
||
}
|
||
return sorted[i]
|
||
}
|
||
|
||
// String renders the report as the comparison table #319 pastes into the task —
|
||
// headline accuracy, the two clarify directions apart, latency, and the slices
|
||
// that name where a path is weak.
|
||
func (r Report) String() string {
|
||
var b strings.Builder
|
||
fmt.Fprintf(&b, "%s: %d/%d cases (%.1f%% full, %.1f%% intent-only)\n",
|
||
r.Name, r.Passed, r.Total, 100*r.Accuracy(), 100*r.IntentAccuracy())
|
||
fmt.Fprintf(&b, " clarify: %d false (asked, shouldn't) / %d missed (guessed, shouldn't) | errors: %d | slots deferred to daemon: %d\n",
|
||
r.FalseClarify, r.MissedClarify, r.Errors, r.SlotsDeferred)
|
||
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
|
||
fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang))
|
||
fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag))
|
||
if len(r.Confusion) > 0 {
|
||
fmt.Fprintf(&b, " confusion: %s\n", renderCounts(r.Confusion))
|
||
}
|
||
return b.String()
|
||
}
|
||
|
||
// Failures — the per-case detail, sorted by ID so two runs diff cleanly.
|
||
func (r Report) Failures() string {
|
||
var b strings.Builder
|
||
out := append([]Outcome(nil), r.Outcomes...)
|
||
sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID })
|
||
for _, o := range out {
|
||
if o.Pass {
|
||
continue
|
||
}
|
||
fmt.Fprintf(&b, " %s %q: %s\n", o.Case.ID, o.Case.Utterance, strings.Join(o.Reasons, "; "))
|
||
}
|
||
return b.String()
|
||
}
|
||
|
||
func renderStats(m map[string]TagStat) string {
|
||
keys := make([]string, 0, len(m))
|
||
for k := range m {
|
||
keys = append(keys, k)
|
||
}
|
||
sort.Strings(keys)
|
||
parts := make([]string, 0, len(keys))
|
||
for _, k := range keys {
|
||
s := m[k]
|
||
parts = append(parts, fmt.Sprintf("%s %d/%d", k, s.Passed, s.Total))
|
||
}
|
||
return strings.Join(parts, " ")
|
||
}
|
||
|
||
func renderCounts(m map[string]int) string {
|
||
keys := make([]string, 0, len(m))
|
||
for k := range m {
|
||
keys = append(keys, k)
|
||
}
|
||
sort.Slice(keys, func(i, j int) bool {
|
||
if m[keys[i]] != m[keys[j]] {
|
||
return m[keys[i]] > m[keys[j]]
|
||
}
|
||
return keys[i] < keys[j]
|
||
})
|
||
parts := make([]string, 0, len(keys))
|
||
for _, k := range keys {
|
||
parts = append(parts, fmt.Sprintf("%s ×%d", k, m[k]))
|
||
}
|
||
return strings.Join(parts, " ")
|
||
}
|