diff --git a/internal/router/eval/eval.go b/internal/router/eval/eval.go index a9ca05c..f4bcb78 100644 --- a/internal/router/eval/eval.go +++ b/internal/router/eval/eval.go @@ -36,17 +36,27 @@ var fixtureJSON []byte // // Intent is empty exactly when WantClarify is set: the contract there is that // the router refuses instead of guessing. +// +// WantSource is a pointer because the destination has three states and a bare +// string only has two (V-659). Absent means the case does not score a +// destination at all, which is every intent but query: a fact, a reminder, a +// note, an act, a chat or a system turn never reaches queryWalk. Present and +// empty is the SourceUnknown contract — the decider must name nothing and let +// the daemon walk the whole chain, which is the right answer whenever two +// destinations can both answer and the utterance does not choose. Present and +// named is a destination the route must produce. type Case struct { - ID string `json:"id"` - Utterance string `json:"utterance"` - Lang string `json:"lang"` - Intent router.Intent `json:"intent"` - WantTime bool `json:"want_time"` - WantFn bool `json:"want_fn"` - WantFactKey string `json:"want_fact_key"` - WantClarify bool `json:"want_clarify"` - Tags []string `json:"tags"` - Note string `json:"note"` + ID string `json:"id"` + Utterance string `json:"utterance"` + Lang string `json:"lang"` + Intent router.Intent `json:"intent"` + WantTime bool `json:"want_time"` + WantFn bool `json:"want_fn"` + WantFactKey string `json:"want_fact_key"` + WantClarify bool `json:"want_clarify"` + WantSource *router.Source `json:"want_source,omitempty"` + Tags []string `json:"tags"` + Note string `json:"note"` } // Fixture — the versioned envelope, same shape as @@ -118,6 +128,11 @@ type Outcome struct { // (a slot gap is a parser fix; a wrong intent is a router fix). IntentOK bool Reasons []string + // SourceReason is set when the case labelled a destination and the route + // named a different one. It is kept out of Reasons on purpose: the + // destination is the second half of a route and it is scored separately, + // so a wrong destination must not move the intent number (V-659). + SourceReason string } // Report — the aggregate. Accuracy is the headline; the rest exists so a @@ -139,7 +154,15 @@ type Report struct { // (reminder grammar → applyAction's time parser). Not a miss, but not a // full router-level win either; tracked so the two aren't conflated. SlotsDeferred int - Outcomes []Outcome + // SourceTotal counts the cases carrying a want_source, and SourceHit the + // ones whose route named it. Reported apart from Passed because intent and + // destination are two decisions, and one number hides which one moved. + SourceTotal int + SourceHit int + // SourceConfusion counts want→got destination pairs. "" reads as the + // SourceUnknown floor on either side. + SourceConfusion map[string]int + Outcomes []Outcome // Confusion counts want→got intent pairs, decided cases only. Confusion map[string]int // ByTag accuracy for the fixture's tags ("hard", "homelab", …). @@ -172,6 +195,17 @@ func (r Report) IntentAccuracy() float64 { return float64(r.IntentHit) / float64(r.Total) } +// SourceAccuracy — fraction of the labelled cases whose route named the right +// destination. Denominator is SourceTotal and not Total, because most of the +// fixture never reaches a query source and scoring those would report a +// percentage of nothing. +func (r Report) SourceAccuracy() float64 { + if r.SourceTotal == 0 { + return 0 + } + return float64(r.SourceHit) / float64(r.SourceTotal) +} + // Score runs every case through r and aggregates. It never fails the run on a // route error — an erroring case scores as a miss and is counted in Errors, // because "the model was down" and "the model was wrong" are different numbers @@ -186,11 +220,12 @@ func Score(ctx context.Context, name string, r Router, f Fixture) (Report, error return Report{}, err } rep := Report{ - Name: name, - Total: len(f.Cases), - Confusion: map[string]int{}, - ByTag: map[string]TagStat{}, - ByLang: map[string]TagStat{}, + Name: name, + Total: len(f.Cases), + Confusion: map[string]int{}, + SourceConfusion: map[string]int{}, + ByTag: map[string]TagStat{}, + ByLang: map[string]TagStat{}, } lat := make([]time.Duration, 0, len(f.Cases)) @@ -242,6 +277,23 @@ func Score(ctx context.Context, name string, r Router, f Fixture) (Report, error } } + // The destination is scored outside the switch and outside Pass. A case + // that clarified or landed the wrong intent named no destination, and + // that is a real miss rather than a case to skip — otherwise the + // denominator quietly drops every turn the route already lost. Only a + // route error is skipped, because "the model was down" is the Errors + // number and not a destination result. + if c.WantSource != nil && err == nil { + rep.SourceTotal++ + switch { + case d.Source == *c.WantSource: + rep.SourceHit++ + default: + rep.SourceConfusion[string(*c.WantSource)+"→"+string(d.Source)]++ + o.SourceReason = fmt.Sprintf("source %q, want %q", d.Source, *c.WantSource) + } + } + o.Pass = len(o.Reasons) == 0 if o.Pass { rep.Passed++ @@ -298,25 +350,40 @@ func (r Report) String() string { r.Name, r.Passed, r.Total, 100*r.Accuracy(), 100*r.IntentAccuracy()) fmt.Fprintf(&b, " clarify: %d false (asked, shouldn't) / %d missed (guessed, shouldn't) | errors: %d | slots deferred to daemon: %d\n", r.FalseClarify, r.MissedClarify, r.Errors, r.SlotsDeferred) + if r.SourceTotal > 0 { + fmt.Fprintf(&b, " destination: %d/%d labelled cases (%.1f%%)\n", + r.SourceHit, r.SourceTotal, 100*r.SourceAccuracy()) + } fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max) fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang)) fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag)) if len(r.Confusion) > 0 { fmt.Fprintf(&b, " confusion: %s\n", renderCounts(r.Confusion)) } + if len(r.SourceConfusion) > 0 { + fmt.Fprintf(&b, " destination confusion: %s\n", renderCounts(r.SourceConfusion)) + } return b.String() } -// Failures — the per-case detail, sorted by ID so two runs diff cleanly. +// Failures — the per-case detail, sorted by ID so two runs diff cleanly. A case +// that landed its intent and missed its destination is listed too, marked, so +// the half that moved is readable without diffing two percentages. func (r Report) Failures() string { var b strings.Builder out := append([]Outcome(nil), r.Outcomes...) sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID }) for _, o := range out { - if o.Pass { - continue + switch { + case !o.Pass: + reasons := o.Reasons + if o.SourceReason != "" { + reasons = append(append([]string(nil), reasons...), o.SourceReason) + } + fmt.Fprintf(&b, " %s %q: %s\n", o.Case.ID, o.Case.Utterance, strings.Join(reasons, "; ")) + case o.SourceReason != "": + fmt.Fprintf(&b, " %s %q: route ok, %s\n", o.Case.ID, o.Case.Utterance, o.SourceReason) } - fmt.Fprintf(&b, " %s %q: %s\n", o.Case.ID, o.Case.Utterance, strings.Join(o.Reasons, "; ")) } return b.String() }