Score the destination apart from the intent (V-659)
The fixture measured the first half of a route and stopped. V-655 split a routing decision in two, and the second half arrived with no fixture, so Decision.Source had no accuracy number at all. want_source is a pointer because the destination has three states and a bare string has two. Absent is every intent but query, which never reaches queryWalk. Present and empty is the SourceUnknown contract: name nothing and let the daemon walk the chain, which is right whenever two destinations can both answer and the utterance does not choose. Present and named is a destination the route must produce. A destination miss does not fail the case. It goes in SourceReason, never in Reasons, so Accuracy and IntentAccuracy stay the numbers they were and 69/91 still means what it meant. SourceAccuracy is the second number, over the labelled cases only, because a percentage of the whole fixture would be a percentage of turns that never ask a query source. A clarified or mis-routed case still counts in the denominator. It named no destination and that is a miss, not a case to skip, or the denominator drops every turn the route already lost. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013ptwopxyo3Z2kwFckHkLvN
This commit is contained in:
@@ -36,17 +36,27 @@ var fixtureJSON []byte
|
||||
//
|
||||
// Intent is empty exactly when WantClarify is set: the contract there is that
|
||||
// the router refuses instead of guessing.
|
||||
//
|
||||
// WantSource is a pointer because the destination has three states and a bare
|
||||
// string only has two (V-659). Absent means the case does not score a
|
||||
// destination at all, which is every intent but query: a fact, a reminder, a
|
||||
// note, an act, a chat or a system turn never reaches queryWalk. Present and
|
||||
// empty is the SourceUnknown contract — the decider must name nothing and let
|
||||
// the daemon walk the whole chain, which is the right answer whenever two
|
||||
// destinations can both answer and the utterance does not choose. Present and
|
||||
// named is a destination the route must produce.
|
||||
type Case struct {
|
||||
ID string `json:"id"`
|
||||
Utterance string `json:"utterance"`
|
||||
Lang string `json:"lang"`
|
||||
Intent router.Intent `json:"intent"`
|
||||
WantTime bool `json:"want_time"`
|
||||
WantFn bool `json:"want_fn"`
|
||||
WantFactKey string `json:"want_fact_key"`
|
||||
WantClarify bool `json:"want_clarify"`
|
||||
Tags []string `json:"tags"`
|
||||
Note string `json:"note"`
|
||||
ID string `json:"id"`
|
||||
Utterance string `json:"utterance"`
|
||||
Lang string `json:"lang"`
|
||||
Intent router.Intent `json:"intent"`
|
||||
WantTime bool `json:"want_time"`
|
||||
WantFn bool `json:"want_fn"`
|
||||
WantFactKey string `json:"want_fact_key"`
|
||||
WantClarify bool `json:"want_clarify"`
|
||||
WantSource *router.Source `json:"want_source,omitempty"`
|
||||
Tags []string `json:"tags"`
|
||||
Note string `json:"note"`
|
||||
}
|
||||
|
||||
// Fixture — the versioned envelope, same shape as
|
||||
@@ -118,6 +128,11 @@ type Outcome struct {
|
||||
// (a slot gap is a parser fix; a wrong intent is a router fix).
|
||||
IntentOK bool
|
||||
Reasons []string
|
||||
// SourceReason is set when the case labelled a destination and the route
|
||||
// named a different one. It is kept out of Reasons on purpose: the
|
||||
// destination is the second half of a route and it is scored separately,
|
||||
// so a wrong destination must not move the intent number (V-659).
|
||||
SourceReason string
|
||||
}
|
||||
|
||||
// Report — the aggregate. Accuracy is the headline; the rest exists so a
|
||||
@@ -139,7 +154,15 @@ type Report struct {
|
||||
// (reminder grammar → applyAction's time parser). Not a miss, but not a
|
||||
// full router-level win either; tracked so the two aren't conflated.
|
||||
SlotsDeferred int
|
||||
Outcomes []Outcome
|
||||
// SourceTotal counts the cases carrying a want_source, and SourceHit the
|
||||
// ones whose route named it. Reported apart from Passed because intent and
|
||||
// destination are two decisions, and one number hides which one moved.
|
||||
SourceTotal int
|
||||
SourceHit int
|
||||
// SourceConfusion counts want→got destination pairs. "" reads as the
|
||||
// SourceUnknown floor on either side.
|
||||
SourceConfusion map[string]int
|
||||
Outcomes []Outcome
|
||||
// Confusion counts want→got intent pairs, decided cases only.
|
||||
Confusion map[string]int
|
||||
// ByTag accuracy for the fixture's tags ("hard", "homelab", …).
|
||||
@@ -172,6 +195,17 @@ func (r Report) IntentAccuracy() float64 {
|
||||
return float64(r.IntentHit) / float64(r.Total)
|
||||
}
|
||||
|
||||
// SourceAccuracy — fraction of the labelled cases whose route named the right
|
||||
// destination. Denominator is SourceTotal and not Total, because most of the
|
||||
// fixture never reaches a query source and scoring those would report a
|
||||
// percentage of nothing.
|
||||
func (r Report) SourceAccuracy() float64 {
|
||||
if r.SourceTotal == 0 {
|
||||
return 0
|
||||
}
|
||||
return float64(r.SourceHit) / float64(r.SourceTotal)
|
||||
}
|
||||
|
||||
// Score runs every case through r and aggregates. It never fails the run on a
|
||||
// route error — an erroring case scores as a miss and is counted in Errors,
|
||||
// because "the model was down" and "the model was wrong" are different numbers
|
||||
@@ -186,11 +220,12 @@ func Score(ctx context.Context, name string, r Router, f Fixture) (Report, error
|
||||
return Report{}, err
|
||||
}
|
||||
rep := Report{
|
||||
Name: name,
|
||||
Total: len(f.Cases),
|
||||
Confusion: map[string]int{},
|
||||
ByTag: map[string]TagStat{},
|
||||
ByLang: map[string]TagStat{},
|
||||
Name: name,
|
||||
Total: len(f.Cases),
|
||||
Confusion: map[string]int{},
|
||||
SourceConfusion: map[string]int{},
|
||||
ByTag: map[string]TagStat{},
|
||||
ByLang: map[string]TagStat{},
|
||||
}
|
||||
lat := make([]time.Duration, 0, len(f.Cases))
|
||||
|
||||
@@ -242,6 +277,23 @@ func Score(ctx context.Context, name string, r Router, f Fixture) (Report, error
|
||||
}
|
||||
}
|
||||
|
||||
// The destination is scored outside the switch and outside Pass. A case
|
||||
// that clarified or landed the wrong intent named no destination, and
|
||||
// that is a real miss rather than a case to skip — otherwise the
|
||||
// denominator quietly drops every turn the route already lost. Only a
|
||||
// route error is skipped, because "the model was down" is the Errors
|
||||
// number and not a destination result.
|
||||
if c.WantSource != nil && err == nil {
|
||||
rep.SourceTotal++
|
||||
switch {
|
||||
case d.Source == *c.WantSource:
|
||||
rep.SourceHit++
|
||||
default:
|
||||
rep.SourceConfusion[string(*c.WantSource)+"→"+string(d.Source)]++
|
||||
o.SourceReason = fmt.Sprintf("source %q, want %q", d.Source, *c.WantSource)
|
||||
}
|
||||
}
|
||||
|
||||
o.Pass = len(o.Reasons) == 0
|
||||
if o.Pass {
|
||||
rep.Passed++
|
||||
@@ -298,25 +350,40 @@ func (r Report) String() string {
|
||||
r.Name, r.Passed, r.Total, 100*r.Accuracy(), 100*r.IntentAccuracy())
|
||||
fmt.Fprintf(&b, " clarify: %d false (asked, shouldn't) / %d missed (guessed, shouldn't) | errors: %d | slots deferred to daemon: %d\n",
|
||||
r.FalseClarify, r.MissedClarify, r.Errors, r.SlotsDeferred)
|
||||
if r.SourceTotal > 0 {
|
||||
fmt.Fprintf(&b, " destination: %d/%d labelled cases (%.1f%%)\n",
|
||||
r.SourceHit, r.SourceTotal, 100*r.SourceAccuracy())
|
||||
}
|
||||
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
|
||||
fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang))
|
||||
fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag))
|
||||
if len(r.Confusion) > 0 {
|
||||
fmt.Fprintf(&b, " confusion: %s\n", renderCounts(r.Confusion))
|
||||
}
|
||||
if len(r.SourceConfusion) > 0 {
|
||||
fmt.Fprintf(&b, " destination confusion: %s\n", renderCounts(r.SourceConfusion))
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// Failures — the per-case detail, sorted by ID so two runs diff cleanly.
|
||||
// Failures — the per-case detail, sorted by ID so two runs diff cleanly. A case
|
||||
// that landed its intent and missed its destination is listed too, marked, so
|
||||
// the half that moved is readable without diffing two percentages.
|
||||
func (r Report) Failures() string {
|
||||
var b strings.Builder
|
||||
out := append([]Outcome(nil), r.Outcomes...)
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID })
|
||||
for _, o := range out {
|
||||
if o.Pass {
|
||||
continue
|
||||
switch {
|
||||
case !o.Pass:
|
||||
reasons := o.Reasons
|
||||
if o.SourceReason != "" {
|
||||
reasons = append(append([]string(nil), reasons...), o.SourceReason)
|
||||
}
|
||||
fmt.Fprintf(&b, " %s %q: %s\n", o.Case.ID, o.Case.Utterance, strings.Join(reasons, "; "))
|
||||
case o.SourceReason != "":
|
||||
fmt.Fprintf(&b, " %s %q: route ok, %s\n", o.Case.ID, o.Case.Utterance, o.SourceReason)
|
||||
}
|
||||
fmt.Fprintf(&b, " %s %q: %s\n", o.Case.ID, o.Case.Utterance, strings.Join(o.Reasons, "; "))
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user