measure what each claimant on an utterance reports (V-565)
Two reporting tests over the 91-case RU fixture, no ratchet: a ratchet here would freeze a number nobody has decided to hold. TestStage0Contention runs the 21 grammars one at a time instead of stopping at the first match. One case of 91 draws two, ru-query-019, where calendar-query beats agenda-query by list position alone. TestONNXClaimConfidenceDistribution buckets the reported confidence by the layer that produced it. Stage 0 is 20/20 at a hardcoded 1.0. The classifier scores 62% below its median and 62% above, across a cosine range of 0.859 to 0.942, with a top-two margin of p50 0.009. The float is not a confidence. newBaselineClassifier and baselineGrammars split out of newBaselineRouter so the measurement runs the same rules the daemon runs. TestONNXBaseline is unchanged at 64/91.
This commit is contained in:
@@ -0,0 +1,230 @@
|
||||
package eval
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// Package-level note for V-565. The cascade's arbitration is list order, and
|
||||
// the reason is that no two claimants report a comparable number. These tests
|
||||
// measure what each claimant actually reports across the 91-case RU fixture,
|
||||
// so the ordinal band set in docs/plans/19-dialogue-arbitration.md is argued
|
||||
// from a distribution rather than from taste. They report and never assert:
|
||||
// a ratchet here would freeze a number nobody has decided to hold yet.
|
||||
|
||||
// TestStage0Contention — how often more than one stage-0 grammar matches the
|
||||
// same utterance. Every one of them reports Confidence 1.0, so where two
|
||||
// match, list order is the entire decision and nothing in the Decision says a
|
||||
// second rule wanted the turn.
|
||||
func TestStage0Contention(t *testing.T) {
|
||||
f, err := Load()
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
grammars := baselineGrammars(router.DefaultActMatcher{Fns: actFns})
|
||||
t.Logf("stage 0: %d grammars over %d cases", len(grammars), len(f.Cases))
|
||||
|
||||
matched, contended := 0, 0
|
||||
pairs := map[string]int{}
|
||||
for _, c := range f.Cases {
|
||||
claimants := matchingGrammars(grammars, c.Utterance)
|
||||
if len(claimants) == 0 {
|
||||
continue
|
||||
}
|
||||
matched++
|
||||
if len(claimants) < 2 {
|
||||
continue
|
||||
}
|
||||
contended++
|
||||
t.Logf(" contended %s %q: %v (winner %q by order)", c.ID, c.Utterance, claimants, claimants[0])
|
||||
for _, loser := range claimants[1:] {
|
||||
pairs[claimants[0]+" beats "+loser]++
|
||||
}
|
||||
}
|
||||
t.Logf("stage 0 claimed %d/%d cases, %d of those with more than one claimant", matched, len(f.Cases), contended)
|
||||
for _, k := range sortedKeys(pairs) {
|
||||
t.Logf(" %s ×%d", k, pairs[k])
|
||||
}
|
||||
}
|
||||
|
||||
// matchingGrammars — every grammar whose pattern matches AND whose Build
|
||||
// accepts, in the daemon's order. Route stops at the first; this does not.
|
||||
func matchingGrammars(grammars []router.Grammar, utterance string) []string {
|
||||
stripped, hadWake := router.StripWakeToken(utterance)
|
||||
var out []string
|
||||
for _, g := range grammars {
|
||||
m := g.Pattern.FindStringSubmatch(utterance)
|
||||
if m == nil && hadWake {
|
||||
m = g.Pattern.FindStringSubmatch(stripped)
|
||||
}
|
||||
if m == nil {
|
||||
continue
|
||||
}
|
||||
if _, ok := g.Build(m); !ok {
|
||||
continue
|
||||
}
|
||||
out = append(out, g.Name)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestClaimConfidenceDistributionHash — the confidence each claimant reports,
|
||||
// on the deterministic hash embedder so it runs anywhere. The ONNX run below
|
||||
// is the one whose cosines are the deployed numbers.
|
||||
func TestClaimConfidenceDistributionHash(t *testing.T) {
|
||||
reportConfidences(t, "hash", router.NewHashEmbedder(1024))
|
||||
}
|
||||
|
||||
// TestONNXClaimConfidenceDistribution — the same measurement on the embedder
|
||||
// homesrv runs, so the cosine column is the real one. Opt-in via
|
||||
// MAVEN_ONNX_LIB, same as TestONNXBaseline, and one TestONNX* per process.
|
||||
func TestONNXClaimConfidenceDistribution(t *testing.T) {
|
||||
lib := os.Getenv("MAVEN_ONNX_LIB")
|
||||
if lib == "" {
|
||||
t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing")
|
||||
}
|
||||
model := filepath.Join("../../..", "models/embedder/multilingual-e5-small/model_quantized.onnx")
|
||||
tok := filepath.Join("../../..", "models/embedder/multilingual-e5-small/tokenizer.json")
|
||||
for _, p := range []string{lib, model, tok} {
|
||||
if _, err := os.Stat(p); err != nil {
|
||||
t.Skipf("missing %s: %v", p, err)
|
||||
}
|
||||
}
|
||||
emb, err := router.NewONNXEmbedder(model, tok, lib)
|
||||
if err != nil {
|
||||
t.Skipf("onnx embedder unavailable: %v", err)
|
||||
}
|
||||
defer emb.Close()
|
||||
reportConfidences(t, "onnx", emb)
|
||||
}
|
||||
|
||||
// reportConfidences runs the fixture through the deployed cascade and buckets
|
||||
// the reported confidence by which layer produced it, then reports how well
|
||||
// each bucket predicts a correct route. A band is only worth defining if the
|
||||
// accuracy inside it differs from the accuracy outside it.
|
||||
func reportConfidences(t *testing.T, name string, emb router.Embedder) {
|
||||
t.Helper()
|
||||
f, err := Load()
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
now, err := f.Now()
|
||||
if err != nil {
|
||||
t.Fatalf("Now: %v", err)
|
||||
}
|
||||
r := newBaselineRouter(t, emb, nil)
|
||||
cls := newBaselineClassifier(t, emb)
|
||||
|
||||
type bucket struct{ n, correct int }
|
||||
byValue := map[string]*bucket{}
|
||||
byMargin := map[string]*bucket{}
|
||||
var cosines, margins []float64
|
||||
for _, c := range f.Cases {
|
||||
d, err := r.Route(context.Background(), c.Utterance, now)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: %v", c.ID, err)
|
||||
}
|
||||
layer := "classifier"
|
||||
if d.Stage == 0 {
|
||||
layer = "stage0"
|
||||
} else {
|
||||
cosines = append(cosines, d.Confidence)
|
||||
}
|
||||
key := fmt.Sprintf("%s conf=%.2f", layer, d.Confidence)
|
||||
if layer == "classifier" {
|
||||
key = fmt.Sprintf("%s conf=%.1f..%.1f", layer, floorTo(d.Confidence, 0.1), floorTo(d.Confidence, 0.1)+0.1)
|
||||
}
|
||||
b := byValue[key]
|
||||
if b == nil {
|
||||
b = &bucket{}
|
||||
byValue[key] = b
|
||||
}
|
||||
ok := routeCorrect(c, d)
|
||||
b.n++
|
||||
if ok {
|
||||
b.correct++
|
||||
}
|
||||
|
||||
// The margin between the classifier's top two intents is the other
|
||||
// float one could call a confidence. Measured on the same cases, so
|
||||
// the ledger's "is a calibrated float available cheaply" question
|
||||
// gets an answer instead of an assumption.
|
||||
if layer != "classifier" {
|
||||
continue
|
||||
}
|
||||
res, err := cls.Classify(context.Background(), c.Utterance)
|
||||
if err != nil || len(res) < 2 {
|
||||
continue
|
||||
}
|
||||
margin := res[0].Score - res[1].Score
|
||||
margins = append(margins, margin)
|
||||
mk := fmt.Sprintf("margin %.2f..%.2f", floorTo(margin, 0.02), floorTo(margin, 0.02)+0.02)
|
||||
mb := byMargin[mk]
|
||||
if mb == nil {
|
||||
mb = &bucket{}
|
||||
byMargin[mk] = mb
|
||||
}
|
||||
mb.n++
|
||||
if ok {
|
||||
mb.correct++
|
||||
}
|
||||
}
|
||||
|
||||
t.Logf("%s: confidence buckets over %d cases (correct = right intent, or clarified when the fixture wants a refusal)", name, len(f.Cases))
|
||||
for _, k := range sortedKeys2(byValue) {
|
||||
b := byValue[k]
|
||||
t.Logf(" %-32s n=%2d correct=%2d (%.0f%%)", k, b.n, b.correct, 100*float64(b.correct)/float64(b.n))
|
||||
}
|
||||
if len(cosines) > 0 {
|
||||
sort.Float64s(cosines)
|
||||
t.Logf(" classifier cosine spread: min %.3f p25 %.3f p50 %.3f p75 %.3f max %.3f",
|
||||
cosines[0], cosines[len(cosines)/4], cosines[len(cosines)/2],
|
||||
cosines[3*len(cosines)/4], cosines[len(cosines)-1])
|
||||
}
|
||||
if len(margins) > 0 {
|
||||
sort.Float64s(margins)
|
||||
t.Logf(" classifier top1-top2 margin: min %.3f p50 %.3f max %.3f",
|
||||
margins[0], margins[len(margins)/2], margins[len(margins)-1])
|
||||
for _, k := range sortedKeys2(byMargin) {
|
||||
b := byMargin[k]
|
||||
t.Logf(" %-32s n=%2d correct=%2d (%.0f%%)", k, b.n, b.correct, 100*float64(b.correct)/float64(b.n))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// routeCorrect — the intent contract only. Slots are a parser question and
|
||||
// would blur what the confidence number is being asked to predict.
|
||||
func routeCorrect(c Case, d router.Decision) bool {
|
||||
if c.WantClarify {
|
||||
return d.Clarify
|
||||
}
|
||||
return d.Intent == c.Intent && !d.Clarify
|
||||
}
|
||||
|
||||
func floorTo(v, step float64) float64 {
|
||||
return float64(int(v/step)) * step
|
||||
}
|
||||
|
||||
func sortedKeys(m map[string]int) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
|
||||
func sortedKeys2[T any](m map[string]T) []string {
|
||||
out := make([]string, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
sort.Strings(out)
|
||||
return out
|
||||
}
|
||||
@@ -216,6 +216,27 @@ func TestONNXBaseline(t *testing.T) {
|
||||
func newBaselineRouter(t *testing.T, emb router.Embedder, llmR *router.LLMRouter) *router.Router {
|
||||
t.Helper()
|
||||
acts := router.DefaultActMatcher{Fns: actFns}
|
||||
cls := newBaselineClassifier(t, emb)
|
||||
return router.New(router.Config{
|
||||
Grammars: baselineGrammars(acts),
|
||||
Classifier: cls,
|
||||
Extractor: router.Extractor{
|
||||
Time: router.StubDateTimeParser{},
|
||||
Acts: acts,
|
||||
Facts: router.DefaultFactParser{},
|
||||
},
|
||||
// The deployed gate, not a test-local one: a fixture scored at a looser
|
||||
// threshold reports an accuracy no real turn would see.
|
||||
Threshold: config.DefaultRouterThreshold,
|
||||
LLM: llmR,
|
||||
})
|
||||
}
|
||||
|
||||
// newBaselineClassifier — the seeded nearest-centroid classifier the cascade
|
||||
// runs. Split out of newBaselineRouter so the claim measurement can ask it for
|
||||
// its full ranking, not just the winner the Decision carries.
|
||||
func newBaselineClassifier(t *testing.T, emb router.Embedder) *router.Classifier {
|
||||
t.Helper()
|
||||
cls := router.NewClassifier(emb)
|
||||
ctx := context.Background()
|
||||
seeds := seedsWithIntent(t)
|
||||
@@ -231,6 +252,15 @@ func newBaselineRouter(t *testing.T, emb router.Embedder, llmR *router.LLMRouter
|
||||
t.Fatalf("seed %q: %v", text, err)
|
||||
}
|
||||
}
|
||||
return cls
|
||||
}
|
||||
|
||||
// baselineGrammars — the stage-0 rule set in the daemon's order (buildRouter in
|
||||
// cmd/mavend/voicewire.go). Split out of newBaselineRouter so the claim
|
||||
// measurement can run the same rules one at a time and see which of them
|
||||
// contend for the same utterance, which the cascade hides by stopping at the
|
||||
// first match.
|
||||
func baselineGrammars(acts router.ActMatcher) []router.Grammar {
|
||||
grammars := router.DefaultGrammars(acts)
|
||||
grammars = append(grammars, router.SystemTimeDateGrammars()...)
|
||||
// Same order as buildRouter (voicewire.go). The fixture is only worth
|
||||
@@ -248,19 +278,7 @@ func newBaselineRouter(t *testing.T, emb router.Embedder, llmR *router.LLMRouter
|
||||
// "расскажи про X" is a world question the model called a fact, and the
|
||||
// rule goes last because it matches on the first word alone (Vikunja #498).
|
||||
grammars = append(grammars, router.NarrativeQueryGrammars()...)
|
||||
return router.New(router.Config{
|
||||
Grammars: grammars,
|
||||
Classifier: cls,
|
||||
Extractor: router.Extractor{
|
||||
Time: router.StubDateTimeParser{},
|
||||
Acts: acts,
|
||||
Facts: router.DefaultFactParser{},
|
||||
},
|
||||
// The deployed gate, not a test-local one: a fixture scored at a looser
|
||||
// threshold reports an accuracy no real turn would see.
|
||||
Threshold: config.DefaultRouterThreshold,
|
||||
LLM: llmR,
|
||||
})
|
||||
return grammars
|
||||
}
|
||||
|
||||
// seedOrder — fixed iteration order over the corpus. Not cosmetic: a few
|
||||
|
||||
Reference in New Issue
Block a user