Harden semantic boundaries and repair dialogue state
Replace nearest-neighbour personal routing with a frozen class-balanced linear head measured on historical, stratified, cross-validation, holdout, and fresh challenge gates (V-702). Close the four repair handoff holes, preserve nested clarification flows, and route Russian possession statements through structural grammar rather than lexical exceptions (V-573). Owner explicitly requested direct commits to master.
This commit is contained in:
@@ -0,0 +1,405 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
_ "embed"
|
||||
"encoding/json"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// This fixture is intentionally separate from personalboundary_test.go. The
|
||||
// small regression table there explains individual fixes; this matrix measures
|
||||
// the boundary as a classifier and prevents a repaired sentence shape from
|
||||
// standing in for language and subject coverage.
|
||||
//
|
||||
//go:embed testdata/personal_boundary_v1.json
|
||||
var personalBoundaryFixtureJSON []byte
|
||||
|
||||
type personalBoundaryEvalCase struct {
|
||||
ID string `json:"id"`
|
||||
Utterance string `json:"utterance"`
|
||||
Lang string `json:"lang"`
|
||||
Want string `json:"want"`
|
||||
Stratum string `json:"stratum"`
|
||||
}
|
||||
|
||||
type personalBoundaryEvalFixture struct {
|
||||
SchemaVersion int `json:"schema_version"`
|
||||
Name string `json:"name"`
|
||||
Notes []string `json:"notes"`
|
||||
Cases []personalBoundaryEvalCase `json:"cases"`
|
||||
}
|
||||
|
||||
var personalBoundaryEvalStrata = []string{
|
||||
"remembered_speech",
|
||||
"possession",
|
||||
"narrative",
|
||||
"first_person_preamble",
|
||||
"advice_current_info",
|
||||
"public_proper_nouns",
|
||||
}
|
||||
|
||||
func loadPersonalBoundaryEvalFixture(t *testing.T) personalBoundaryEvalFixture {
|
||||
t.Helper()
|
||||
var fixture personalBoundaryEvalFixture
|
||||
if err := json.Unmarshal(personalBoundaryFixtureJSON, &fixture); err != nil {
|
||||
t.Fatalf("parse personal boundary fixture: %v", err)
|
||||
}
|
||||
if fixture.SchemaVersion != 1 {
|
||||
t.Fatalf("personal boundary fixture schema_version = %d, want 1", fixture.SchemaVersion)
|
||||
}
|
||||
if fixture.Name != "personal_boundary_v1" {
|
||||
t.Fatalf("personal boundary fixture name = %q, want personal_boundary_v1", fixture.Name)
|
||||
}
|
||||
return fixture
|
||||
}
|
||||
|
||||
// TestPersonalBoundaryEvalFixture enforces the sampling contract separately
|
||||
// from the model measurement. It runs in ordinary CI even when ONNX Runtime is
|
||||
// absent, so a fixture edit cannot silently unbalance a language, side or
|
||||
// sentence shape, or turn a production seed into a held-out case.
|
||||
func TestPersonalBoundaryEvalFixture(t *testing.T) {
|
||||
fixture := loadPersonalBoundaryEvalFixture(t)
|
||||
const wantPerCell = 3
|
||||
const wantTotal = 6 * 2 * 2 * wantPerCell
|
||||
if len(fixture.Cases) != wantTotal {
|
||||
t.Errorf("fixture has %d cases, want %d", len(fixture.Cases), wantTotal)
|
||||
}
|
||||
|
||||
validStrata := make(map[string]bool, len(personalBoundaryEvalStrata))
|
||||
for _, stratum := range personalBoundaryEvalStrata {
|
||||
validStrata[stratum] = true
|
||||
}
|
||||
seedSource := make(map[string]string, len(personalSeeds)+len(worldSeeds))
|
||||
for _, seed := range personalSeeds {
|
||||
seedSource[normalizePersonalBoundaryEval(seed)] = "personalSeeds"
|
||||
}
|
||||
for _, seed := range worldSeeds {
|
||||
seedSource[normalizePersonalBoundaryEval(seed)] = "worldSeeds"
|
||||
}
|
||||
|
||||
seenID := make(map[string]bool, len(fixture.Cases))
|
||||
seenUtterance := make(map[string]string, len(fixture.Cases))
|
||||
cells := make(map[string]int)
|
||||
for _, c := range fixture.Cases {
|
||||
if strings.TrimSpace(c.ID) == "" || seenID[c.ID] {
|
||||
t.Errorf("case %q: empty or duplicate id", c.ID)
|
||||
}
|
||||
seenID[c.ID] = true
|
||||
if c.Lang != "ru" && c.Lang != "en" {
|
||||
t.Errorf("%s: lang = %q, want ru|en", c.ID, c.Lang)
|
||||
}
|
||||
if c.Want != "personal" && c.Want != "world" {
|
||||
t.Errorf("%s: want = %q, want personal|world", c.ID, c.Want)
|
||||
}
|
||||
if !validStrata[c.Stratum] {
|
||||
t.Errorf("%s: stratum = %q, not one of the six declared strata", c.ID, c.Stratum)
|
||||
}
|
||||
|
||||
normalized := normalizePersonalBoundaryEval(c.Utterance)
|
||||
if normalized == "" {
|
||||
t.Errorf("%s: empty utterance", c.ID)
|
||||
}
|
||||
if previous, ok := seenUtterance[normalized]; ok {
|
||||
t.Errorf("%s: utterance duplicates %s after normalization", c.ID, previous)
|
||||
}
|
||||
seenUtterance[normalized] = c.ID
|
||||
if source, ok := seedSource[normalized]; ok {
|
||||
t.Errorf("%s: %q is verbatim in %s, so it is not held out", c.ID, c.Utterance, source)
|
||||
}
|
||||
// The original failure names Baikal. Replacing that sentence's verb or
|
||||
// punctuation would measure an exception, not the boundary. This corpus
|
||||
// instead varies people, places, products and events.
|
||||
if strings.Contains(normalized, "байкал") || strings.Contains(normalized, "baikal") {
|
||||
t.Errorf("%s: the stratified fixture must not copy the Baikal regression", c.ID)
|
||||
}
|
||||
cells[c.Stratum+"/"+c.Lang+"/"+c.Want]++
|
||||
}
|
||||
|
||||
for _, stratum := range personalBoundaryEvalStrata {
|
||||
for _, lang := range []string{"ru", "en"} {
|
||||
for _, want := range []string{"personal", "world"} {
|
||||
cell := stratum + "/" + lang + "/" + want
|
||||
if got := cells[cell]; got != wantPerCell {
|
||||
t.Errorf("fixture cell %s has %d cases, want %d", cell, got, wantPerCell)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// normalizePersonalBoundaryEval compares content rather than typography:
|
||||
// case, punctuation and repeated whitespace cannot disguise a copied seed or
|
||||
// duplicate case. This is fixture hygiene only; it does not participate in the
|
||||
// production boundary.
|
||||
func normalizePersonalBoundaryEval(s string) string {
|
||||
var b strings.Builder
|
||||
space := true
|
||||
for _, r := range strings.ToLower(s) {
|
||||
if unicode.IsLetter(r) || unicode.IsNumber(r) {
|
||||
b.WriteRune(r)
|
||||
space = false
|
||||
continue
|
||||
}
|
||||
if !space {
|
||||
b.WriteByte(' ')
|
||||
space = true
|
||||
}
|
||||
}
|
||||
return strings.TrimSpace(b.String())
|
||||
}
|
||||
|
||||
type personalBoundaryEvalStat struct {
|
||||
Correct int
|
||||
Total int
|
||||
}
|
||||
|
||||
type personalBoundaryEvalReport struct {
|
||||
Name string
|
||||
Correct int
|
||||
Total int
|
||||
MinimumMargin float64
|
||||
ByStratum map[string]personalBoundaryEvalStat
|
||||
ByLanguage map[string]personalBoundaryEvalStat
|
||||
ByExpectedClass map[string]personalBoundaryEvalStat
|
||||
ByCell map[string]personalBoundaryEvalStat
|
||||
}
|
||||
|
||||
func newPersonalBoundaryEvalReport(name string) *personalBoundaryEvalReport {
|
||||
return &personalBoundaryEvalReport{
|
||||
Name: name,
|
||||
MinimumMargin: math.Inf(1),
|
||||
ByStratum: make(map[string]personalBoundaryEvalStat),
|
||||
ByLanguage: make(map[string]personalBoundaryEvalStat),
|
||||
ByExpectedClass: make(map[string]personalBoundaryEvalStat),
|
||||
ByCell: make(map[string]personalBoundaryEvalStat),
|
||||
}
|
||||
}
|
||||
|
||||
func (r *personalBoundaryEvalReport) add(c personalBoundaryEvalCase, gotPersonal bool, personal, world float64) {
|
||||
wantPersonal := c.Want == "personal"
|
||||
correct := gotPersonal == wantPersonal
|
||||
r.Total++
|
||||
if correct {
|
||||
r.Correct++
|
||||
}
|
||||
signedMargin := personal - world
|
||||
if !wantPersonal {
|
||||
signedMargin = -signedMargin
|
||||
}
|
||||
if signedMargin < r.MinimumMargin {
|
||||
r.MinimumMargin = signedMargin
|
||||
}
|
||||
add := func(stats map[string]personalBoundaryEvalStat, key string) {
|
||||
stat := stats[key]
|
||||
stat.Total++
|
||||
if correct {
|
||||
stat.Correct++
|
||||
}
|
||||
stats[key] = stat
|
||||
}
|
||||
add(r.ByStratum, c.Stratum)
|
||||
add(r.ByLanguage, c.Lang)
|
||||
add(r.ByExpectedClass, c.Want)
|
||||
add(r.ByCell, c.Stratum+"/"+c.Lang+"/"+c.Want)
|
||||
}
|
||||
|
||||
// TestONNXPersonalBoundaryStratified scores the model homesrv actually runs.
|
||||
// Production is read from personalBoundary.score; top1, top2, top3 and a
|
||||
// whole-class centroid are diagnostics over the same embedded seeds. Today
|
||||
// production and top3 coincide, but keeping them separate means a later scoring
|
||||
// experiment can be compared without rewriting this evaluation or putting its
|
||||
// candidate math in runtime code. The privacy boundary is a hard contract, so
|
||||
// every production miss is a test failure rather than an accuracy target to
|
||||
// average away.
|
||||
func TestONNXPersonalBoundaryStratified(t *testing.T) {
|
||||
if os.Getenv("MAVEN_EVAL_PERSONAL_BOUNDARY") == "" {
|
||||
t.Skip("set MAVEN_EVAL_PERSONAL_BOUNDARY=1 to run the deliberately strict V-702 matrix")
|
||||
}
|
||||
lib := os.Getenv("MAVEN_ONNX_LIB")
|
||||
if lib == "" {
|
||||
t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing")
|
||||
}
|
||||
modelDir := filepath.Join("../..", "models/embedder/multilingual-e5-small")
|
||||
model := filepath.Join(modelDir, "model_quantized.onnx")
|
||||
tokenizer := filepath.Join(modelDir, "tokenizer.json")
|
||||
for _, path := range []string{lib, model, tokenizer} {
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
t.Skipf("personal boundary eval dependency %s unavailable: %v", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
embedder, err := router.NewONNXEmbedder(model, tokenizer, lib)
|
||||
if err != nil {
|
||||
t.Skipf("onnx embedder unavailable: %v", err)
|
||||
}
|
||||
defer embedder.Close()
|
||||
|
||||
ctx := context.Background()
|
||||
boundary := &personalBoundary{}
|
||||
boundary.load(ctx, embedder)
|
||||
if !boundary.loaded {
|
||||
t.Fatal("personal boundary seeds did not load with a working embedder")
|
||||
}
|
||||
// Production loads its model-ID-pinned frozen head and deliberately skips
|
||||
// the 132 corpus embeddings on a user's first query. This test still needs
|
||||
// those vectors for the historical top-k/centroid diagnostics, so build
|
||||
// them here without putting that latency back in runtime code.
|
||||
embedCorpus := func(values []string) [][]float32 {
|
||||
vectors := make([][]float32, len(values))
|
||||
for i, value := range values {
|
||||
vector, err := router.EmbedQuery(ctx, embedder, value)
|
||||
if err != nil {
|
||||
t.Fatalf("embed diagnostic corpus %q: %v", value, err)
|
||||
}
|
||||
vectors[i] = vector
|
||||
}
|
||||
return vectors
|
||||
}
|
||||
boundary.personal = embedCorpus(personalSeeds)
|
||||
boundary.world = embedCorpus(worldSeeds)
|
||||
personalCentroid := personalBoundaryEvalCentroid(boundary.personal)
|
||||
worldCentroid := personalBoundaryEvalCentroid(boundary.world)
|
||||
if len(personalCentroid) == 0 || len(worldCentroid) == 0 {
|
||||
t.Fatal("personal boundary seed vectors do not share a dimension")
|
||||
}
|
||||
|
||||
type candidate struct {
|
||||
name string
|
||||
score func([]float32) (float64, float64)
|
||||
}
|
||||
candidates := []candidate{
|
||||
{name: "production", score: func(vec []float32) (float64, float64) {
|
||||
personal, world, ok := boundary.score(vec)
|
||||
if !ok {
|
||||
t.Fatal("loaded personal boundary declined to score")
|
||||
}
|
||||
return personal, world
|
||||
}},
|
||||
{name: "top1", score: func(vec []float32) (float64, float64) {
|
||||
return meanNearest(vec, boundary.personal, 1), meanNearest(vec, boundary.world, 1)
|
||||
}},
|
||||
{name: "top2", score: func(vec []float32) (float64, float64) {
|
||||
return meanNearest(vec, boundary.personal, 2), meanNearest(vec, boundary.world, 2)
|
||||
}},
|
||||
{name: "top3", score: func(vec []float32) (float64, float64) {
|
||||
return meanNearest(vec, boundary.personal, 3), meanNearest(vec, boundary.world, 3)
|
||||
}},
|
||||
{name: "centroid", score: func(vec []float32) (float64, float64) {
|
||||
return cosine(vec, personalCentroid), cosine(vec, worldCentroid)
|
||||
}},
|
||||
}
|
||||
reports := make(map[string]*personalBoundaryEvalReport, len(candidates))
|
||||
for _, candidate := range candidates {
|
||||
reports[candidate.name] = newPersonalBoundaryEvalReport(candidate.name)
|
||||
}
|
||||
|
||||
fixture := loadPersonalBoundaryEvalFixture(t)
|
||||
for _, c := range fixture.Cases {
|
||||
vec, err := router.EmbedQuery(ctx, embedder, c.Utterance)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: embed %q: %v", c.ID, c.Utterance, err)
|
||||
}
|
||||
for _, candidate := range candidates {
|
||||
personal, world := candidate.score(vec)
|
||||
gotPersonal := personal > world
|
||||
reports[candidate.name].add(c, gotPersonal, personal, world)
|
||||
if candidate.name == "production" && gotPersonal != (c.Want == "personal") {
|
||||
t.Errorf("%s [%s/%s]: got %s, want %s (personal %.4f world %.4f delta %+.4f): %q",
|
||||
c.ID, c.Lang, c.Stratum, boundaryEvalSide(gotPersonal), c.Want,
|
||||
personal, world, personal-world, c.Utterance)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for _, candidate := range candidates {
|
||||
report := reports[candidate.name]
|
||||
t.Logf("candidate %-15s %2d/%d (%.1f%%), minimum signed margin %+.4f",
|
||||
report.Name, report.Correct, report.Total,
|
||||
100*float64(report.Correct)/float64(report.Total), report.MinimumMargin)
|
||||
}
|
||||
production := reports["production"]
|
||||
for _, lang := range []string{"ru", "en"} {
|
||||
stat := production.ByLanguage[lang]
|
||||
t.Logf("production language %-2s %2d/%d", lang, stat.Correct, stat.Total)
|
||||
}
|
||||
for _, side := range []string{"personal", "world"} {
|
||||
stat := production.ByExpectedClass[side]
|
||||
t.Logf("production expected %-8s %2d/%d", side, stat.Correct, stat.Total)
|
||||
}
|
||||
strata := append([]string(nil), personalBoundaryEvalStrata...)
|
||||
sort.Strings(strata)
|
||||
for _, stratum := range strata {
|
||||
stat := production.ByStratum[stratum]
|
||||
ruPersonal := production.ByCell[stratum+"/ru/personal"]
|
||||
ruWorld := production.ByCell[stratum+"/ru/world"]
|
||||
enPersonal := production.ByCell[stratum+"/en/personal"]
|
||||
enWorld := production.ByCell[stratum+"/en/world"]
|
||||
t.Logf("production stratum %-21s %2d/%d | ru personal %d/%d world %d/%d | en personal %d/%d world %d/%d",
|
||||
stratum, stat.Correct, stat.Total,
|
||||
ruPersonal.Correct, ruPersonal.Total, ruWorld.Correct, ruWorld.Total,
|
||||
enPersonal.Correct, enPersonal.Total, enWorld.Correct, enWorld.Total)
|
||||
}
|
||||
}
|
||||
|
||||
func personalBoundaryEvalCentroid(vectors [][]float32) []float32 {
|
||||
if len(vectors) == 0 {
|
||||
return nil
|
||||
}
|
||||
centroid := make([]float32, len(vectors[0]))
|
||||
for _, vector := range vectors {
|
||||
if len(vector) != len(centroid) {
|
||||
return nil
|
||||
}
|
||||
for i, value := range vector {
|
||||
centroid[i] += value
|
||||
}
|
||||
}
|
||||
for i := range centroid {
|
||||
centroid[i] /= float32(len(vectors))
|
||||
}
|
||||
return centroid
|
||||
}
|
||||
|
||||
func boundaryEvalSide(personal bool) string {
|
||||
if personal {
|
||||
return "personal"
|
||||
}
|
||||
return "world"
|
||||
}
|
||||
|
||||
// meanNearest is an evaluation baseline retained beside the strict fixture;
|
||||
// production uses the linear head in personalboundary.go.
|
||||
func meanNearest(vec []float32, seeds [][]float32, k int) float64 {
|
||||
if len(seeds) == 0 || k <= 0 {
|
||||
return -1
|
||||
}
|
||||
if k > len(seeds) {
|
||||
k = len(seeds)
|
||||
}
|
||||
top := make([]float64, k)
|
||||
for i := range top {
|
||||
top[i] = -1
|
||||
}
|
||||
for _, seed := range seeds {
|
||||
candidate := cosine(vec, seed)
|
||||
for i := range top {
|
||||
if candidate > top[i] {
|
||||
candidate, top[i] = top[i], candidate
|
||||
}
|
||||
}
|
||||
}
|
||||
var sum float64
|
||||
for _, similarity := range top {
|
||||
sum += similarity
|
||||
}
|
||||
return sum / float64(k)
|
||||
}
|
||||
Reference in New Issue
Block a user