package main import ( "context" _ "embed" "encoding/json" "math" "os" "path/filepath" "sort" "strings" "testing" "unicode" "github.com/kami/maven/internal/router" ) // This fixture is intentionally separate from personalboundary_test.go. The // small regression table there explains individual fixes; this matrix measures // the boundary as a classifier and prevents a repaired sentence shape from // standing in for language and subject coverage. // //go:embed testdata/personal_boundary_v1.json var personalBoundaryFixtureJSON []byte type personalBoundaryEvalCase struct { ID string `json:"id"` Utterance string `json:"utterance"` Lang string `json:"lang"` Want string `json:"want"` Stratum string `json:"stratum"` } type personalBoundaryEvalFixture struct { SchemaVersion int `json:"schema_version"` Name string `json:"name"` Notes []string `json:"notes"` Cases []personalBoundaryEvalCase `json:"cases"` } var personalBoundaryEvalStrata = []string{ "remembered_speech", "possession", "narrative", "first_person_preamble", "advice_current_info", "public_proper_nouns", } func loadPersonalBoundaryEvalFixture(t *testing.T) personalBoundaryEvalFixture { t.Helper() var fixture personalBoundaryEvalFixture if err := json.Unmarshal(personalBoundaryFixtureJSON, &fixture); err != nil { t.Fatalf("parse personal boundary fixture: %v", err) } if fixture.SchemaVersion != 1 { t.Fatalf("personal boundary fixture schema_version = %d, want 1", fixture.SchemaVersion) } if fixture.Name != "personal_boundary_v1" { t.Fatalf("personal boundary fixture name = %q, want personal_boundary_v1", fixture.Name) } return fixture } // TestPersonalBoundaryEvalFixture enforces the sampling contract separately // from the model measurement. It runs in ordinary CI even when ONNX Runtime is // absent, so a fixture edit cannot silently unbalance a language, side or // sentence shape, or turn a production seed into a held-out case. func TestPersonalBoundaryEvalFixture(t *testing.T) { fixture := loadPersonalBoundaryEvalFixture(t) const wantPerCell = 3 const wantTotal = 6 * 2 * 2 * wantPerCell if len(fixture.Cases) != wantTotal { t.Errorf("fixture has %d cases, want %d", len(fixture.Cases), wantTotal) } validStrata := make(map[string]bool, len(personalBoundaryEvalStrata)) for _, stratum := range personalBoundaryEvalStrata { validStrata[stratum] = true } seedSource := make(map[string]string, len(personalSeeds)+len(worldSeeds)) for _, seed := range personalSeeds { seedSource[normalizePersonalBoundaryEval(seed)] = "personalSeeds" } for _, seed := range worldSeeds { seedSource[normalizePersonalBoundaryEval(seed)] = "worldSeeds" } seenID := make(map[string]bool, len(fixture.Cases)) seenUtterance := make(map[string]string, len(fixture.Cases)) cells := make(map[string]int) for _, c := range fixture.Cases { if strings.TrimSpace(c.ID) == "" || seenID[c.ID] { t.Errorf("case %q: empty or duplicate id", c.ID) } seenID[c.ID] = true if c.Lang != "ru" && c.Lang != "en" { t.Errorf("%s: lang = %q, want ru|en", c.ID, c.Lang) } if c.Want != "personal" && c.Want != "world" { t.Errorf("%s: want = %q, want personal|world", c.ID, c.Want) } if !validStrata[c.Stratum] { t.Errorf("%s: stratum = %q, not one of the six declared strata", c.ID, c.Stratum) } normalized := normalizePersonalBoundaryEval(c.Utterance) if normalized == "" { t.Errorf("%s: empty utterance", c.ID) } if previous, ok := seenUtterance[normalized]; ok { t.Errorf("%s: utterance duplicates %s after normalization", c.ID, previous) } seenUtterance[normalized] = c.ID if source, ok := seedSource[normalized]; ok { t.Errorf("%s: %q is verbatim in %s, so it is not held out", c.ID, c.Utterance, source) } // The original failure names Baikal. Replacing that sentence's verb or // punctuation would measure an exception, not the boundary. This corpus // instead varies people, places, products and events. if strings.Contains(normalized, "байкал") || strings.Contains(normalized, "baikal") { t.Errorf("%s: the stratified fixture must not copy the Baikal regression", c.ID) } cells[c.Stratum+"/"+c.Lang+"/"+c.Want]++ } for _, stratum := range personalBoundaryEvalStrata { for _, lang := range []string{"ru", "en"} { for _, want := range []string{"personal", "world"} { cell := stratum + "/" + lang + "/" + want if got := cells[cell]; got != wantPerCell { t.Errorf("fixture cell %s has %d cases, want %d", cell, got, wantPerCell) } } } } } // normalizePersonalBoundaryEval compares content rather than typography: // case, punctuation and repeated whitespace cannot disguise a copied seed or // duplicate case. This is fixture hygiene only; it does not participate in the // production boundary. func normalizePersonalBoundaryEval(s string) string { var b strings.Builder space := true for _, r := range strings.ToLower(s) { if unicode.IsLetter(r) || unicode.IsNumber(r) { b.WriteRune(r) space = false continue } if !space { b.WriteByte(' ') space = true } } return strings.TrimSpace(b.String()) } type personalBoundaryEvalStat struct { Correct int Total int } type personalBoundaryEvalReport struct { Name string Correct int Total int MinimumMargin float64 ByStratum map[string]personalBoundaryEvalStat ByLanguage map[string]personalBoundaryEvalStat ByExpectedClass map[string]personalBoundaryEvalStat ByCell map[string]personalBoundaryEvalStat } func newPersonalBoundaryEvalReport(name string) *personalBoundaryEvalReport { return &personalBoundaryEvalReport{ Name: name, MinimumMargin: math.Inf(1), ByStratum: make(map[string]personalBoundaryEvalStat), ByLanguage: make(map[string]personalBoundaryEvalStat), ByExpectedClass: make(map[string]personalBoundaryEvalStat), ByCell: make(map[string]personalBoundaryEvalStat), } } func (r *personalBoundaryEvalReport) add(c personalBoundaryEvalCase, gotPersonal bool, personal, world float64) { wantPersonal := c.Want == "personal" correct := gotPersonal == wantPersonal r.Total++ if correct { r.Correct++ } signedMargin := personal - world if !wantPersonal { signedMargin = -signedMargin } if signedMargin < r.MinimumMargin { r.MinimumMargin = signedMargin } add := func(stats map[string]personalBoundaryEvalStat, key string) { stat := stats[key] stat.Total++ if correct { stat.Correct++ } stats[key] = stat } add(r.ByStratum, c.Stratum) add(r.ByLanguage, c.Lang) add(r.ByExpectedClass, c.Want) add(r.ByCell, c.Stratum+"/"+c.Lang+"/"+c.Want) } // TestONNXPersonalBoundaryStratified scores the model homesrv actually runs. // Production is read from personalBoundary.score; top1, top2, top3 and a // whole-class centroid are diagnostics over the same embedded seeds. Today // production and top3 coincide, but keeping them separate means a later scoring // experiment can be compared without rewriting this evaluation or putting its // candidate math in runtime code. The privacy boundary is a hard contract, so // every production miss is a test failure rather than an accuracy target to // average away. func TestONNXPersonalBoundaryStratified(t *testing.T) { if os.Getenv("MAVEN_EVAL_PERSONAL_BOUNDARY") == "" { t.Skip("set MAVEN_EVAL_PERSONAL_BOUNDARY=1 to run the deliberately strict V-702 matrix") } lib := os.Getenv("MAVEN_ONNX_LIB") if lib == "" { t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing") } modelDir := filepath.Join("../..", "models/embedder/multilingual-e5-small") model := filepath.Join(modelDir, "model_quantized.onnx") tokenizer := filepath.Join(modelDir, "tokenizer.json") for _, path := range []string{lib, model, tokenizer} { if _, err := os.Stat(path); err != nil { t.Skipf("personal boundary eval dependency %s unavailable: %v", path, err) } } embedder, err := router.NewONNXEmbedder(model, tokenizer, lib) if err != nil { t.Skipf("onnx embedder unavailable: %v", err) } defer embedder.Close() ctx := context.Background() boundary := &personalBoundary{} boundary.load(ctx, embedder) if !boundary.loaded { t.Fatal("personal boundary seeds did not load with a working embedder") } // Production loads its model-ID-pinned frozen head and deliberately skips // the 132 corpus embeddings on a user's first query. This test still needs // those vectors for the historical top-k/centroid diagnostics, so build // them here without putting that latency back in runtime code. embedCorpus := func(values []string) [][]float32 { vectors := make([][]float32, len(values)) for i, value := range values { vector, err := router.EmbedQuery(ctx, embedder, value) if err != nil { t.Fatalf("embed diagnostic corpus %q: %v", value, err) } vectors[i] = vector } return vectors } boundary.personal = embedCorpus(personalSeeds) boundary.world = embedCorpus(worldSeeds) personalCentroid := personalBoundaryEvalCentroid(boundary.personal) worldCentroid := personalBoundaryEvalCentroid(boundary.world) if len(personalCentroid) == 0 || len(worldCentroid) == 0 { t.Fatal("personal boundary seed vectors do not share a dimension") } type candidate struct { name string score func([]float32) (float64, float64) } candidates := []candidate{ {name: "production", score: func(vec []float32) (float64, float64) { personal, world, ok := boundary.score(vec) if !ok { t.Fatal("loaded personal boundary declined to score") } return personal, world }}, {name: "top1", score: func(vec []float32) (float64, float64) { return meanNearest(vec, boundary.personal, 1), meanNearest(vec, boundary.world, 1) }}, {name: "top2", score: func(vec []float32) (float64, float64) { return meanNearest(vec, boundary.personal, 2), meanNearest(vec, boundary.world, 2) }}, {name: "top3", score: func(vec []float32) (float64, float64) { return meanNearest(vec, boundary.personal, 3), meanNearest(vec, boundary.world, 3) }}, {name: "centroid", score: func(vec []float32) (float64, float64) { return cosine(vec, personalCentroid), cosine(vec, worldCentroid) }}, } reports := make(map[string]*personalBoundaryEvalReport, len(candidates)) for _, candidate := range candidates { reports[candidate.name] = newPersonalBoundaryEvalReport(candidate.name) } fixture := loadPersonalBoundaryEvalFixture(t) for _, c := range fixture.Cases { vec, err := router.EmbedQuery(ctx, embedder, c.Utterance) if err != nil { t.Fatalf("%s: embed %q: %v", c.ID, c.Utterance, err) } for _, candidate := range candidates { personal, world := candidate.score(vec) gotPersonal := personal > world reports[candidate.name].add(c, gotPersonal, personal, world) if candidate.name == "production" && gotPersonal != (c.Want == "personal") { t.Errorf("%s [%s/%s]: got %s, want %s (personal %.4f world %.4f delta %+.4f): %q", c.ID, c.Lang, c.Stratum, boundaryEvalSide(gotPersonal), c.Want, personal, world, personal-world, c.Utterance) } } } for _, candidate := range candidates { report := reports[candidate.name] t.Logf("candidate %-15s %2d/%d (%.1f%%), minimum signed margin %+.4f", report.Name, report.Correct, report.Total, 100*float64(report.Correct)/float64(report.Total), report.MinimumMargin) } production := reports["production"] for _, lang := range []string{"ru", "en"} { stat := production.ByLanguage[lang] t.Logf("production language %-2s %2d/%d", lang, stat.Correct, stat.Total) } for _, side := range []string{"personal", "world"} { stat := production.ByExpectedClass[side] t.Logf("production expected %-8s %2d/%d", side, stat.Correct, stat.Total) } strata := append([]string(nil), personalBoundaryEvalStrata...) sort.Strings(strata) for _, stratum := range strata { stat := production.ByStratum[stratum] ruPersonal := production.ByCell[stratum+"/ru/personal"] ruWorld := production.ByCell[stratum+"/ru/world"] enPersonal := production.ByCell[stratum+"/en/personal"] enWorld := production.ByCell[stratum+"/en/world"] t.Logf("production stratum %-21s %2d/%d | ru personal %d/%d world %d/%d | en personal %d/%d world %d/%d", stratum, stat.Correct, stat.Total, ruPersonal.Correct, ruPersonal.Total, ruWorld.Correct, ruWorld.Total, enPersonal.Correct, enPersonal.Total, enWorld.Correct, enWorld.Total) } } func personalBoundaryEvalCentroid(vectors [][]float32) []float32 { if len(vectors) == 0 { return nil } centroid := make([]float32, len(vectors[0])) for _, vector := range vectors { if len(vector) != len(centroid) { return nil } for i, value := range vector { centroid[i] += value } } for i := range centroid { centroid[i] /= float32(len(vectors)) } return centroid } func boundaryEvalSide(personal bool) string { if personal { return "personal" } return "world" } // meanNearest is an evaluation baseline retained beside the strict fixture; // production uses the linear head in personalboundary.go. func meanNearest(vec []float32, seeds [][]float32, k int) float64 { if len(seeds) == 0 || k <= 0 { return -1 } if k > len(seeds) { k = len(seeds) } top := make([]float64, k) for i := range top { top[i] = -1 } for _, seed := range seeds { candidate := cosine(vec, seed) for i := range top { if candidate > top[i] { candidate, top[i] = top[i], candidate } } } var sum float64 for _, similarity := range top { sum += similarity } return sum / float64(k) }