Gate recall on the margin over the runner-up, not just the score

The e5 embedder puts every cosine in one narrow band (0.79-0.89), so the
absolute query_min_score gate cannot tell a real hit from a made-up
question: any value under the band answers everything, any value above it
answers nothing. False recall was 5/5.

New gate asks whether one note is clearly the best instead: top1 - top2 >
delta. New query_min_margin config knob, default 0.008, read off the sweep
in the recall harness. The absolute floor stays as a second check.

On the recall fixture with e5: answered 72% -> 68%, false recall 5/5 -> 1/5.

Vikunja #359

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
This commit is contained in:
kami
2026-07-31 11:50:39 +04:00
parent 34521c30b8
commit 11831c6ace
9 changed files with 237 additions and 50 deletions
+5 -8
View File
@@ -8,16 +8,13 @@ import "github.com/kami/maven/internal/memory"
// that can answer "when did I last …?" from a captured fact). A note hit here
// is redundant with the notes-RAG path — by design; the two indexes can diverge
// once the backend is swapped for a persistent/external store. ok=false when
// there's no hit above the threshold or the hit carries no text.
func bestRecall(results []memory.Result, min float64) (string, bool) {
if len(results) == 0 {
// the hit fails the confidence gate (see memory.Confident: an absolute floor
// plus a margin over the runner-up) or carries no text.
func bestRecall(results []memory.Result, minScore, minMargin float64) (string, bool) {
if !memory.Confident(results, minScore, minMargin) {
return "", false
}
top := results[0]
if top.Score < min {
return "", false
}
text := top.Meta["text"]
text := results[0].Meta["text"]
if text == "" {
return "", false
}
+17 -4
View File
@@ -8,23 +8,24 @@ import (
func TestBestRecall(t *testing.T) {
const min = 0.55
const margin = 0.008
t.Run("empty results", func(t *testing.T) {
if _, ok := bestRecall(nil, min); ok {
if _, ok := bestRecall(nil, min, margin); ok {
t.Error("empty results returned ok")
}
})
t.Run("top below threshold", func(t *testing.T) {
res := []memory.Result{{Score: 0.4, Meta: map[string]string{"text": "выпил воды"}}}
if _, ok := bestRecall(res, min); ok {
if _, ok := bestRecall(res, min, margin); ok {
t.Error("below-threshold hit returned ok")
}
})
t.Run("hit without text meta", func(t *testing.T) {
res := []memory.Result{{Score: 0.9, Meta: map[string]string{"type": "fact"}}}
if _, ok := bestRecall(res, min); ok {
if _, ok := bestRecall(res, min, margin); ok {
t.Error("textless hit returned ok")
}
})
@@ -34,7 +35,7 @@ func TestBestRecall(t *testing.T) {
{Score: 0.82, Meta: map[string]string{"text": "выпил воды в три часа", "type": "fact"}},
{Score: 0.60, Meta: map[string]string{"text": "другое"}},
}
got, ok := bestRecall(res, min)
got, ok := bestRecall(res, min, margin)
if !ok {
t.Fatal("clearing hit not returned")
}
@@ -42,4 +43,16 @@ func TestBestRecall(t *testing.T) {
t.Errorf("wrong text: %q", got)
}
})
// The runner-up is almost as close, so the embedder cannot tell the two
// notes apart. Silence beats reading back a coin flip.
t.Run("runner-up too close", func(t *testing.T) {
res := []memory.Result{
{Score: 0.860, Meta: map[string]string{"text": "выпил воды в три часа"}},
{Score: 0.858, Meta: map[string]string{"text": "другое"}},
}
if _, ok := bestRecall(res, min, margin); ok {
t.Error("thin-margin hit returned ok")
}
})
}
+16 -7
View File
@@ -257,6 +257,7 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
clarifyStore: clarifyStore,
extractor: router.Extractor{Time: timeParser, Acts: matcher, Facts: router.DefaultFactParser{}},
queryMinScore: cfg.Voice.QueryMinScore,
queryMinMargin: cfg.Voice.QueryMinMargin,
timeParser: timeParser,
ecosystem: eco,
}
@@ -300,6 +301,9 @@ type reactiveHandler struct {
// load-bearing math (same posture as the presence thresholds). Set by
// wireVoice from VoiceConfig; default 0.55.
queryMinScore float64
// queryMinMargin — the second half of that gate: how far the top hit must
// beat the runner-up. 0 ⇒ margin off.
queryMinMargin float64
// timeParser — used as a fallback for stage-0 reminder grammar matches
// (where the extractor didn't run). Shared with the router's extractor.
@@ -760,18 +764,23 @@ func (h *reactiveHandler) applyAction(ctx context.Context, dec router.Decision)
log.Printf("voice: query notes: %v", err)
return "не получилось найти ответ."
}
// Confidence gate: below threshold, say "I don't know" rather than read
// back the least-unrelated note — a confident wrong recall is worse than
// a gap (spec's "not a guesser-of-truth"). Same instinct as the loop's
// since(key)==null → don't fire. Tuned for the ONNX embedder; the Hash
// floor scores lexically and may rarely clear it.
if len(notes) == 0 || notes[0].Score < h.queryMinScore {
// Confidence gate: below it, say "I don't know" rather than read back
// the least-unrelated note — a confident wrong recall is worse than a
// gap (spec's "not a guesser-of-truth"). Same instinct as the loop's
// since(key)==null → don't fire. Two parts: an absolute cosine floor,
// and a margin over the runner-up, which is the part that works with
// the e5 embedder's narrow score band. See memory.Confident.
noteScores := make([]float64, len(notes))
for i, n := range notes {
noteScores[i] = n.Score
}
if !memory.ConfidentScores(noteScores, h.queryMinScore, h.queryMinMargin) {
// Long-term memory recall (notes + facts) before general knowledge:
// the notes table can't answer fact questions, but the memory store
// indexes both. Only runs when notes-RAG already gave up → additive.
if h.memStore != nil {
if hits, herr := h.memStore.Search(ctx, vec, 3); herr == nil {
if text, ok := bestRecall(hits, h.queryMinScore); ok {
if text, ok := bestRecall(hits, h.queryMinScore, h.queryMinMargin); ok {
return text
}
}
+2
View File
@@ -41,6 +41,8 @@
"lib_path": "/opt/maven/lib/libonnxruntime.so"
},
"llm_router": false,
"query_min_score": 0.55,
"query_min_margin": 0.008,
"tool_timeout": "30s",
"tools": [
{ "name": "status", "cmd": ["systemctl", "status"], "scope": "homelab", "destructive": false },
+22 -1
View File
@@ -277,6 +277,14 @@ type VoiceConfig struct {
// default if unset.
QueryMinScore float64 `json:"query_min_score,omitempty"`
// QueryMinMargin — the second half of the recall gate: the top hit must
// beat the runner-up by more than this. The absolute score above cannot do
// the job on its own, because the e5 embedder puts every cosine in one
// narrow high band, so a made-up question scores as high as a real one.
// The margin asks whether one note is clearly the best instead.
// Negative ⇒ off. 0 ⇒ the default below.
QueryMinMargin float64 `json:"query_min_margin,omitempty"`
// Persona — optional prompt prefix that tunes maven's character. Prepended
// to every LLM system prompt (nudge phrasing, note queries, general
// knowledge). Empty string ⇒ current hardcoded persona (feminine-gendered
@@ -404,7 +412,12 @@ const (
DefaultAutotuneInterval = 10 * time.Minute
DefaultRouterThreshold = 0.55
DefaultQueryMinScore = 0.55
DefaultToolTimeout = 30 * time.Second
// Read off the margin sweep in internal/memory/recalleval on the e5
// embedder: 0.008 answers 68% of real questions (down from 72%) and cuts
// false recall from 5/5 to 1/5. Every larger delta costs real recall
// without removing that last one until 0.020, which drops recall to 44%.
DefaultQueryMinMargin = 0.008
DefaultToolTimeout = 30 * time.Second
DefaultFactEnrichmentInterval = 30 * time.Second
)
@@ -486,6 +499,14 @@ func (c *Config) applyDefaults() {
if c.Voice.QueryMinScore <= 0 {
c.Voice.QueryMinScore = DefaultQueryMinScore
}
// Unset ⇒ default. Negative is how you turn the margin off on purpose,
// so it is clamped to 0 rather than replaced by the default.
switch {
case c.Voice.QueryMinMargin == 0:
c.Voice.QueryMinMargin = DefaultQueryMinMargin
case c.Voice.QueryMinMargin < 0:
c.Voice.QueryMinMargin = 0
}
if c.Voice.ToolTimeout <= 0 {
c.Voice.ToolTimeout = Duration(DefaultToolTimeout)
}
+38
View File
@@ -0,0 +1,38 @@
package memory
// Confidence gate for a recall. Two checks, both must pass before Maven says a
// note back:
//
// - minScore — an absolute cosine floor.
// - minMargin — the top hit must beat the runner-up by more than this.
//
// The margin is the one that carries the weight. The e5 embedder packs every
// score into a narrow high band (0.79-0.89 on the recall fixture), so an
// absolute floor cannot tell a real hit from a confident-looking miss: every
// value under the band admits everything, every value above it answers nothing.
// A margin asks a different question — "is this note clearly the best one, or
// is the whole shelf equally close?" — and a made-up question has no clear best.
//
// With one hit and no runner-up there is nothing to compare, so only the floor
// applies.
// ConfidentScores reports whether the top score clears both gates. scores must
// be sorted highest first. minMargin <= 0 turns the margin check off.
func ConfidentScores(scores []float64, minScore, minMargin float64) bool {
if len(scores) == 0 || scores[0] < minScore {
return false
}
if minMargin > 0 && len(scores) > 1 && scores[0]-scores[1] <= minMargin {
return false
}
return true
}
// Confident is ConfidentScores for search results.
func Confident(results []Result, minScore, minMargin float64) bool {
scores := make([]float64, len(results))
for i, r := range results {
scores[i] = r.Score
}
return ConfidentScores(scores, minScore, minMargin)
}
+45
View File
@@ -0,0 +1,45 @@
package memory
import "testing"
func TestConfidentScores(t *testing.T) {
cases := []struct {
name string
scores []float64
minScore float64
minMargin float64
want bool
}{
{"no hits", nil, 0.55, 0.008, false},
{"below the floor", []float64{0.40, 0.10}, 0.55, 0.008, false},
{"clear winner", []float64{0.86, 0.70}, 0.55, 0.008, true},
{"runner-up too close", []float64{0.860, 0.858}, 0.55, 0.008, false},
// The rule is "beats the runner-up by MORE than delta". Not testing an
// exactly-equal margin: no pair of these decimals subtracts to exactly
// 0.008 in binary float, so such a test would pin rounding, not the rule.
{"margin just under delta", []float64{0.8079, 0.8}, 0.55, 0.008, false},
{"margin just over delta", []float64{0.8081, 0.8}, 0.55, 0.008, true},
// One hit: nothing to compare against, so only the floor applies.
{"single hit clears", []float64{0.86}, 0.55, 0.008, true},
{"single hit below floor", []float64{0.10}, 0.55, 0.008, false},
// Margin off — the old absolute-only behaviour.
{"margin off admits a tie", []float64{0.86, 0.86}, 0.55, 0, true},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if got := ConfidentScores(c.scores, c.minScore, c.minMargin); got != c.want {
t.Errorf("got %v, want %v", got, c.want)
}
})
}
}
func TestConfidentReadsResultScores(t *testing.T) {
res := []Result{{ID: "a", Score: 0.86}, {ID: "b", Score: 0.858}}
if Confident(res, 0.55, 0.008) {
t.Error("thin margin passed the gate")
}
if !Confident(res, 0.55, 0) {
t.Error("margin off should fall back to the floor alone")
}
}
+42 -19
View File
@@ -177,6 +177,8 @@ type Outcome struct {
Tied bool
TopID string
TopScor float64
// Margin — top1 top2. 0 when fewer than two hits came back.
Margin float64
Reasons []string
}
@@ -184,8 +186,10 @@ type Outcome struct {
// ranks first but is silenced by query_min_score is a threshold problem, and a
// note that never ranks first is an embedder problem. Those are different fixes.
type Report struct {
Name string
MinScore float64
Name string
MinScore float64
// MinMargin — how far the top hit must beat the runner-up. 0 ⇒ off.
MinMargin float64
Total int
Answerable int
Rank1 int
@@ -212,9 +216,14 @@ type Report struct {
// where the right note ranked first, and for the no-answer cases. The gap
// between these two distributions is what a defensible query_min_score
// would have to sit inside; if they overlap, no threshold separates them.
CorrectTop []float64
NoAnswerTop []float64
P50, P95, Max time.Duration
CorrectTop []float64
NoAnswerTop []float64
// CorrectMargin / NoAnswerMargin — the same two groups, but top1 top2
// instead of top1. This is the pair the margin gate has to separate, and
// unlike the absolute scores it is what the sweep reads.
CorrectMargin []float64
NoAnswerMargin []float64
P50, P95, Max time.Duration
}
// TagStat — passed/total for one slice of the fixture.
@@ -245,13 +254,14 @@ func ratio(n, d int) float64 {
// the run on an embed or search error: an erroring case scores as a miss and is
// counted in Errors, because "the embedder was down" and "the embedder was
// wrong" are different numbers.
func Score(ctx context.Context, name string, emb router.Embedder, newStore NewStore, minScore float64, f Fixture) (Report, error) {
func Score(ctx context.Context, name string, emb router.Embedder, newStore NewStore, minScore, minMargin float64, f Fixture) (Report, error) {
rep := Report{
Name: name,
MinScore: minScore,
Total: len(f.Cases),
ByTag: map[string]TagStat{},
ByLang: map[string]TagStat{},
Name: name,
MinScore: minScore,
MinMargin: minMargin,
Total: len(f.Cases),
ByTag: map[string]TagStat{},
ByLang: map[string]TagStat{},
}
lat := make([]time.Duration, 0, len(f.Cases))
@@ -261,7 +271,7 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
} else {
rep.NoAnswer++
}
o, err := scoreCase(ctx, emb, newStore, minScore, c, f.Filler)
o, err := scoreCase(ctx, emb, newStore, minScore, minMargin, c, f.Filler)
if err != nil {
return Report{}, err
}
@@ -285,6 +295,11 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
} else if !o.Rank1 {
rep.WrongTop++
}
if o.Rank1 {
// Margins are collected on rank, not on the gate, so the
// distribution does not move as the sweep changes the gate.
rep.CorrectMargin = append(rep.CorrectMargin, o.Margin)
}
if o.Rank1 && o.Recalled != "" {
rep.CorrectTop = append(rep.CorrectTop, o.TopScor)
}
@@ -293,6 +308,7 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
rep.FalseRecall++
}
rep.NoAnswerTop = append(rep.NoAnswerTop, o.TopScor)
rep.NoAnswerMargin = append(rep.NoAnswerMargin, o.Margin)
}
if o.Pass {
@@ -307,6 +323,8 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
sort.Float64s(rep.CorrectTop)
sort.Float64s(rep.NoAnswerTop)
sort.Float64s(rep.CorrectMargin)
sort.Float64s(rep.NoAnswerMargin)
sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] })
rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95)
if len(lat) > 0 {
@@ -318,7 +336,7 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
// scoreCase inserts the case's notes into a fresh store, then runs the read
// path the daemon runs. The returned error is fatal (the harness is broken);
// an embedder or store failure on the query lands in Outcome.Err instead.
func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minScore float64, c Case, filler []StoredNote) (Outcome, error) {
func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minScore, minMargin float64, c Case, filler []StoredNote) (Outcome, error) {
st, release, err := newStore()
if err != nil {
return Outcome{}, fmt.Errorf("%s: new store: %w", c.ID, err)
@@ -357,7 +375,10 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
if len(hits) > 0 {
o.TopID, o.TopScor = hits[0].ID, hits[0].Score
o.Recalled = bestRecall(hits, minScore)
if len(hits) > 1 {
o.Margin = hits[0].Score - hits[1].Score
}
o.Recalled = bestRecall(hits, minScore, minMargin)
}
for i, h := range hits {
if h.ID != c.Want {
@@ -378,14 +399,14 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
switch {
case !c.Answerable():
if o.Recalled != "" {
o.Reasons = append(o.Reasons, fmt.Sprintf("false recall: %q at %.3f, want silence", o.TopID, o.TopScor))
o.Reasons = append(o.Reasons, fmt.Sprintf("false recall: %q at %.3f (margin %.3f), want silence", o.TopID, o.TopScor, o.Margin))
}
case o.Tied:
o.Reasons = append(o.Reasons, fmt.Sprintf("tie at %.3f — the right note is on top only by sort order", o.TopScor))
case !o.Rank1:
o.Reasons = append(o.Reasons, fmt.Sprintf("top hit %q (%.3f), want %q%s", o.TopID, o.TopScor, c.Want, rankNote(o.Rank3)))
case o.Recalled == "":
o.Reasons = append(o.Reasons, fmt.Sprintf("right note ranked first but scored %.3f < gate %.2f — daemon says \"не знаю\"", o.TopScor, minScore))
o.Reasons = append(o.Reasons, fmt.Sprintf("right note ranked first at %.3f (margin %.3f) but the gate silenced it — daemon says \"не знаю\"", o.TopScor, o.Margin))
}
o.Pass = len(o.Reasons) == 0
return o, nil
@@ -401,8 +422,8 @@ func rankNote(inTop3 bool) string {
// bestRecall mirrors cmd/mavend/recall.go — the gate the daemon actually
// applies to a memory hit. Duplicated rather than imported because package main
// is not importable; recalleval_test.go asserts the two agree in behaviour.
func bestRecall(results []memory.Result, min float64) string {
if len(results) == 0 || results[0].Score < min {
func bestRecall(results []memory.Result, minScore, minMargin float64) string {
if !memory.Confident(results, minScore, minMargin) {
return ""
}
return results[0].Meta["text"]
@@ -437,7 +458,7 @@ func percentile(sorted []time.Duration, p float64) time.Duration {
// the slices that name where the path is weak.
func (r Report) String() string {
var b strings.Builder
fmt.Fprintf(&b, "%s: %d/%d cases pass (gate %.2f)\n", r.Name, r.Passed, r.Total, r.MinScore)
fmt.Fprintf(&b, "%s: %d/%d cases pass (gate %.2f, margin %.3f)\n", r.Name, r.Passed, r.Total, r.MinScore, r.MinMargin)
fmt.Fprintf(&b, " recall@1 %.1f%% (%d/%d) recall@3 %.1f%% (%d/%d) answered after gate %.1f%% (%d/%d)\n",
100*r.Recall1(), r.Rank1, r.Answerable,
100*r.Recall3(), r.Rank3, r.Answerable,
@@ -448,6 +469,8 @@ func (r Report) String() string {
100*r.FalseRecallRate(), r.FalseRecall, r.NoAnswer)
fmt.Fprintf(&b, " top-1 score, right note first: %s\n", spread(r.CorrectTop))
fmt.Fprintf(&b, " top-1 score, must be silent: %s\n", spread(r.NoAnswerTop))
fmt.Fprintf(&b, " margin top1-top2, right note first: %s\n", spread(r.CorrectMargin))
fmt.Fprintf(&b, " margin top1-top2, must be silent: %s\n", spread(r.NoAnswerMargin))
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang))
fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag))
+50 -11
View File
@@ -142,21 +142,40 @@ func words(s string) []string {
// cmd/mavend/recall.go (package main is not importable). This pins the copy to
// the original's three rules: no hits, below the gate, or no text ⇒ silence.
func TestBestRecallMatchesDaemon(t *testing.T) {
if got := bestRecall(nil, 0.55); got != "" {
if got := bestRecall(nil, 0.55, 0); got != "" {
t.Errorf("no hits: got %q, want silence", got)
}
low := []memory.Result{{ID: "a", Score: 0.4, Meta: map[string]string{"text": "чай"}}}
if got := bestRecall(low, 0.55); got != "" {
if got := bestRecall(low, 0.55, 0); got != "" {
t.Errorf("below gate: got %q, want silence", got)
}
noText := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{}}}
if got := bestRecall(noText, 0.55); got != "" {
if got := bestRecall(noText, 0.55, 0); got != "" {
t.Errorf("no text: got %q, want silence", got)
}
ok := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{"text": "чай"}}}
if got := bestRecall(ok, 0.55); got != "чай" {
if got := bestRecall(ok, 0.55, 0); got != "чай" {
t.Errorf("above gate: got %q, want %q", got, "чай")
}
// Margin: a close runner-up means the embedder cannot tell the two apart,
// so Maven stays silent even though both clear the absolute floor.
close := []memory.Result{
{ID: "a", Score: 0.86, Meta: map[string]string{"text": "чай"}},
{ID: "b", Score: 0.85, Meta: map[string]string{"text": "кофе"}},
}
if got := bestRecall(close, 0.55, 0.03); got != "" {
t.Errorf("thin margin: got %q, want silence", got)
}
if got := bestRecall(close, 0.55, 0); got != "чай" {
t.Errorf("margin off: got %q, want %q", got, "чай")
}
clear := []memory.Result{
{ID: "a", Score: 0.86, Meta: map[string]string{"text": "чай"}},
{ID: "b", Score: 0.70, Meta: map[string]string{"text": "кофе"}},
}
if got := bestRecall(clear, 0.55, 0.03); got != "чай" {
t.Errorf("wide margin: got %q, want %q", got, "чай")
}
}
// TestHashRecallBaseline — the CI ratchet. HashEmbedder, so it needs no model
@@ -171,7 +190,7 @@ func TestHashRecallBaseline(t *testing.T) {
t.Fatalf("Load: %v", err)
}
rep, err := Score(context.Background(), "recall+hash", router.NewHashEmbedder(hashDim), InMemory,
config.DefaultQueryMinScore, f)
config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
if err != nil {
t.Fatalf("Score: %v", err)
}
@@ -201,11 +220,11 @@ func TestPersistentStoreScoresTheSame(t *testing.T) {
t.Fatalf("Load: %v", err)
}
emb := router.NewHashEmbedder(hashDim)
inMem, err := Score(context.Background(), "recall+hash+memory", emb, InMemory, config.DefaultQueryMinScore, f)
inMem, err := Score(context.Background(), "recall+hash+memory", emb, InMemory, config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
if err != nil {
t.Fatalf("Score in-memory: %v", err)
}
persistent, err := Score(context.Background(), "recall+hash+sqlite", emb, sqliteStores(t), config.DefaultQueryMinScore, f)
persistent, err := Score(context.Background(), "recall+hash+sqlite", emb, sqliteStores(t), config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
if err != nil {
t.Fatalf("Score sqlite: %v", err)
}
@@ -263,14 +282,16 @@ func TestONNXRecall(t *testing.T) {
if err != nil {
t.Fatalf("Load: %v", err)
}
rep, err := Score(context.Background(), "recall+onnx", emb, InMemory, config.DefaultQueryMinScore, f)
rep, err := Score(context.Background(), "recall+onnx", emb, InMemory, config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
if err != nil {
t.Fatalf("Score: %v", err)
}
t.Log("\n" + rep.String() + rep.Failures())
// Cached for the sweep only: the headline run above must pay the real
// Cached for the sweeps only: the headline run above must pay the real
// embedder cost so its latency numbers mean something.
t.Log("\ngate sweep:\n" + sweep(t, Cache(emb), f))
cached := Cache(emb)
t.Log("\ngate sweep (margin off):\n" + sweep(t, cached, f))
t.Log("\nmargin sweep (gate 0.55):\n" + marginSweep(t, cached, f))
}
// sweep scores the fixture at a range of gates and renders one line each. Two
@@ -281,7 +302,7 @@ func sweep(t *testing.T, emb router.Embedder, f Fixture) string {
t.Helper()
var b strings.Builder
for _, gate := range []float64{0.0, 0.30, 0.40, 0.50, 0.55, 0.60, 0.70, 0.80, 0.90} {
rep, err := Score(context.Background(), "sweep", emb, InMemory, gate, f)
rep, err := Score(context.Background(), "sweep", emb, InMemory, gate, 0, f)
if err != nil {
t.Fatalf("sweep at %.2f: %v", gate, err)
}
@@ -290,3 +311,21 @@ func sweep(t *testing.T, emb router.Embedder, f Fixture) string {
}
return b.String()
}
// marginSweep is the same idea for the margin gate (top1 top2 > delta), with
// the absolute gate held at its default. The absolute score cannot separate a
// real hit from a made-up question under e5 — every score lands in one narrow
// band — so this sweep is the one that picks a number.
func marginSweep(t *testing.T, emb router.Embedder, f Fixture) string {
t.Helper()
var b strings.Builder
for _, d := range []float64{0, 0.002, 0.005, 0.008, 0.01, 0.012, 0.015, 0.02, 0.025, 0.03, 0.04, 0.05, 0.06} {
rep, err := Score(context.Background(), "margin sweep", emb, InMemory, config.DefaultQueryMinScore, d, f)
if err != nil {
t.Fatalf("margin sweep at %.3f: %v", d, err)
}
fmt.Fprintf(&b, " delta %.3f: answered %d/%d (%.0f%%) false recall %d/%d\n",
d, rep.Rank1-rep.Gated, rep.Answerable, 100*rep.Answered(), rep.FalseRecall, rep.NoAnswer)
}
return b.String()
}