package eval import ( "context" "os" "strings" "testing" "time" "github.com/kami/maven/internal/dialogue" "github.com/kami/maven/internal/llm" "github.com/kami/maven/internal/persona" "github.com/kami/maven/internal/phraser" ) // perPathMinimum — the resolution floor. A per-path score built on a handful of // cases moves by 12% when a single reply changes, which cannot distinguish a // prompt regression from noise. const perPathMinimum = 8 // TestTalkFixture — the fixture itself has to be sound before any score off it // means anything. func TestTalkFixture(t *testing.T) { f, err := LoadTalk() if err != nil { t.Fatalf("LoadTalk: %v", err) } seen := map[string]bool{} byPath := map[string]int{} for _, c := range f.Cases { if seen[c.ID] { t.Errorf("duplicate case id %q", c.ID) } seen[c.ID] = true switch c.Path { case PathChat, PathQuery, PathKnowledge: default: t.Errorf("%s: unknown path %q", c.ID, c.Path) } byPath[c.Path]++ if strings.TrimSpace(c.Utterance) == "" { t.Errorf("%s: empty utterance", c.ID) } if len(c.WantAny) == 0 { t.Errorf("%s: no want_any — the reply cannot be checked for topic", c.ID) } // A query case with no notes would silently score the knowledge path. if c.Path == PathQuery && len(c.Notes) == 0 { t.Errorf("%s: query case has no notes", c.ID) } if c.Path == PathKnowledge && len(c.Notes) > 0 { t.Errorf("%s: knowledge case must have no notes", c.ID) } } for _, p := range TalkPaths { if byPath[p] < perPathMinimum { t.Errorf("path %s has %d cases, want at least %d", p, byPath[p], perPathMinimum) } } } // fakeTalker — a scripted Talker, so the scorer is testable without a model. type fakeTalker struct{ reply string } func (f fakeTalker) PhraseChat(context.Context, string, []dialogue.Turn) (string, error) { return f.reply, nil } func (f fakeTalker) PhraseQuery(context.Context, string, []string) (string, error) { return f.reply, nil } // TestScoreTalkCounts — a reply that fails on purpose must be counted on every // path, so a real run cannot report a hidden zero. func TestScoreTalkCounts(t *testing.T) { f, err := LoadTalk() if err != nil { t.Fatalf("LoadTalk: %v", err) } // Formal address, off-topic, trailing ellipsis: three checks fail at once. rep, err := ScoreTalk(context.Background(), "fake", fakeTalker{"Приходите, я вас жду…"}, f) if err != nil { t.Fatalf("ScoreTalk: %v", err) } if rep.Total != len(f.Cases) || rep.Passed != 0 { t.Errorf("got %d/%d passing, want 0/%d", rep.Passed, rep.Total, len(f.Cases)) } if rep.ByCheck[CheckAddress] != 0 { t.Errorf("formal reply passed the address check %d times", rep.ByCheck[CheckAddress]) } if rep.ByCheck[CheckEllipsis] != 0 { t.Errorf("truncated reply passed the ellipsis check %d times", rep.ByCheck[CheckEllipsis]) } for _, p := range TalkPaths { if rep.ByPath[p].Total == 0 { t.Errorf("path %s missing from the report", p) } } if !strings.Contains(rep.String(), "by path") { t.Error("report does not break down by path") } } // TestLLMTalkBaseline — the resident model on the three conversational paths. // Opt-in exactly like TestLLMPhrasingBaseline: CI has no model and a run costs // minutes on the CPU target. // // MAVEN_LLM_URL=http://127.0.0.1:18099 \ // go test -run TestLLMTalkBaseline ./internal/phraser/eval/ // // Reports, does not assert a quality bar — the numbers are the input to tuning // the persona prompt. The one thing worth failing on is a harness fault. func TestLLMTalkBaseline(t *testing.T) { base := os.Getenv("MAVEN_LLM_URL") if base == "" { t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server (see doc comment)") } noProxyLoopback(t) ctx := context.Background() f, err := LoadTalk() if err != nil { t.Fatalf("LoadTalk: %v", err) } cfg := phraser.DefaultConfig("") cfg.Timeout = 5 * time.Minute cfg.ContextBlock = func() string { return persona.Facts{}.Block(time.Now()) } p := phraser.NewLLMPhraserAt(base, cfg) defer p.Close() // Unreachable server is fatal here, not a logged warning, and that differs // from the nudge test on purpose. PhraseNudge returns its errors, so a dead // server there shows up honestly in the Errors column. PhraseChat and // PhraseQuery do NOT: they swallow every failure and return a canned string // ("поговорили.", "не знаю.", "вот что я нашла: …"). So on these three paths // a dead server produces a full report with 0 errors and a terrible score — // a number that looks like bad phrasing and is really no phrasing at all. // Refusing to score without a confirmed model is the only guard available // until the phraser reports its failures (Vikunja #397). model, err := llm.ModelID(ctx, base) if err != nil { t.Fatalf("no model at %s: %v — refusing to score, these paths hide their errors "+ "and would report a plausible-looking result off a dead server", base, err) } t.Logf("scoring model %s at %s", model, base) rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", p, f) if err != nil { t.Fatalf("ScoreTalk: %v", err) } t.Log("\n" + rep.String() + "\nreplies:\n" + rep.Replies() + "\nfailures:\n" + rep.Failures()) // And again afterwards: the run takes minutes, and a server that died or got // OOM-killed halfway through would leave the first cases scored and the rest // silently canned. Checking only at the start would not catch that. if _, err := llm.ModelID(ctx, base); err != nil { t.Fatalf("model at %s went away during the run: %v — the score above is not trustworthy", base, err) } }