Files
Maven/internal/phraser/eval/talk_test.go
T
kami 50ca8c8b5a Score the chat, query and knowledge phrasing paths (#395)
The phrasing fixture was 15 nudge cases, so every prompt change we
measured only told us about nudges. But the shared context block sits in
front of five prompts, and three of them — chat, note query, general
knowledge — had no scorer at all. Those are the long free-form replies,
where a persona break is most likely and where nothing could see one.

27 cases, nine per path. Nine rather than five because the nudge fixture
already cannot resolve a change smaller than about three cases, and a
per-path score off five would be worse.

Reuses the persona checks instead of copying them. Length, mood and
"no questions" are left out on purpose: these paths return no mood, and
a follow-up question is a feature in chat, not a fault.

The run refuses to score unless the model answers before and after it.
PhraseChat and PhraseQuery swallow model errors and return a canned
string, so without that guard a dead server produces a full report with
zero errors and a bad score — which reads as bad phrasing rather than as
nothing measured. Vikunja #397 is the real fix.
2026-07-31 16:51:52 +04:00

164 lines
5.5 KiB
Go

package eval
import (
"context"
"os"
"strings"
"testing"
"time"
"github.com/kami/maven/internal/dialogue"
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/persona"
"github.com/kami/maven/internal/phraser"
)
// perPathMinimum — the resolution floor. A per-path score built on a handful of
// cases moves by 12% when a single reply changes, which cannot distinguish a
// prompt regression from noise.
const perPathMinimum = 8
// TestTalkFixture — the fixture itself has to be sound before any score off it
// means anything.
func TestTalkFixture(t *testing.T) {
f, err := LoadTalk()
if err != nil {
t.Fatalf("LoadTalk: %v", err)
}
seen := map[string]bool{}
byPath := map[string]int{}
for _, c := range f.Cases {
if seen[c.ID] {
t.Errorf("duplicate case id %q", c.ID)
}
seen[c.ID] = true
switch c.Path {
case PathChat, PathQuery, PathKnowledge:
default:
t.Errorf("%s: unknown path %q", c.ID, c.Path)
}
byPath[c.Path]++
if strings.TrimSpace(c.Utterance) == "" {
t.Errorf("%s: empty utterance", c.ID)
}
if len(c.WantAny) == 0 {
t.Errorf("%s: no want_any — the reply cannot be checked for topic", c.ID)
}
// A query case with no notes would silently score the knowledge path.
if c.Path == PathQuery && len(c.Notes) == 0 {
t.Errorf("%s: query case has no notes", c.ID)
}
if c.Path == PathKnowledge && len(c.Notes) > 0 {
t.Errorf("%s: knowledge case must have no notes", c.ID)
}
}
for _, p := range TalkPaths {
if byPath[p] < perPathMinimum {
t.Errorf("path %s has %d cases, want at least %d", p, byPath[p], perPathMinimum)
}
}
}
// fakeTalker — a scripted Talker, so the scorer is testable without a model.
type fakeTalker struct{ reply string }
func (f fakeTalker) PhraseChat(context.Context, string, []dialogue.Turn) (string, error) {
return f.reply, nil
}
func (f fakeTalker) PhraseQuery(context.Context, string, []string) (string, error) {
return f.reply, nil
}
// TestScoreTalkCounts — a reply that fails on purpose must be counted on every
// path, so a real run cannot report a hidden zero.
func TestScoreTalkCounts(t *testing.T) {
f, err := LoadTalk()
if err != nil {
t.Fatalf("LoadTalk: %v", err)
}
// Formal address, off-topic, trailing ellipsis: three checks fail at once.
rep, err := ScoreTalk(context.Background(), "fake", fakeTalker{"Приходите, я вас жду…"}, f)
if err != nil {
t.Fatalf("ScoreTalk: %v", err)
}
if rep.Total != len(f.Cases) || rep.Passed != 0 {
t.Errorf("got %d/%d passing, want 0/%d", rep.Passed, rep.Total, len(f.Cases))
}
if rep.ByCheck[CheckAddress] != 0 {
t.Errorf("formal reply passed the address check %d times", rep.ByCheck[CheckAddress])
}
if rep.ByCheck[CheckEllipsis] != 0 {
t.Errorf("truncated reply passed the ellipsis check %d times", rep.ByCheck[CheckEllipsis])
}
for _, p := range TalkPaths {
if rep.ByPath[p].Total == 0 {
t.Errorf("path %s missing from the report", p)
}
}
if !strings.Contains(rep.String(), "by path") {
t.Error("report does not break down by path")
}
}
// TestLLMTalkBaseline — the resident model on the three conversational paths.
// Opt-in exactly like TestLLMPhrasingBaseline: CI has no model and a run costs
// minutes on the CPU target.
//
// MAVEN_LLM_URL=http://127.0.0.1:18099 \
// go test -run TestLLMTalkBaseline ./internal/phraser/eval/
//
// Reports, does not assert a quality bar — the numbers are the input to tuning
// the persona prompt. The one thing worth failing on is a harness fault.
func TestLLMTalkBaseline(t *testing.T) {
base := os.Getenv("MAVEN_LLM_URL")
if base == "" {
t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server (see doc comment)")
}
noProxyLoopback(t)
ctx := context.Background()
f, err := LoadTalk()
if err != nil {
t.Fatalf("LoadTalk: %v", err)
}
cfg := phraser.DefaultConfig("")
cfg.Timeout = 5 * time.Minute
cfg.ContextBlock = func() string { return persona.Facts{}.Block(time.Now()) }
p := phraser.NewLLMPhraserAt(base, cfg)
defer p.Close()
// Unreachable server is fatal here, not a logged warning, and that differs
// from the nudge test on purpose. PhraseNudge returns its errors, so a dead
// server there shows up honestly in the Errors column. PhraseChat and
// PhraseQuery do NOT: they swallow every failure and return a canned string
// ("поговорили.", "не знаю.", "вот что я нашла: …"). So on these three paths
// a dead server produces a full report with 0 errors and a terrible score —
// a number that looks like bad phrasing and is really no phrasing at all.
// Refusing to score without a confirmed model is the only guard available
// until the phraser reports its failures (Vikunja #397).
model, err := llm.ModelID(ctx, base)
if err != nil {
t.Fatalf("no model at %s: %v — refusing to score, these paths hide their errors "+
"and would report a plausible-looking result off a dead server", base, err)
}
t.Logf("scoring model %s at %s", model, base)
rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", p, f)
if err != nil {
t.Fatalf("ScoreTalk: %v", err)
}
t.Log("\n" + rep.String() + "\nreplies:\n" + rep.Replies() + "\nfailures:\n" + rep.Failures())
// And again afterwards: the run takes minutes, and a server that died or got
// OOM-killed halfway through would leave the first cases scored and the rest
// silently canned. Checking only at the start would not catch that.
if _, err := llm.ModelID(ctx, base); err != nil {
t.Fatalf("model at %s went away during the run: %v — the score above is not trustworthy", base, err)
}
}