35018226ef
Nine reply cases and a fourth column in the talk report. The reply path is a separate object from the phraser in the daemon, so Pair joins a Talker and a Confirmer for a run that covers everything Maven says. Cases carry intent/key/value because the replier is phrased from the decision the router resolved, not from the raw utterance. Three of them are baits the other paths cannot produce: a masculine verb about himself that she must not copy onto herself, a polite plural input that must still come back на ты, and an unresolved note that invites a question a confirmation is not allowed to ask. Not scored against a model here — this box has no llama-server, and the baseline test is opt-in on MAVEN_LLM_URL.
179 lines
6.1 KiB
Go
179 lines
6.1 KiB
Go
package eval
|
|
|
|
import (
|
|
"context"
|
|
"os"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/kami/maven/internal/dialogue"
|
|
"github.com/kami/maven/internal/llm"
|
|
"github.com/kami/maven/internal/persona"
|
|
"github.com/kami/maven/internal/phraser"
|
|
"github.com/kami/maven/internal/router"
|
|
)
|
|
|
|
// perPathMinimum — the resolution floor. A per-path score built on a handful of
|
|
// cases moves by 12% when a single reply changes, which cannot distinguish a
|
|
// prompt regression from noise.
|
|
const perPathMinimum = 8
|
|
|
|
// TestTalkFixture — the fixture itself has to be sound before any score off it
|
|
// means anything.
|
|
func TestTalkFixture(t *testing.T) {
|
|
f, err := LoadTalk()
|
|
if err != nil {
|
|
t.Fatalf("LoadTalk: %v", err)
|
|
}
|
|
|
|
seen := map[string]bool{}
|
|
byPath := map[string]int{}
|
|
for _, c := range f.Cases {
|
|
if seen[c.ID] {
|
|
t.Errorf("duplicate case id %q", c.ID)
|
|
}
|
|
seen[c.ID] = true
|
|
|
|
switch c.Path {
|
|
case PathChat, PathQuery, PathKnowledge:
|
|
case PathReply:
|
|
if c.Intent == "" {
|
|
t.Errorf("%s: reply case has no intent — the replier is phrased from the decision", c.ID)
|
|
}
|
|
default:
|
|
t.Errorf("%s: unknown path %q", c.ID, c.Path)
|
|
}
|
|
byPath[c.Path]++
|
|
|
|
if strings.TrimSpace(c.Utterance) == "" {
|
|
t.Errorf("%s: empty utterance", c.ID)
|
|
}
|
|
if len(c.WantAny) == 0 {
|
|
t.Errorf("%s: no want_any — the reply cannot be checked for topic", c.ID)
|
|
}
|
|
// A query case with no notes would silently score the knowledge path.
|
|
if c.Path == PathQuery && len(c.Notes) == 0 {
|
|
t.Errorf("%s: query case has no notes", c.ID)
|
|
}
|
|
if c.Path == PathKnowledge && len(c.Notes) > 0 {
|
|
t.Errorf("%s: knowledge case must have no notes", c.ID)
|
|
}
|
|
}
|
|
|
|
for _, p := range TalkPaths {
|
|
if byPath[p] < perPathMinimum {
|
|
t.Errorf("path %s has %d cases, want at least %d", p, byPath[p], perPathMinimum)
|
|
}
|
|
}
|
|
}
|
|
|
|
// fakeTalker — a scripted Talker, so the scorer is testable without a model.
|
|
type fakeTalker struct{ reply string }
|
|
|
|
func (f fakeTalker) PhraseChat(context.Context, string, []dialogue.Turn) (string, error) {
|
|
return f.reply, nil
|
|
}
|
|
|
|
func (f fakeTalker) PhraseQuery(context.Context, string, []string) (string, error) {
|
|
return f.reply, nil
|
|
}
|
|
|
|
func (f fakeTalker) PhraseReply(context.Context, router.Decision) (string, error) {
|
|
return f.reply, nil
|
|
}
|
|
|
|
// TestScoreTalkCounts — a reply that fails on purpose must be counted on every
|
|
// path, so a real run cannot report a hidden zero.
|
|
func TestScoreTalkCounts(t *testing.T) {
|
|
f, err := LoadTalk()
|
|
if err != nil {
|
|
t.Fatalf("LoadTalk: %v", err)
|
|
}
|
|
// Formal address, off-topic, trailing ellipsis: three checks fail at once.
|
|
rep, err := ScoreTalk(context.Background(), "fake", fakeTalker{"Приходите, я вас жду…"}, f)
|
|
if err != nil {
|
|
t.Fatalf("ScoreTalk: %v", err)
|
|
}
|
|
if rep.Total != len(f.Cases) || rep.Passed != 0 {
|
|
t.Errorf("got %d/%d passing, want 0/%d", rep.Passed, rep.Total, len(f.Cases))
|
|
}
|
|
if rep.ByCheck[CheckAddress] != 0 {
|
|
t.Errorf("formal reply passed the address check %d times", rep.ByCheck[CheckAddress])
|
|
}
|
|
if rep.ByCheck[CheckEllipsis] != 0 {
|
|
t.Errorf("truncated reply passed the ellipsis check %d times", rep.ByCheck[CheckEllipsis])
|
|
}
|
|
for _, p := range TalkPaths {
|
|
if rep.ByPath[p].Total == 0 {
|
|
t.Errorf("path %s missing from the report", p)
|
|
}
|
|
}
|
|
if !strings.Contains(rep.String(), "by path") {
|
|
t.Error("report does not break down by path")
|
|
}
|
|
}
|
|
|
|
// TestLLMTalkBaseline — the resident model on all four phrasing paths.
|
|
// Opt-in exactly like TestLLMPhrasingBaseline: CI has no model and a run costs
|
|
// minutes on the CPU target.
|
|
//
|
|
// MAVEN_LLM_URL=http://127.0.0.1:18099 \
|
|
// go test -run TestLLMTalkBaseline ./internal/phraser/eval/
|
|
//
|
|
// Reports, does not assert a quality bar — the numbers are the input to tuning
|
|
// the persona prompt. The one thing worth failing on is a harness fault.
|
|
func TestLLMTalkBaseline(t *testing.T) {
|
|
base := os.Getenv("MAVEN_LLM_URL")
|
|
if base == "" {
|
|
t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server (see doc comment)")
|
|
}
|
|
noProxyLoopback(t)
|
|
|
|
ctx := context.Background()
|
|
f, err := LoadTalk()
|
|
if err != nil {
|
|
t.Fatalf("LoadTalk: %v", err)
|
|
}
|
|
|
|
cfg := phraser.DefaultConfig("")
|
|
cfg.Timeout = 5 * time.Minute
|
|
cfg.ContextBlock = func() string { return persona.Facts{}.Block(time.Now()) }
|
|
p := phraser.NewLLMPhraserAt(base, cfg)
|
|
defer p.Close()
|
|
|
|
// Unreachable server is fatal here, not a logged warning, and that differs
|
|
// from the nudge test on purpose. PhraseNudge returns its errors, so a dead
|
|
// server there shows up honestly in the Errors column. PhraseChat and
|
|
// PhraseQuery do NOT: they swallow every failure and return a canned string
|
|
// ("поговорили.", "не знаю.", "вот что я нашла: …"). So on these three paths
|
|
// a dead server produces a full report with 0 errors and a terrible score —
|
|
// a number that looks like bad phrasing and is really no phrasing at all.
|
|
// Refusing to score without a confirmed model is the only guard available
|
|
// until the phraser reports its failures (Vikunja #397).
|
|
model, err := llm.ModelID(ctx, base)
|
|
if err != nil {
|
|
t.Fatalf("no model at %s: %v — refusing to score, these paths hide their errors "+
|
|
"and would report a plausible-looking result off a dead server", base, err)
|
|
}
|
|
t.Logf("scoring model %s at %s", model, base)
|
|
|
|
// The reply path is a separate object in the daemon too: the phraser owns its
|
|
// own llama-server, the replier is handed an llm.Client. Pair scores both.
|
|
block := func() string { return persona.Facts{}.Block(time.Now()) }
|
|
target := Pair{Talker: p, Confirmer: phraser.NewReplier(llm.New(base, cfg.Timeout), block)}
|
|
|
|
rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", target, f)
|
|
if err != nil {
|
|
t.Fatalf("ScoreTalk: %v", err)
|
|
}
|
|
t.Log("\n" + rep.String() + "\nreplies:\n" + rep.Replies() + "\nfailures:\n" + rep.Failures())
|
|
|
|
// And again afterwards: the run takes minutes, and a server that died or got
|
|
// OOM-killed halfway through would leave the first cases scored and the rest
|
|
// silently canned. Checking only at the start would not catch that.
|
|
if _, err := llm.ModelID(ctx, base); err != nil {
|
|
t.Fatalf("model at %s went away during the run: %v — the score above is not trustworthy", base, err)
|
|
}
|
|
}
|