Files
Maven/internal/phraser/eval/talk.go
T
kami 64e5f3bdc1 Score the chat, query and knowledge phrasing paths (#395)
The nudge fixture only covered nudges. The shared persona block now goes into
five prompts, and the three conversational ones were unmeasured — those are the
long free-form replies where a persona break is most likely.

Adds talk_v1.json (27 Russian cases, 9 per path) and ScoreTalk, reporting
per-path as well as per-check so a chat regression can be told apart from a
knowledge one. Reuses the persona checks; the nudge-only ones (length, mood,
no questions) are left out, since a chat reply is allowed 1-3 sentences and a
follow-up question. The LLM run is opt-in on MAVEN_LLM_URL as before.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 16:28:11 +04:00

266 lines
8.2 KiB
Go

package eval
// This file scores the CONVERSATIONAL paths, the ones the nudge fixture never
// touches: chat, query-with-notes, and general knowledge. All three now carry
// the shared persona block (internal/persona), and all three produce long
// free-form Russian — which is exactly where a persona break (formality, third
// person, masculine self-reference) is most likely and where, until this file,
// nothing could see one.
//
// Why a second fixture instead of more nudge cases: the checks differ. A nudge
// must be one short sentence with no question in it; a chat reply is allowed
// 1-3 sentences and a follow-up question is a FEATURE there. Mixing them would
// need per-case check masks, and the nudge scorer stays untouched this way.
//
// Why per-path reporting: a chat regression and a knowledge regression have
// different causes (chat prompt vs router.KnowledgePrompt), and one blended
// percentage cannot tell them apart.
import (
"context"
_ "embed"
"encoding/json"
"fmt"
"sort"
"strings"
"time"
"github.com/kami/maven/internal/dialogue"
)
//go:embed talk_v1.json
var talkFixtureJSON []byte
// The three phrasing paths under test. Values match the fixture's "path" field.
const (
PathChat = "chat" // PhraseChat
PathQuery = "query" // PhraseQuery with notes
PathKnowledge = "knowledge" // PhraseQuery with no notes
)
// TalkPaths — report order.
var TalkPaths = []string{PathChat, PathQuery, PathKnowledge}
// TalkCheckNames — the checks that apply to a free-form reply, in report order.
// Deliberately a subset of CheckNames: length, mood and "no questions" are nudge
// properties and would fail a correct chat reply. These paths return no mood at
// all, so there is nothing to check there.
var TalkCheckNames = []string{
CheckNonEmpty, CheckEllipsis, CheckLang, CheckFeminine, CheckAddress, CheckOnTopic,
}
// TalkCase — one turn as the daemon would present it.
//
// History is flat text because that is all PhraseChat uses (it concatenates
// turn texts into one user message); intents and slots would be dead fields.
// Notes are what the store would have matched for a query.
//
// WantAny is the on-topic contract: at least one lowercased fragment must appear
// in the reply. Fragments are stems ("пароль" → "парол") so declension does not
// defeat them.
type TalkCase struct {
ID string `json:"id"`
Path string `json:"path"`
Utterance string `json:"utterance"`
History []string `json:"history,omitempty"`
Notes []string `json:"notes,omitempty"`
WantAny []string `json:"want_any"`
Tags []string `json:"tags,omitempty"`
Note string `json:"note,omitempty"`
}
// TalkFixture — the versioned envelope, same gating as Fixture.
type TalkFixture struct {
SchemaVersion int `json:"schema_version"`
Name string `json:"name"`
Notes []string `json:"notes"`
Cases []TalkCase `json:"cases"`
}
// LoadTalk returns the embedded conversational fixture.
func LoadTalk() (TalkFixture, error) {
var f TalkFixture
if err := json.Unmarshal(talkFixtureJSON, &f); err != nil {
return TalkFixture{}, fmt.Errorf("parse talk fixture: %w", err)
}
if f.SchemaVersion != SchemaVersion {
return TalkFixture{}, fmt.Errorf("talk fixture schema_version %d, want %d", f.SchemaVersion, SchemaVersion)
}
if len(f.Cases) == 0 {
return TalkFixture{}, fmt.Errorf("talk fixture has no cases")
}
return f, nil
}
// Talker — the two methods a conversational path must have to be scorable.
// *phraser.LLMPhraser satisfies it; same trick as Nudger.
type Talker interface {
PhraseChat(ctx context.Context, utterance string, history []dialogue.Turn) (string, error)
PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error)
}
// TalkOutcome — one scored case.
type TalkOutcome struct {
Case TalkCase
Reply string
Err error
Latency time.Duration
Pass bool
Failed []string
Reasons []string
}
// TalkReport — the aggregate. ByPath is the point of this scorer.
type TalkReport struct {
Name string
Total int
Passed int
Errors int
ByCheck map[string]int
ByPath map[string]TagStat
Outcomes []TalkOutcome
P50 time.Duration
P95 time.Duration
Max time.Duration
}
// Accuracy — fraction of cases that passed every check.
func (r TalkReport) Accuracy() float64 {
if r.Total == 0 {
return 0
}
return float64(r.Passed) / float64(r.Total)
}
// ScoreTalk runs every case through t and aggregates. A phrasing error scores as
// a miss and is counted separately: "the model was down" and "the model wrote
// something bad" must not be the same number.
func ScoreTalk(ctx context.Context, name string, t Talker, f TalkFixture) (TalkReport, error) {
rep := TalkReport{
Name: name,
Total: len(f.Cases),
ByCheck: map[string]int{},
ByPath: map[string]TagStat{},
}
for _, n := range TalkCheckNames {
rep.ByCheck[n] = 0
}
lat := make([]time.Duration, 0, len(f.Cases))
for _, c := range f.Cases {
start := time.Now()
reply, err := c.run(ctx, t)
o := TalkOutcome{Case: c, Reply: reply, Err: err, Latency: time.Since(start)}
lat = append(lat, o.Latency)
if err != nil {
rep.Errors++
o.Failed = append(o.Failed, "call")
o.Reasons = append(o.Reasons, fmt.Sprintf("phrase error: %v", err))
} else {
for _, res := range RunTalkChecks(c, reply) {
if res.Pass {
rep.ByCheck[res.Name]++
continue
}
o.Failed = append(o.Failed, res.Name)
o.Reasons = append(o.Reasons, res.Name+": "+res.Detail)
}
}
o.Pass = len(o.Failed) == 0
if o.Pass {
rep.Passed++
}
bump(rep.ByPath, c.Path, o.Pass)
rep.Outcomes = append(rep.Outcomes, o)
}
sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] })
rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95)
if len(lat) > 0 {
rep.Max = lat[len(lat)-1]
}
return rep, nil
}
// run dispatches the case to its path. knowledge and query are the same method;
// the empty notes slice is what selects the no-notes branch inside PhraseQuery.
func (c TalkCase) run(ctx context.Context, t Talker) (string, error) {
switch c.Path {
case PathChat:
return t.PhraseChat(ctx, c.Utterance, c.turns())
case PathQuery:
return t.PhraseQuery(ctx, c.Utterance, c.Notes)
case PathKnowledge:
return t.PhraseQuery(ctx, c.Utterance, nil)
}
return "", fmt.Errorf("unknown path %q", c.Path)
}
func (c TalkCase) turns() []dialogue.Turn {
turns := make([]dialogue.Turn, 0, len(c.History))
for _, h := range c.History {
turns = append(turns, dialogue.Turn{Text: h})
}
return turns
}
// RunTalkChecks scores one reply. Order matches TalkCheckNames.
func RunTalkChecks(c TalkCase, reply string) []Result {
return []Result{
checkNonEmpty(reply),
checkEllipsis(reply),
checkLang(reply),
checkFeminine(reply),
checkAddress(reply),
checkOnTopicAny(c.WantAny, reply),
}
}
// String renders the comparison table — composite, then per-check so a
// regression names the property, then per-path so it names the prompt.
func (r TalkReport) String() string {
var b strings.Builder
fmt.Fprintf(&b, "%s: %d/%d cases pass every check (%.1f%%), %d errors\n",
r.Name, r.Passed, r.Total, 100*r.Accuracy(), r.Errors)
for _, name := range TalkCheckNames {
fmt.Fprintf(&b, " %-10s %d/%d\n", name, r.ByCheck[name], r.Total)
}
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
fmt.Fprintf(&b, " by path: %s\n", renderStats(r.ByPath))
return b.String()
}
// Failures — per-case detail, sorted by ID so two runs diff cleanly.
func (r TalkReport) Failures() string {
var b strings.Builder
for _, o := range r.sorted() {
if o.Pass {
continue
}
fmt.Fprintf(&b, " %s %q\n %s\n", o.Case.ID, o.Reply, strings.Join(o.Reasons, "; "))
}
return b.String()
}
// Replies — every generated reply verbatim. This is what a human reads to judge
// tone; the score only says which checks fired.
func (r TalkReport) Replies() string {
var b strings.Builder
for _, o := range r.sorted() {
mark := "ok "
if !o.Pass {
mark = "FAIL"
}
fmt.Fprintf(&b, " %s %-9s %-22s %q\n", mark, o.Case.Path, o.Case.ID, o.Reply)
}
return b.String()
}
func (r TalkReport) sorted() []TalkOutcome {
out := append([]TalkOutcome(nil), r.Outcomes...)
sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID })
return out
}