Score the chat, query and knowledge phrasing paths (#395)

The nudge fixture only covered nudges. The shared persona block now goes into
five prompts, and the three conversational ones were unmeasured — those are the
long free-form replies where a persona break is most likely.

Adds talk_v1.json (27 Russian cases, 9 per path) and ScoreTalk, reporting
per-path as well as per-check so a chat regression can be told apart from a
knowledge one. Reuses the persona checks; the nudge-only ones (length, mood,
no questions) are left out, since a chat reply is allowed 1-3 sentences and a
follow-up question. The LLM run is opt-in on MAVEN_LLM_URL as before.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
This commit is contained in:
kami
2026-07-31 16:28:11 +04:00
parent de09471421
commit 64e5f3bdc1
5 changed files with 685 additions and 5 deletions
+37 -2
View File
@@ -574,12 +574,47 @@ func checkCringe(body string) Result {
// checkOnTopic — the message must name the thing the rule is about. A nudge
// that never mentions water leaves the operator with a chime and no action.
func checkOnTopic(c Case, body string) Result {
return checkOnTopicAny(c.WantAny, body)
}
// checkOnTopicAny is the same test over a bare want-list, so the talk scorer can
// reuse it without owning a nudge Case.
func checkOnTopicAny(wantAny []string, body string) Result {
low := strings.ToLower(body)
for _, want := range c.WantAny {
for _, want := range wantAny {
if strings.Contains(low, strings.ToLower(want)) {
return Result{CheckOnTopic, true, ""}
}
}
return Result{CheckOnTopic, false,
fmt.Sprintf("mentions none of %v", c.WantAny)}
fmt.Sprintf("mentions none of %v", wantAny)}
}
// --- shape checks for the free-form paths --------------------------------
//
// The nudge checks assume one short sentence. Chat and query replies are longer
// by design, so the only shape worth testing there is that the model produced a
// reply at all and did not trail off. Both are failure modes the fallbacks in
// llmphraser.go hide: a truncated or empty generation still returns nil error.
const (
CheckNonEmpty = "nonempty" // she said something
CheckEllipsis = "ellipsis" // she finished the sentence
)
func checkNonEmpty(body string) Result {
if strings.TrimSpace(body) == "" {
return Result{CheckNonEmpty, false, "empty reply"}
}
return Result{CheckNonEmpty, true, ""}
}
// checkEllipsis — a reply ending in "…" or "..." is a generation that ran out of
// tokens, not a stylistic pause. Mid-sentence ellipses are left alone.
func checkEllipsis(body string) Result {
trimmed := strings.TrimRight(strings.TrimSpace(body), `"'»)`)
if strings.HasSuffix(trimmed, "…") || strings.HasSuffix(trimmed, "...") {
return Result{CheckEllipsis, false, "reply trails off in an ellipsis — likely truncated"}
}
return Result{CheckEllipsis, true, ""}
}