From 64e5f3bdc116e3d288e2010797875ec7ab4c1fbe Mon Sep 17 00:00:00 2001 From: kami Date: Fri, 31 Jul 2026 16:28:11 +0400 Subject: [PATCH] Score the chat, query and knowledge phrasing paths (#395) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The nudge fixture only covered nudges. The shared persona block now goes into five prompts, and the three conversational ones were unmeasured — those are the long free-form replies where a persona break is most likely. Adds talk_v1.json (27 Russian cases, 9 per path) and ScoreTalk, reporting per-path as well as per-check so a chat regression can be told apart from a knowledge one. Reuses the persona checks; the nudge-only ones (length, mood, no questions) are left out, since a chat reply is allowed 1-3 sentences and a follow-up question. The LLM run is opt-in on MAVEN_LLM_URL as before. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ --- Makefile | 9 +- internal/phraser/eval/checks.go | 39 ++++- internal/phraser/eval/talk.go | 265 +++++++++++++++++++++++++++++ internal/phraser/eval/talk_test.go | 151 ++++++++++++++++ internal/phraser/eval/talk_v1.json | 226 ++++++++++++++++++++++++ 5 files changed, 685 insertions(+), 5 deletions(-) create mode 100644 internal/phraser/eval/talk.go create mode 100644 internal/phraser/eval/talk_test.go create mode 100644 internal/phraser/eval/talk_v1.json diff --git a/Makefile b/Makefile index 6bfa9c0..5e34b96 100644 --- a/Makefile +++ b/Makefile @@ -103,14 +103,17 @@ eval-router: eval-recall: MAVEN_ONNX_LIB="$(MAVEN_ONNX_LIB)" $(GO) test -v -count=1 ./internal/memory/recalleval/ -# eval-phrasing -- score nudge phrasing (internal/phraser/eval). Verbose so the +# eval-phrasing -- score nudge phrasing AND the conversational paths (chat, +# query, general knowledge) in internal/phraser/eval. Verbose so the # report and every generated message land in the terminal. With no environment # it scores the deterministic Stub only, which is what CI runs. Set # MAVEN_LLM_URL to add the resident model: # MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-phrasing -# The model run is slow (minutes) -- the timeout is raised to match. +# The model run is slow (minutes) -- the timeout is raised to match. It covers +# two fixtures now (15 nudges + 27 conversational cases, and the chat replies are +# the long ones), hence 90m rather than 40m. eval-phrasing: - $(GO) test -v -count=1 -timeout 40m ./internal/phraser/eval/ + $(GO) test -v -count=1 -timeout 90m ./internal/phraser/eval/ # eval-models — score ONE llama-server against the same fixture, for the # resident-model bake-off (#278, #250). Start a server with the gguf you want, diff --git a/internal/phraser/eval/checks.go b/internal/phraser/eval/checks.go index ade60b7..b3ae2e2 100644 --- a/internal/phraser/eval/checks.go +++ b/internal/phraser/eval/checks.go @@ -574,12 +574,47 @@ func checkCringe(body string) Result { // checkOnTopic — the message must name the thing the rule is about. A nudge // that never mentions water leaves the operator with a chime and no action. func checkOnTopic(c Case, body string) Result { + return checkOnTopicAny(c.WantAny, body) +} + +// checkOnTopicAny is the same test over a bare want-list, so the talk scorer can +// reuse it without owning a nudge Case. +func checkOnTopicAny(wantAny []string, body string) Result { low := strings.ToLower(body) - for _, want := range c.WantAny { + for _, want := range wantAny { if strings.Contains(low, strings.ToLower(want)) { return Result{CheckOnTopic, true, ""} } } return Result{CheckOnTopic, false, - fmt.Sprintf("mentions none of %v", c.WantAny)} + fmt.Sprintf("mentions none of %v", wantAny)} +} + +// --- shape checks for the free-form paths -------------------------------- +// +// The nudge checks assume one short sentence. Chat and query replies are longer +// by design, so the only shape worth testing there is that the model produced a +// reply at all and did not trail off. Both are failure modes the fallbacks in +// llmphraser.go hide: a truncated or empty generation still returns nil error. + +const ( + CheckNonEmpty = "nonempty" // she said something + CheckEllipsis = "ellipsis" // she finished the sentence +) + +func checkNonEmpty(body string) Result { + if strings.TrimSpace(body) == "" { + return Result{CheckNonEmpty, false, "empty reply"} + } + return Result{CheckNonEmpty, true, ""} +} + +// checkEllipsis — a reply ending in "…" or "..." is a generation that ran out of +// tokens, not a stylistic pause. Mid-sentence ellipses are left alone. +func checkEllipsis(body string) Result { + trimmed := strings.TrimRight(strings.TrimSpace(body), `"'»)`) + if strings.HasSuffix(trimmed, "…") || strings.HasSuffix(trimmed, "...") { + return Result{CheckEllipsis, false, "reply trails off in an ellipsis — likely truncated"} + } + return Result{CheckEllipsis, true, ""} } diff --git a/internal/phraser/eval/talk.go b/internal/phraser/eval/talk.go new file mode 100644 index 0000000..2d071a7 --- /dev/null +++ b/internal/phraser/eval/talk.go @@ -0,0 +1,265 @@ +package eval + +// This file scores the CONVERSATIONAL paths, the ones the nudge fixture never +// touches: chat, query-with-notes, and general knowledge. All three now carry +// the shared persona block (internal/persona), and all three produce long +// free-form Russian — which is exactly where a persona break (formality, third +// person, masculine self-reference) is most likely and where, until this file, +// nothing could see one. +// +// Why a second fixture instead of more nudge cases: the checks differ. A nudge +// must be one short sentence with no question in it; a chat reply is allowed +// 1-3 sentences and a follow-up question is a FEATURE there. Mixing them would +// need per-case check masks, and the nudge scorer stays untouched this way. +// +// Why per-path reporting: a chat regression and a knowledge regression have +// different causes (chat prompt vs router.KnowledgePrompt), and one blended +// percentage cannot tell them apart. + +import ( + "context" + _ "embed" + "encoding/json" + "fmt" + "sort" + "strings" + "time" + + "github.com/kami/maven/internal/dialogue" +) + +//go:embed talk_v1.json +var talkFixtureJSON []byte + +// The three phrasing paths under test. Values match the fixture's "path" field. +const ( + PathChat = "chat" // PhraseChat + PathQuery = "query" // PhraseQuery with notes + PathKnowledge = "knowledge" // PhraseQuery with no notes +) + +// TalkPaths — report order. +var TalkPaths = []string{PathChat, PathQuery, PathKnowledge} + +// TalkCheckNames — the checks that apply to a free-form reply, in report order. +// Deliberately a subset of CheckNames: length, mood and "no questions" are nudge +// properties and would fail a correct chat reply. These paths return no mood at +// all, so there is nothing to check there. +var TalkCheckNames = []string{ + CheckNonEmpty, CheckEllipsis, CheckLang, CheckFeminine, CheckAddress, CheckOnTopic, +} + +// TalkCase — one turn as the daemon would present it. +// +// History is flat text because that is all PhraseChat uses (it concatenates +// turn texts into one user message); intents and slots would be dead fields. +// Notes are what the store would have matched for a query. +// +// WantAny is the on-topic contract: at least one lowercased fragment must appear +// in the reply. Fragments are stems ("пароль" → "парол") so declension does not +// defeat them. +type TalkCase struct { + ID string `json:"id"` + Path string `json:"path"` + Utterance string `json:"utterance"` + History []string `json:"history,omitempty"` + Notes []string `json:"notes,omitempty"` + WantAny []string `json:"want_any"` + Tags []string `json:"tags,omitempty"` + Note string `json:"note,omitempty"` +} + +// TalkFixture — the versioned envelope, same gating as Fixture. +type TalkFixture struct { + SchemaVersion int `json:"schema_version"` + Name string `json:"name"` + Notes []string `json:"notes"` + Cases []TalkCase `json:"cases"` +} + +// LoadTalk returns the embedded conversational fixture. +func LoadTalk() (TalkFixture, error) { + var f TalkFixture + if err := json.Unmarshal(talkFixtureJSON, &f); err != nil { + return TalkFixture{}, fmt.Errorf("parse talk fixture: %w", err) + } + if f.SchemaVersion != SchemaVersion { + return TalkFixture{}, fmt.Errorf("talk fixture schema_version %d, want %d", f.SchemaVersion, SchemaVersion) + } + if len(f.Cases) == 0 { + return TalkFixture{}, fmt.Errorf("talk fixture has no cases") + } + return f, nil +} + +// Talker — the two methods a conversational path must have to be scorable. +// *phraser.LLMPhraser satisfies it; same trick as Nudger. +type Talker interface { + PhraseChat(ctx context.Context, utterance string, history []dialogue.Turn) (string, error) + PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error) +} + +// TalkOutcome — one scored case. +type TalkOutcome struct { + Case TalkCase + Reply string + Err error + Latency time.Duration + Pass bool + Failed []string + Reasons []string +} + +// TalkReport — the aggregate. ByPath is the point of this scorer. +type TalkReport struct { + Name string + Total int + Passed int + Errors int + ByCheck map[string]int + ByPath map[string]TagStat + Outcomes []TalkOutcome + P50 time.Duration + P95 time.Duration + Max time.Duration +} + +// Accuracy — fraction of cases that passed every check. +func (r TalkReport) Accuracy() float64 { + if r.Total == 0 { + return 0 + } + return float64(r.Passed) / float64(r.Total) +} + +// ScoreTalk runs every case through t and aggregates. A phrasing error scores as +// a miss and is counted separately: "the model was down" and "the model wrote +// something bad" must not be the same number. +func ScoreTalk(ctx context.Context, name string, t Talker, f TalkFixture) (TalkReport, error) { + rep := TalkReport{ + Name: name, + Total: len(f.Cases), + ByCheck: map[string]int{}, + ByPath: map[string]TagStat{}, + } + for _, n := range TalkCheckNames { + rep.ByCheck[n] = 0 + } + lat := make([]time.Duration, 0, len(f.Cases)) + + for _, c := range f.Cases { + start := time.Now() + reply, err := c.run(ctx, t) + o := TalkOutcome{Case: c, Reply: reply, Err: err, Latency: time.Since(start)} + lat = append(lat, o.Latency) + + if err != nil { + rep.Errors++ + o.Failed = append(o.Failed, "call") + o.Reasons = append(o.Reasons, fmt.Sprintf("phrase error: %v", err)) + } else { + for _, res := range RunTalkChecks(c, reply) { + if res.Pass { + rep.ByCheck[res.Name]++ + continue + } + o.Failed = append(o.Failed, res.Name) + o.Reasons = append(o.Reasons, res.Name+": "+res.Detail) + } + } + + o.Pass = len(o.Failed) == 0 + if o.Pass { + rep.Passed++ + } + bump(rep.ByPath, c.Path, o.Pass) + rep.Outcomes = append(rep.Outcomes, o) + } + + sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] }) + rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95) + if len(lat) > 0 { + rep.Max = lat[len(lat)-1] + } + return rep, nil +} + +// run dispatches the case to its path. knowledge and query are the same method; +// the empty notes slice is what selects the no-notes branch inside PhraseQuery. +func (c TalkCase) run(ctx context.Context, t Talker) (string, error) { + switch c.Path { + case PathChat: + return t.PhraseChat(ctx, c.Utterance, c.turns()) + case PathQuery: + return t.PhraseQuery(ctx, c.Utterance, c.Notes) + case PathKnowledge: + return t.PhraseQuery(ctx, c.Utterance, nil) + } + return "", fmt.Errorf("unknown path %q", c.Path) +} + +func (c TalkCase) turns() []dialogue.Turn { + turns := make([]dialogue.Turn, 0, len(c.History)) + for _, h := range c.History { + turns = append(turns, dialogue.Turn{Text: h}) + } + return turns +} + +// RunTalkChecks scores one reply. Order matches TalkCheckNames. +func RunTalkChecks(c TalkCase, reply string) []Result { + return []Result{ + checkNonEmpty(reply), + checkEllipsis(reply), + checkLang(reply), + checkFeminine(reply), + checkAddress(reply), + checkOnTopicAny(c.WantAny, reply), + } +} + +// String renders the comparison table — composite, then per-check so a +// regression names the property, then per-path so it names the prompt. +func (r TalkReport) String() string { + var b strings.Builder + fmt.Fprintf(&b, "%s: %d/%d cases pass every check (%.1f%%), %d errors\n", + r.Name, r.Passed, r.Total, 100*r.Accuracy(), r.Errors) + for _, name := range TalkCheckNames { + fmt.Fprintf(&b, " %-10s %d/%d\n", name, r.ByCheck[name], r.Total) + } + fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max) + fmt.Fprintf(&b, " by path: %s\n", renderStats(r.ByPath)) + return b.String() +} + +// Failures — per-case detail, sorted by ID so two runs diff cleanly. +func (r TalkReport) Failures() string { + var b strings.Builder + for _, o := range r.sorted() { + if o.Pass { + continue + } + fmt.Fprintf(&b, " %s %q\n %s\n", o.Case.ID, o.Reply, strings.Join(o.Reasons, "; ")) + } + return b.String() +} + +// Replies — every generated reply verbatim. This is what a human reads to judge +// tone; the score only says which checks fired. +func (r TalkReport) Replies() string { + var b strings.Builder + for _, o := range r.sorted() { + mark := "ok " + if !o.Pass { + mark = "FAIL" + } + fmt.Fprintf(&b, " %s %-9s %-22s %q\n", mark, o.Case.Path, o.Case.ID, o.Reply) + } + return b.String() +} + +func (r TalkReport) sorted() []TalkOutcome { + out := append([]TalkOutcome(nil), r.Outcomes...) + sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID }) + return out +} diff --git a/internal/phraser/eval/talk_test.go b/internal/phraser/eval/talk_test.go new file mode 100644 index 0000000..71d1963 --- /dev/null +++ b/internal/phraser/eval/talk_test.go @@ -0,0 +1,151 @@ +package eval + +import ( + "context" + "os" + "strings" + "testing" + "time" + + "github.com/kami/maven/internal/dialogue" + "github.com/kami/maven/internal/llm" + "github.com/kami/maven/internal/persona" + "github.com/kami/maven/internal/phraser" +) + +// perPathMinimum — the resolution floor. A per-path score built on a handful of +// cases moves by 12% when a single reply changes, which cannot distinguish a +// prompt regression from noise. +const perPathMinimum = 8 + +// TestTalkFixture — the fixture itself has to be sound before any score off it +// means anything. +func TestTalkFixture(t *testing.T) { + f, err := LoadTalk() + if err != nil { + t.Fatalf("LoadTalk: %v", err) + } + + seen := map[string]bool{} + byPath := map[string]int{} + for _, c := range f.Cases { + if seen[c.ID] { + t.Errorf("duplicate case id %q", c.ID) + } + seen[c.ID] = true + + switch c.Path { + case PathChat, PathQuery, PathKnowledge: + default: + t.Errorf("%s: unknown path %q", c.ID, c.Path) + } + byPath[c.Path]++ + + if strings.TrimSpace(c.Utterance) == "" { + t.Errorf("%s: empty utterance", c.ID) + } + if len(c.WantAny) == 0 { + t.Errorf("%s: no want_any — the reply cannot be checked for topic", c.ID) + } + // A query case with no notes would silently score the knowledge path. + if c.Path == PathQuery && len(c.Notes) == 0 { + t.Errorf("%s: query case has no notes", c.ID) + } + if c.Path == PathKnowledge && len(c.Notes) > 0 { + t.Errorf("%s: knowledge case must have no notes", c.ID) + } + } + + for _, p := range TalkPaths { + if byPath[p] < perPathMinimum { + t.Errorf("path %s has %d cases, want at least %d", p, byPath[p], perPathMinimum) + } + } +} + +// fakeTalker — a scripted Talker, so the scorer is testable without a model. +type fakeTalker struct{ reply string } + +func (f fakeTalker) PhraseChat(context.Context, string, []dialogue.Turn) (string, error) { + return f.reply, nil +} +func (f fakeTalker) PhraseQuery(context.Context, string, []string) (string, error) { + return f.reply, nil +} + +// TestScoreTalkCounts — a reply that fails on purpose must be counted on every +// path, so a real run cannot report a hidden zero. +func TestScoreTalkCounts(t *testing.T) { + f, err := LoadTalk() + if err != nil { + t.Fatalf("LoadTalk: %v", err) + } + // Formal address, off-topic, trailing ellipsis: three checks fail at once. + rep, err := ScoreTalk(context.Background(), "fake", fakeTalker{"Приходите, я вас жду…"}, f) + if err != nil { + t.Fatalf("ScoreTalk: %v", err) + } + if rep.Total != len(f.Cases) || rep.Passed != 0 { + t.Errorf("got %d/%d passing, want 0/%d", rep.Passed, rep.Total, len(f.Cases)) + } + if rep.ByCheck[CheckAddress] != 0 { + t.Errorf("formal reply passed the address check %d times", rep.ByCheck[CheckAddress]) + } + if rep.ByCheck[CheckEllipsis] != 0 { + t.Errorf("truncated reply passed the ellipsis check %d times", rep.ByCheck[CheckEllipsis]) + } + for _, p := range TalkPaths { + if rep.ByPath[p].Total == 0 { + t.Errorf("path %s missing from the report", p) + } + } + if !strings.Contains(rep.String(), "by path") { + t.Error("report does not break down by path") + } +} + +// TestLLMTalkBaseline — the resident model on the three conversational paths. +// Opt-in exactly like TestLLMPhrasingBaseline: CI has no model and a run costs +// minutes on the CPU target. +// +// MAVEN_LLM_URL=http://127.0.0.1:18099 \ +// go test -run TestLLMTalkBaseline ./internal/phraser/eval/ +// +// Reports, does not assert a quality bar — the numbers are the input to tuning +// the persona prompt. The one thing worth failing on is a harness fault. +func TestLLMTalkBaseline(t *testing.T) { + base := os.Getenv("MAVEN_LLM_URL") + if base == "" { + t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server (see doc comment)") + } + noProxyLoopback(t) + + ctx := context.Background() + f, err := LoadTalk() + if err != nil { + t.Fatalf("LoadTalk: %v", err) + } + + cfg := phraser.DefaultConfig("") + cfg.Timeout = 5 * time.Minute + cfg.ContextBlock = func() string { return persona.Facts{}.Block(time.Now()) } + p := phraser.NewLLMPhraserAt(base, cfg) + defer p.Close() + + model, err := llm.ModelID(ctx, base) + if err != nil { + t.Logf("could not read model id from %s: %v — report will say %q", base, err, llm.UnknownModel) + model = llm.UnknownModel + } + t.Logf("scoring model %s at %s", model, base) + + rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", p, f) + if err != nil { + t.Fatalf("ScoreTalk: %v", err) + } + t.Log("\n" + rep.String() + "\nreplies:\n" + rep.Replies() + "\nfailures:\n" + rep.Failures()) + + if rep.Errors == rep.Total { + t.Errorf("all %d cases errored — harness fault, not a measurement", rep.Total) + } +} diff --git a/internal/phraser/eval/talk_v1.json b/internal/phraser/eval/talk_v1.json new file mode 100644 index 0000000..6076079 --- /dev/null +++ b/internal/phraser/eval/talk_v1.json @@ -0,0 +1,226 @@ +{ + "schema_version": 1, + "name": "ru-talk-v1", + "notes": [ + "Scores the three conversational phrasing paths: chat (PhraseChat), query (PhraseQuery with notes) and knowledge (PhraseQuery with no notes). The nudge fixture does not cover any of them.", + "Nine cases per path, not five. The nudge fixture is 15 sampled cases and cannot resolve a change smaller than ~3 cases; a per-path score off five cases would be worse still. More cases per path is the point of this fixture.", + "The owner is a man, addressed informally as ty, living alone with a home server. Every utterance is written the way he actually talks to her.", + "chat-formality-bait and chat-about-me exist to provoke the two persona breaks the nudge eval caught: the formal vy/vas plural, and talking about him in the third person.", + "want_any fragments are stems so Russian declension does not defeat the on-topic check. They are lowercased before comparison.", + "Notes are written as the store would have them: short, first person, no punctuation discipline." + ], + "cases": [ + { + "id": "chat-how-are-you", + "path": "chat", + "utterance": "привет, как дела?", + "want_any": ["норм", "хорош", "порядк", "тут", "работ"], + "tags": ["greeting"], + "note": "The plainest chat turn there is. If the persona breaks anywhere it breaks here first." + }, + { + "id": "chat-formality-bait", + "path": "chat", + "utterance": "не могли бы вы подсказать, чем вы сейчас занимаетесь?", + "want_any": ["сейчас", "ничем", "ничего", "жду", "тут"], + "tags": ["persona-bait", "address"], + "note": "Deliberately polite and plural. A small model mirrors the register and answers with vy/vas — the exact break the address check was written for." + }, + { + "id": "chat-about-me", + "path": "chat", + "utterance": "расскажи обо мне", + "want_any": ["ты", "тебя", "теб"], + "tags": ["persona-bait", "third-person"], + "note": "Baits the third person: she should say 'ты живёшь один', not 'он живёт один', as if reporting to somebody else." + }, + { + "id": "chat-bored-evening", + "path": "chat", + "utterance": "скучно что-то вечером, посоветуй чем заняться", + "want_any": ["можеш", "попробу", "почита", "прогул", "фильм", "серв"], + "tags": ["open-ended"] + }, + { + "id": "chat-followup-server", + "path": "chat", + "utterance": "а стоит его вообще перезагружать?", + "history": ["сервер опять шумит как самолёт", "похоже вентилятор"], + "want_any": ["серв", "перезагру", "вентил", "шум"], + "tags": ["history", "anaphora"], + "note": "The pronoun 'его' only resolves through history. Also the one case where 'он' about the server is legitimate." + }, + { + "id": "chat-tired", + "path": "chat", + "utterance": "устал я сегодня, весь день за компом", + "want_any": ["отдохн", "устал", "перерыв", "спат", "день"], + "tags": ["tone"], + "note": "Invites the fake-concern and emotional-support drift; the reply should stay plain." + }, + { + "id": "chat-thanks", + "path": "chat", + "utterance": "спасибо, выручила", + "want_any": ["пожалуйст", "не за что", "рада", "обращ"], + "tags": ["persona", "feminine"], + "note": "Feminine self-reference is unavoidable in an answer to thanks: 'рада', not 'рад'." + }, + { + "id": "chat-what-can-you-do", + "path": "chat", + "utterance": "что ты вообще умеешь?", + "want_any": ["напомн", "замет", "запис", "могу", "умею"], + "tags": ["self-description", "feminine"] + }, + { + "id": "chat-joke", + "path": "chat", + "utterance": "расскажи что-нибудь смешное", + "want_any": ["анекдот", "шутк", "смешн", "истори"], + "tags": ["open-ended"], + "note": "Longest free-form generation in the chat set — the most likely place for a truncated reply." + }, + { + "id": "query-router-password", + "path": "query", + "utterance": "что я записывал про пароль от роутера?", + "notes": ["пароль от роутера admin/xxK9tp — на наклейке снизу", "роутер висит в коридоре"], + "want_any": ["парол", "роутер", "наклейк"], + "tags": ["notes", "recall"] + }, + { + "id": "query-bedtime-yesterday", + "path": "query", + "utterance": "напомни, во сколько я вчера лёг?", + "notes": ["лёг спать в 02:40", "сегодня встал в 9"], + "want_any": ["02:40", "2:40", "полтрет", "ноч"], + "tags": ["notes", "time"] + }, + { + "id": "query-doctor-name", + "path": "query", + "utterance": "как звали того стоматолога, которого мне советовали?", + "notes": ["стоматолог Игорь Валерьевич, клиника на Ленина, советовал Дима"], + "want_any": ["игор", "валерьев", "стоматолог"], + "tags": ["notes", "recall"] + }, + { + "id": "query-disk-plan", + "path": "query", + "utterance": "я что-то планировал с диском на сервере, что именно?", + "notes": ["купить второй hdd на 4тб под бэкапы", "перенести медиатеку с системного диска"], + "want_any": ["hdd", "бэкап", "диск", "4тб", "медиатек"], + "tags": ["notes", "homeserver"] + }, + { + "id": "query-notes-do-not-answer", + "path": "query", + "utterance": "сколько я заплатил за домен?", + "notes": ["домен продлевается в марте", "хостинг оплачен на год вперёд"], + "want_any": ["домен", "не зна", "не указ", "нет"], + "tags": ["notes", "negative"], + "note": "The notes do not contain the price. The prompt tells her to say so; a made-up number is the failure being watched for." + }, + { + "id": "query-single-note", + "path": "query", + "utterance": "где лежит запасной ключ?", + "notes": ["запасной ключ у соседа с четвёртого этажа"], + "want_any": ["ключ", "сосед", "четверт"], + "tags": ["notes", "single"], + "note": "One note only — PhraseQuery has a separate branch for len(notes) == 1." + }, + { + "id": "query-polite-form", + "path": "query", + "utterance": "подскажите, пожалуйста, что у меня записано по машине?", + "notes": ["замена масла на 92 тысячах", "страховка до 14 сентября"], + "want_any": ["масл", "страховк", "92", "сентябр"], + "tags": ["notes", "persona-bait", "address"], + "note": "Polite plural in the question. The answer must still be ty." + }, + { + "id": "query-shopping", + "path": "query", + "utterance": "что мне надо было купить?", + "notes": ["купить кофе и фильтры", "закончилась паста"], + "want_any": ["кофе", "фильтр", "паст"], + "tags": ["notes", "list"] + }, + { + "id": "query-wifi-guest", + "path": "query", + "utterance": "я записывал гостевой вайфай?", + "notes": ["гостевая сеть maven-guest, пароль 12345678 меняю раз в месяц"], + "want_any": ["guest", "гостев", "12345678", "парол"], + "tags": ["notes", "recall"] + }, + { + "id": "know-sky-blue", + "path": "knowledge", + "utterance": "почему небо синее?", + "want_any": ["све", "рассеи", "атмосфер", "син", "волн"], + "tags": ["general"] + }, + { + "id": "know-boil-egg", + "path": "knowledge", + "utterance": "сколько варить яйцо вкрутую?", + "want_any": ["минут", "8", "9", "10", "варит"], + "tags": ["general", "practical"] + }, + { + "id": "know-ssd-vs-hdd", + "path": "knowledge", + "utterance": "чем ssd отличается от hdd?", + "want_any": ["ssd", "hdd", "быстр", "диск", "механич"], + "tags": ["general", "tech"] + }, + { + "id": "know-cat-purr", + "path": "knowledge", + "utterance": "почему кошки мурчат?", + "want_any": ["кош", "мурч", "вибра", "успока"], + "tags": ["general"] + }, + { + "id": "know-hiccups", + "path": "knowledge", + "utterance": "как быстро избавиться от икоты?", + "want_any": ["икот", "дыха", "вод", "задерж"], + "tags": ["general", "practical"] + }, + { + "id": "know-polite-form", + "path": "knowledge", + "utterance": "не могли бы вы объяснить, что такое vpn?", + "want_any": ["vpn", "туннел", "трафик", "сет", "шифр"], + "tags": ["general", "persona-bait", "address"], + "note": "Polite plural bait on the knowledge prompt, which is a different system prompt from chat and must hold the same line." + }, + { + "id": "know-dont-know", + "path": "knowledge", + "utterance": "как зовут моего соседа снизу?", + "want_any": ["не зна", "не мог", "нет"], + "tags": ["general", "negative"], + "note": "Unanswerable without notes. Admitting it beats inventing a name; watching for the invention." + }, + { + "id": "know-water-per-day", + "path": "knowledge", + "utterance": "сколько воды в день надо пить?", + "want_any": ["вод", "литр", "стакан", "пит"], + "tags": ["general", "health"], + "note": "Overlaps a nudge rule on purpose: the knowledge answer must not turn into a nudge." + }, + { + "id": "know-thunder-delay", + "path": "knowledge", + "utterance": "почему гром слышно позже молнии?", + "want_any": ["звук", "све", "быстр", "гром", "молни"], + "tags": ["general"] + } + ] +}