diff --git a/cmd/mavend/capture.go b/cmd/mavend/capture.go
index 604f96a..8209a74 100644
--- a/cmd/mavend/capture.go
+++ b/cmd/mavend/capture.go
@@ -100,7 +100,7 @@ func (l llmCompleter) Complete(ctx context.Context, system, user string) (string
// grammar, or a llama-server too old to honour one, gets the plain text it used
// to get rather than an empty meeting summary.
func unwrapSummary(raw string) string {
- s := stripThink(strings.TrimSpace(raw))
+ s := phraser.StripThink(strings.TrimSpace(raw))
start := strings.Index(s, "{")
end := strings.LastIndex(s, "}")
if start < 0 || end <= start {
diff --git a/cmd/mavend/replier_llm.go b/cmd/mavend/replier_llm.go
index db6110d..20967dd 100644
--- a/cmd/mavend/replier_llm.go
+++ b/cmd/mavend/replier_llm.go
@@ -2,122 +2,33 @@ package main
import (
"context"
- "encoding/json"
- "strings"
- "time"
- "github.com/kami/maven/internal/llm"
- "github.com/kami/maven/internal/persona"
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/voice"
)
-// completer is the LLM seam for the replier (subset of router.Completer).
-// *llm.Client satisfies it.
-type completer interface {
- Complete(ctx context.Context, r llm.Req) (string, error)
-}
-
-// llmReplier phrases reactive confirmations with the resident model
-// (Qwen3-1.7B). Stub is the
-// floor on any error (offline-safe). Maven speaks as "she", feminine RU.
+// llmReplier is the daemon-side wiring around phraser.Replier: it owns the
+// deterministic floor, and nothing else. The phrasing itself, the prompt and the
+// output parsing live in internal/phraser so the eval can score them (#396).
type llmReplier struct {
- c completer
+ p *phraser.Replier
stub *voice.StubReplier
-
- // block renders the shared context block per turn (who he is, the time).
- // nil ⇒ the prompt stands alone.
- block func() string
}
-func newLLMReplier(c completer, block func() string) *llmReplier {
- return &llmReplier{c: c, stub: voice.NewStubReplier(), block: block}
+func newLLMReplier(c phraser.Completer, block func() string) *llmReplier {
+ return &llmReplier{p: phraser.NewReplier(c, block), stub: voice.NewStubReplier()}
}
-const replySystem = `Ты — Maven, домашняя ассистентка (о себе — в женском роде). Владелец — мужчина, говоришь с ним на "ты", в единственном числе; никогда не "вы"/"ваш" и не "он"/"его". Подтверди действие РОВНО ОДНИМ коротким предложением (≤120 символов), по-русски, спокойно и без официальных формулировок. Не задавай вопросов, не повторяй слова, не добавляй ничего после точки. Отвечай ТОЛЬКО одним объектом JSON с полями "response" (текст) и "mood" (ровно одно из: neutral, happy, thinking, tired, confused).
-Пример: {"response": "Записала, что ты выпил стакан воды.", "mood": "neutral"}
-Никогда не пиши "..." в поле response.`
-
+// Reply never fails: a clarify, a model error and an unusable generation all
+// answer from the stub, which is what keeps a turn from breaking on the model.
func (r *llmReplier) Reply(d router.Decision) string {
if d.Clarify {
return r.stub.Reply(d)
}
- ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
- defer cancel()
- out, err := r.c.Complete(ctx, llm.Req{
- System: persona.Prepend(r.block, replySystem),
- User: replyContext(d),
- Grammar: phraser.ResponseGrammar,
- MaxTokens: 512,
- RepeatPenalty: 1.3,
- })
- if err != nil {
+ out, err := r.p.PhraseReply(context.Background(), d)
+ if err != nil || out == "" {
return r.stub.Reply(d)
}
- out = stripThink(out)
- if response, _ := parseResponseMood(out); response != "" {
- return response
- }
- // fallback: try plain-text parsing
- if out = firstSentence(out); out != "" {
- return out
- }
- return r.stub.Reply(d)
-}
-
-// firstSentence trims the model's output to a single clean confirmation: first
-// line, first sentence, whitespace-normalized — the last-line defense against a
-// small model that rambles past the first period despite the prompt + stop.
-// stripThink removes the block that Thinking-variant models emit.
-func stripThink(s string) string {
- if i := strings.LastIndex(s, ""); i >= 0 {
- s = strings.TrimSpace(s[i+8:])
- }
- return s
-}
-
-func firstSentence(s string) string {
- s = strings.TrimSpace(s)
- if i := strings.IndexByte(s, '\n'); i >= 0 {
- s = s[:i]
- }
- // keep up to and including the first sentence-ending punctuation.
- if i := strings.IndexAny(s, ".!?"); i >= 0 {
- s = s[:i+1]
- }
- return strings.TrimSpace(s)
-}
-
-// parseResponseMood extracts {"response","mood"} from LLM output, tolerant
-// of thinking tokens and extra text before/after the JSON block.
-func parseResponseMood(raw string) (response, mood string) {
- cleaned := strings.TrimSpace(raw)
- start := strings.Index(cleaned, "{")
- end := strings.LastIndex(cleaned, "}")
- if start < 0 || end < 0 || end <= start {
- return "", ""
- }
- var parsed struct {
- Response string `json:"response"`
- Mood string `json:"mood"`
- }
- if err := json.Unmarshal([]byte(cleaned[start:end+1]), &parsed); err != nil {
- return "", ""
- }
- return parsed.Response, parsed.Mood
-}
-
-// replyContext renders the decision into a compact RU description for the model.
-func replyContext(d router.Decision) string {
- switch d.Intent {
- case router.IntentFact:
- return "записала факт: " + d.Slots.Key + " " + d.Slots.Value
- case router.IntentNote:
- return "сохранила заметку: " + d.Slots.Text
- case router.IntentReminder:
- return "поставила напоминание: " + d.Slots.Text
- default:
- return string(d.Intent) + ": " + d.Slots.Text
- }
+ return out
}
diff --git a/cmd/mavend/replier_llm_test.go b/cmd/mavend/replier_llm_test.go
index 5e084f9..fae6d39 100644
--- a/cmd/mavend/replier_llm_test.go
+++ b/cmd/mavend/replier_llm_test.go
@@ -5,28 +5,22 @@ import (
"testing"
"github.com/kami/maven/internal/llm"
- "github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/voice"
)
-type mockCompleter struct {
+// The phrasing itself is tested in internal/phraser. What is left here is the
+// only thing the daemon adds: the stub floor, on the three ways a reply can
+// fail to arrive.
+type stubCompleter struct {
out string
err error
}
-func (m mockCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return m.out, m.err }
+func (s stubCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return s.out, s.err }
-func TestLLMReplierReturnsLLMReply(t *testing.T) {
- r := newLLMReplier(mockCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
- got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
- if got != "записала, кофе закончился" {
- t.Errorf("got %q, want %q", got, "записала, кофе закончился")
- }
-}
-
-func TestLLMReplierFallsBackToPlainText(t *testing.T) {
- r := newLLMReplier(mockCompleter{out: "записала, кофе закончился"}, nil)
+func TestLLMReplierPassesTheModelReplyThrough(t *testing.T) {
+ r := newLLMReplier(stubCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
if got != "записала, кофе закончился" {
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
@@ -34,54 +28,30 @@ func TestLLMReplierFallsBackToPlainText(t *testing.T) {
}
func TestLLMReplierFallsBackToStubOnError(t *testing.T) {
- r := newLLMReplier(mockCompleter{err: errTestLLMDown}, nil)
- noteDec := router.Decision{Intent: router.IntentNote}
- got := r.Reply(noteDec)
- want := voice.NewStubReplier().Reply(noteDec)
- if got != want {
- t.Errorf("on llm error: got %q, want stub %q", got, want)
- }
+ r := newLLMReplier(stubCompleter{err: errReplierTest}, nil)
+ assertStub(t, r, router.Decision{Intent: router.IntentNote}, "llm error")
}
func TestLLMReplierFallsBackToStubOnEmpty(t *testing.T) {
- r := newLLMReplier(mockCompleter{out: ""}, nil)
- noteDec := router.Decision{Intent: router.IntentNote}
- got := r.Reply(noteDec)
- want := voice.NewStubReplier().Reply(noteDec)
- if got != want {
- t.Errorf("on empty llm: got %q, want stub %q", got, want)
- }
+ r := newLLMReplier(stubCompleter{out: ""}, nil)
+ assertStub(t, r, router.Decision{Intent: router.IntentNote}, "empty llm")
}
func TestLLMReplierClarifyUsesStub(t *testing.T) {
- r := newLLMReplier(mockCompleter{out: "я всё поняла"}, nil)
- clarifyDec := router.Decision{Clarify: true}
- got := r.Reply(clarifyDec)
- want := voice.NewStubReplier().Reply(clarifyDec)
+ r := newLLMReplier(stubCompleter{out: "я всё поняла"}, nil)
+ assertStub(t, r, router.Decision{Clarify: true}, "clarify")
+}
+
+func assertStub(t *testing.T, r *llmReplier, d router.Decision, what string) {
+ t.Helper()
+ got, want := r.Reply(d), voice.NewStubReplier().Reply(d)
if got != want {
- t.Errorf("on clarify: got %q, want stub %q", got, want)
+ t.Errorf("on %s: got %q, want stub %q", what, got, want)
}
}
-var errTestLLMDown = errTest("llm down")
+var errReplierTest = errTest("llm down")
type errTest string
func (e errTest) Error() string { return string(e) }
-
-// grammarRecorder captures the request so the grammar can be asserted on.
-type grammarRecorder struct{ req llm.Req }
-
-func (g *grammarRecorder) Complete(_ context.Context, r llm.Req) (string, error) {
- g.req = r
- return `{"response":"записала","mood":"neutral"}`, nil
-}
-
-func TestLLMReplierCarriesTheResponseGrammar(t *testing.T) {
- rec := &grammarRecorder{}
- r := newLLMReplier(rec, nil)
- r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
- if rec.req.Grammar != phraser.ResponseGrammar {
- t.Errorf("grammar = %q, want phraser.ResponseGrammar", rec.req.Grammar)
- }
-}
diff --git a/internal/phraser/eval/talk.go b/internal/phraser/eval/talk.go
index 2d071a7..6d239f8 100644
--- a/internal/phraser/eval/talk.go
+++ b/internal/phraser/eval/talk.go
@@ -26,20 +26,22 @@ import (
"time"
"github.com/kami/maven/internal/dialogue"
+ "github.com/kami/maven/internal/router"
)
//go:embed talk_v1.json
var talkFixtureJSON []byte
-// The three phrasing paths under test. Values match the fixture's "path" field.
+// The phrasing paths under test. Values match the fixture's "path" field.
const (
PathChat = "chat" // PhraseChat
PathQuery = "query" // PhraseQuery with notes
PathKnowledge = "knowledge" // PhraseQuery with no notes
+ PathReply = "reply" // PhraseReply, the reactive confirmation
)
// TalkPaths — report order.
-var TalkPaths = []string{PathChat, PathQuery, PathKnowledge}
+var TalkPaths = []string{PathChat, PathQuery, PathKnowledge, PathReply}
// TalkCheckNames — the checks that apply to a free-form reply, in report order.
// Deliberately a subset of CheckNames: length, mood and "no questions" are nudge
@@ -58,12 +60,19 @@ var TalkCheckNames = []string{
// WantAny is the on-topic contract: at least one lowercased fragment must appear
// in the reply. Fragments are stems ("пароль" → "парол") so declension does not
// defeat them.
+//
+// Intent, Key and Value carry the reply path's decision: that path is phrased
+// from what the router already resolved, not from the raw utterance. Utterance
+// stays filled anyway, because it is what a human reads in the report.
type TalkCase struct {
ID string `json:"id"`
Path string `json:"path"`
Utterance string `json:"utterance"`
History []string `json:"history,omitempty"`
Notes []string `json:"notes,omitempty"`
+ Intent string `json:"intent,omitempty"`
+ Key string `json:"key,omitempty"`
+ Value string `json:"value,omitempty"`
WantAny []string `json:"want_any"`
Tags []string `json:"tags,omitempty"`
Note string `json:"note,omitempty"`
@@ -92,13 +101,27 @@ func LoadTalk() (TalkFixture, error) {
return f, nil
}
-// Talker — the two methods a conversational path must have to be scorable.
-// *phraser.LLMPhraser satisfies it; same trick as Nudger.
+// Talker — the methods a conversational path must have to be scorable.
+// *phraser.LLMPhraser satisfies the first two; *phraser.Replier satisfies the
+// third, so a run that scores all four paths passes a Pair.
type Talker interface {
PhraseChat(ctx context.Context, utterance string, history []dialogue.Turn) (string, error)
PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error)
}
+// Confirmer — the reply path. *phraser.Replier satisfies it.
+type Confirmer interface {
+ PhraseReply(ctx context.Context, d router.Decision) (string, error)
+}
+
+// Pair joins the two objects the daemon wires separately — the phraser and the
+// replier — so one ScoreTalk call covers every path Maven speaks through. A bare
+// Talker still works; its reply cases score as errors, which is honest.
+type Pair struct {
+ Talker
+ Confirmer
+}
+
// TalkOutcome — one scored case.
type TalkOutcome struct {
Case TalkCase
@@ -194,10 +217,30 @@ func (c TalkCase) run(ctx context.Context, t Talker) (string, error) {
return t.PhraseQuery(ctx, c.Utterance, c.Notes)
case PathKnowledge:
return t.PhraseQuery(ctx, c.Utterance, nil)
+ case PathReply:
+ conf, ok := t.(Confirmer)
+ if !ok {
+ return "", fmt.Errorf("target cannot phrase replies — pass a Pair")
+ }
+ return conf.PhraseReply(ctx, c.decision())
}
return "", fmt.Errorf("unknown path %q", c.Path)
}
+// decision rebuilds what the router would have handed the replier. Text is the
+// utterance for a note or a reminder, which is what the router puts there.
+func (c TalkCase) decision() router.Decision {
+ return router.Decision{
+ Intent: router.Intent(c.Intent),
+ Slots: router.Slots{
+ Key: c.Key,
+ Value: c.Value,
+ Text: c.Utterance,
+ HasKey: c.Key != "",
+ },
+ }
+}
+
func (c TalkCase) turns() []dialogue.Turn {
turns := make([]dialogue.Turn, 0, len(c.History))
for _, h := range c.History {
diff --git a/internal/phraser/eval/talk_test.go b/internal/phraser/eval/talk_test.go
index dcff579..7b993c9 100644
--- a/internal/phraser/eval/talk_test.go
+++ b/internal/phraser/eval/talk_test.go
@@ -11,6 +11,7 @@ import (
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/persona"
"github.com/kami/maven/internal/phraser"
+ "github.com/kami/maven/internal/router"
)
// perPathMinimum — the resolution floor. A per-path score built on a handful of
@@ -36,6 +37,10 @@ func TestTalkFixture(t *testing.T) {
switch c.Path {
case PathChat, PathQuery, PathKnowledge:
+ case PathReply:
+ if c.Intent == "" {
+ t.Errorf("%s: reply case has no intent — the replier is phrased from the decision", c.ID)
+ }
default:
t.Errorf("%s: unknown path %q", c.ID, c.Path)
}
@@ -69,10 +74,15 @@ type fakeTalker struct{ reply string }
func (f fakeTalker) PhraseChat(context.Context, string, []dialogue.Turn) (string, error) {
return f.reply, nil
}
+
func (f fakeTalker) PhraseQuery(context.Context, string, []string) (string, error) {
return f.reply, nil
}
+func (f fakeTalker) PhraseReply(context.Context, router.Decision) (string, error) {
+ return f.reply, nil
+}
+
// TestScoreTalkCounts — a reply that fails on purpose must be counted on every
// path, so a real run cannot report a hidden zero.
func TestScoreTalkCounts(t *testing.T) {
@@ -104,7 +114,7 @@ func TestScoreTalkCounts(t *testing.T) {
}
}
-// TestLLMTalkBaseline — the resident model on the three conversational paths.
+// TestLLMTalkBaseline — the resident model on all four phrasing paths.
// Opt-in exactly like TestLLMPhrasingBaseline: CI has no model and a run costs
// minutes on the CPU target.
//
@@ -148,7 +158,12 @@ func TestLLMTalkBaseline(t *testing.T) {
}
t.Logf("scoring model %s at %s", model, base)
- rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", p, f)
+ // The reply path is a separate object in the daemon too: the phraser owns its
+ // own llama-server, the replier is handed an llm.Client. Pair scores both.
+ block := func() string { return persona.Facts{}.Block(time.Now()) }
+ target := Pair{Talker: p, Confirmer: phraser.NewReplier(llm.New(base, cfg.Timeout), block)}
+
+ rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", target, f)
if err != nil {
t.Fatalf("ScoreTalk: %v", err)
}
diff --git a/internal/phraser/eval/talk_v1.json b/internal/phraser/eval/talk_v1.json
index 64b428b..389cf3f 100644
--- a/internal/phraser/eval/talk_v1.json
+++ b/internal/phraser/eval/talk_v1.json
@@ -222,6 +222,90 @@
"utterance": "почему гром слышно позже молнии?",
"want_any": ["звук", "све", "быстр", "гром", "молни"],
"tags": ["general"]
+ },
+ {
+ "id": "reply-fact-coffee",
+ "path": "reply",
+ "intent": "fact",
+ "key": "кофе",
+ "value": "закончился",
+ "utterance": "кофе закончился",
+ "want_any": ["коф"],
+ "tags": ["fact"],
+ "note": "The plainest confirmation there is, and the sentence he hears most often."
+ },
+ {
+ "id": "reply-fact-weight",
+ "path": "reply",
+ "intent": "fact",
+ "key": "вес",
+ "value": "82",
+ "utterance": "мой вес 82",
+ "want_any": ["вес", "82"],
+ "tags": ["fact", "number"],
+ "note": "A number must survive into the confirmation; a paraphrase that drops it is useless."
+ },
+ {
+ "id": "reply-fact-pill",
+ "path": "reply",
+ "intent": "fact",
+ "key": "таблетки",
+ "value": "выпил",
+ "utterance": "таблетки выпил",
+ "want_any": ["таблетк"],
+ "tags": ["fact", "feminine"],
+ "note": "He says 'выпил', masculine and about himself. She must not copy the form onto herself."
+ },
+ {
+ "id": "reply-note-router",
+ "path": "reply",
+ "intent": "note",
+ "utterance": "роутер перезагружается сам по ночам",
+ "want_any": ["роутер"],
+ "tags": ["note"]
+ },
+ {
+ "id": "reply-note-long",
+ "path": "reply",
+ "intent": "note",
+ "utterance": "если диск снова отвалится, посмотреть кабель, а не контроллер, в прошлый раз был кабель",
+ "want_any": ["диск", "кабел"],
+ "tags": ["note", "length"],
+ "note": "A long note baits a long confirmation. One sentence is the contract."
+ },
+ {
+ "id": "reply-reminder-evening",
+ "path": "reply",
+ "intent": "reminder",
+ "utterance": "напомни вечером полить цветы",
+ "want_any": ["цвет", "полит", "вечер"],
+ "tags": ["reminder"]
+ },
+ {
+ "id": "reply-reminder-tomorrow",
+ "path": "reply",
+ "intent": "reminder",
+ "utterance": "напомни завтра позвонить в поликлинику",
+ "want_any": ["поликлиник", "позвон", "звон"],
+ "tags": ["reminder"]
+ },
+ {
+ "id": "reply-formality-bait",
+ "path": "reply",
+ "intent": "note",
+ "utterance": "запишите пожалуйста что счётчики я сдал",
+ "want_any": ["счётчик", "счетчик"],
+ "tags": ["note", "persona-bait", "address"],
+ "note": "Polite plural in the input. The confirmation must still be на ты."
+ },
+ {
+ "id": "reply-question-bait",
+ "path": "reply",
+ "intent": "note",
+ "utterance": "надо купить фильтр для воды, не помню какой",
+ "want_any": ["фильтр"],
+ "tags": ["note", "no-question"],
+ "note": "An unresolved note invites her to ask which filter. A confirmation does not ask."
}
]
}
diff --git a/internal/phraser/replier.go b/internal/phraser/replier.go
new file mode 100644
index 0000000..b61246a
--- /dev/null
+++ b/internal/phraser/replier.go
@@ -0,0 +1,112 @@
+// phraser/replier.go — reactive reply phrasing, the confirmation he hears
+// after every fact, note and reminder.
+//
+// It lived in cmd/mavend as package main until Vikunja #396, which meant the
+// most frequently heard sentence Maven says was the one path the phrasing eval
+// could not import, let alone score. Nothing here talks to the daemon: the
+// caller supplies the completer and the context block, and cmd/mavend keeps the
+// stub fallback so a model error still answers.
+package phraser
+
+import (
+ "context"
+ "strings"
+ "time"
+
+ "github.com/kami/maven/internal/llm"
+ "github.com/kami/maven/internal/persona"
+ "github.com/kami/maven/internal/router"
+)
+
+// Completer is the model seam for the replier, a subset of router.Completer.
+// *llm.Client satisfies it.
+type Completer interface {
+ Complete(ctx context.Context, r llm.Req) (string, error)
+}
+
+// replyTimeout bounds one reply. Generous because the resident model on the CPU
+// floor is slow and the caller has a deterministic fallback anyway.
+const replyTimeout = 60 * time.Second
+
+// ReplySystemPrompt — the reactive confirmation contract: one short Russian
+// sentence, feminine self-reference, informal address, no question.
+const ReplySystemPrompt = `Ты — Maven, домашняя ассистентка (о себе — в женском роде). Владелец — мужчина, говоришь с ним на "ты", в единственном числе; никогда не "вы"/"ваш" и не "он"/"его". Подтверди действие РОВНО ОДНИМ коротким предложением (≤120 символов), по-русски, спокойно и без официальных формулировок. Не задавай вопросов, не повторяй слова, не добавляй ничего после точки. Отвечай ТОЛЬКО одним объектом JSON с полями "response" (текст) и "mood" (ровно одно из: neutral, happy, thinking, tired, confused).
+Пример: {"response": "Записала, что ты выпил стакан воды.", "mood": "neutral"}
+Никогда не пиши "..." в поле response.`
+
+// Replier phrases reactive confirmations with the resident model. It has no
+// fallback of its own: an error is returned, and the daemon answers from the
+// deterministic stub. That is also what makes it scorable — a dead server shows
+// up as an error rather than as bad phrasing.
+type Replier struct {
+ c Completer
+
+ // block renders the shared context block per turn (who he is, the time).
+ // nil ⇒ the prompt stands alone.
+ block func() string
+}
+
+// NewReplier builds a replier over c. block may be nil.
+func NewReplier(c Completer, block func() string) *Replier {
+ return &Replier{c: c, block: block}
+}
+
+// PhraseReply returns the confirmation for one decision. An empty string with a
+// nil error means the model produced nothing usable, which the caller must
+// treat exactly like an error.
+func (r *Replier) PhraseReply(ctx context.Context, d router.Decision) (string, error) {
+ ctx, cancel := context.WithTimeout(ctx, replyTimeout)
+ defer cancel()
+ out, err := r.c.Complete(ctx, llm.Req{
+ System: persona.Prepend(r.block, ReplySystemPrompt),
+ User: replyContext(d),
+ Grammar: ResponseGrammar,
+ MaxTokens: 512,
+ RepeatPenalty: 1.3,
+ })
+ if err != nil {
+ return "", err
+ }
+ out = stripThink(out)
+ if response, _, perr := parseResponseMood(out); perr != nil {
+ return "", perr
+ } else if response != "" {
+ return response, nil
+ }
+ // fallback: the model answered in bare prose, which is fine here.
+ return firstSentence(out), nil
+}
+
+// firstSentence trims the model's output to a single clean confirmation: first
+// line, first sentence, whitespace-normalized — the last-line defense against a
+// small model that rambles past the first period despite the prompt + stop.
+func firstSentence(s string) string {
+ s = strings.TrimSpace(s)
+ if i := strings.IndexByte(s, '\n'); i >= 0 {
+ s = s[:i]
+ }
+ // keep up to and including the first sentence-ending punctuation.
+ if i := strings.IndexAny(s, ".!?"); i >= 0 {
+ s = s[:i+1]
+ }
+ return strings.TrimSpace(s)
+}
+
+// replyContext renders the decision into a compact RU description for the model.
+func replyContext(d router.Decision) string {
+ switch d.Intent {
+ case router.IntentFact:
+ return "записала факт: " + d.Slots.Key + " " + d.Slots.Value
+ case router.IntentNote:
+ return "сохранила заметку: " + d.Slots.Text
+ case router.IntentReminder:
+ return "поставила напоминание: " + d.Slots.Text
+ default:
+ return string(d.Intent) + ": " + d.Slots.Text
+ }
+}
+
+// StripThink removes the block a Thinking-variant model emits before its
+// answer. Exported for the daemon's own model callers, which parse output that
+// never passes through a phraser method.
+func StripThink(s string) string { return stripThink(s) }
diff --git a/internal/phraser/replier_test.go b/internal/phraser/replier_test.go
new file mode 100644
index 0000000..9e3e026
--- /dev/null
+++ b/internal/phraser/replier_test.go
@@ -0,0 +1,90 @@
+package phraser
+
+import (
+ "context"
+ "testing"
+
+ "github.com/kami/maven/internal/llm"
+ "github.com/kami/maven/internal/router"
+)
+
+type mockCompleter struct {
+ out string
+ err error
+}
+
+func (m mockCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return m.out, m.err }
+
+func TestReplierReturnsLLMReply(t *testing.T) {
+ r := NewReplier(mockCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
+ got, err := r.PhraseReply(context.Background(), noteDecision())
+ if err != nil || got != "записала, кофе закончился" {
+ t.Errorf("got %q, %v, want %q, nil", got, err, "записала, кофе закончился")
+ }
+}
+
+func TestReplierFallsBackToPlainText(t *testing.T) {
+ r := NewReplier(mockCompleter{out: "записала, кофе закончился"}, nil)
+ got, err := r.PhraseReply(context.Background(), noteDecision())
+ if err != nil || got != "записала, кофе закончился" {
+ t.Errorf("got %q, %v, want %q, nil", got, err, "записала, кофе закончился")
+ }
+}
+
+func TestReplierReportsTheModelError(t *testing.T) {
+ r := NewReplier(mockCompleter{err: errTestLLMDown}, nil)
+ got, err := r.PhraseReply(context.Background(), noteDecision())
+ if err == nil {
+ t.Errorf("got %q, nil error — a dead model must be reported, not phrased around", got)
+ }
+}
+
+// A fragment the grammar left half-open is a failed generation. It must come
+// back as an error so the daemon reaches its stub, not as a reply.
+func TestReplierRejectsBrokenJSON(t *testing.T) {
+ r := NewReplier(mockCompleter{out: `{"response":"запис`}, nil)
+ got, err := r.PhraseReply(context.Background(), noteDecision())
+ if err == nil || got != "" {
+ t.Errorf("got %q, %v, want empty and an error", got, err)
+ }
+}
+
+func TestReplierEmptyOutputIsEmpty(t *testing.T) {
+ r := NewReplier(mockCompleter{out: ""}, nil)
+ got, err := r.PhraseReply(context.Background(), noteDecision())
+ if err != nil || got != "" {
+ t.Errorf("got %q, %v, want empty and no error", got, err)
+ }
+}
+
+// grammarRecorder captures the request so the grammar can be asserted on.
+type grammarRecorder struct{ req llm.Req }
+
+func (g *grammarRecorder) Complete(_ context.Context, r llm.Req) (string, error) {
+ g.req = r
+ return `{"response":"записала","mood":"neutral"}`, nil
+}
+
+func TestReplierCarriesTheResponseGrammar(t *testing.T) {
+ rec := &grammarRecorder{}
+ r := NewReplier(rec, nil)
+ if _, err := r.PhraseReply(context.Background(), noteDecision()); err != nil {
+ t.Fatalf("PhraseReply: %v", err)
+ }
+ if rec.req.Grammar != ResponseGrammar {
+ t.Errorf("grammar = %q, want ResponseGrammar", rec.req.Grammar)
+ }
+ if rec.req.System != ReplySystemPrompt {
+ t.Errorf("system prompt = %q, want ReplySystemPrompt", rec.req.System)
+ }
+}
+
+func noteDecision() router.Decision {
+ return router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}}
+}
+
+var errTestLLMDown = errTest("llm down")
+
+type errTest string
+
+func (e errTest) Error() string { return string(e) }