grammar: the last two model calls that were still free text

The replier and the meeting summariser were the two call sites without a
GBNF. Both are exactly the shape that makes a Thinking variant answer with
its reasoning as prose, and neither had anything downstream that could
remove it.

The replier already parses {"response","mood"}, so it now sends the phraser's
grammar for that contract, exported once as phraser.ResponseGrammar so the
two definitions cannot drift.

The summariser stays text-in/text-out. The JSON wrapper is attached and
unwrapped in the daemon's Completer, so internal/capture is unchanged and a
Completer without a grammar still works.

The simulator told routing from phrasing by "has a grammar", which stopped
being true here; it now looks for the intent enum.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
kami
2026-08-02 02:28:53 +04:00
parent 14e98334ad
commit 612ca8cf1b
7 changed files with 118 additions and 4 deletions
+52 -1
View File
@@ -31,9 +31,11 @@ package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"log"
"strings"
"sync"
"time"
@@ -54,16 +56,65 @@ import (
// asks Maven to stop recording gets the transcript back in seconds.
const captureSummaryTimeout = 20 * time.Minute
// summaryGrammar — GBNF pinning a summarisation call to one JSON object holding
// the summary and nothing else. Same reasoning as responseGrammar and memeval's
// evalGrammar: the resident model is a Thinking variant, and a summarisation
// prompt is exactly the shape that invites it to answer with its reasoning as
// plain text. Demanding JSON leaves the reasoning nowhere to go.
//
// The bound is 2000 characters, twice the phraser's, because a reduce step over
// a two-hour meeting is a paragraph and not a sentence. Newlines are escaped by
// the escape rule, so the bullet list the prompt asks for survives the wrapper.
const summaryGrammar = `
root ::= "{" ws "\"summary\"" ws ":" ws string ws "}"
string ::= "\"" ([^"\\] | "\\" ["\\/bfnrt]){0,2000} "\""
ws ::= [ \t\n]*
`
// llmCompleter adapts *llm.Client to capture.Completer. The pure package names
// the two strings it needs and stays free of the llm request struct; the client
// itself is the swap-aware one from llmClientFor, so a model swap re-points it.
//
// The JSON wrapper lives here, not in internal/capture: that package is
// text-in/text-out by design, and the map/reduce steps still see plain prose.
type llmCompleter struct {
c *llm.Client
maxTokens int
}
func (l llmCompleter) Complete(ctx context.Context, system, user string) (string, error) {
return l.c.Complete(ctx, llm.Req{System: system, User: user, MaxTokens: l.maxTokens})
out, err := l.c.Complete(ctx, llm.Req{
System: system,
User: user,
Grammar: summaryGrammar,
MaxTokens: l.maxTokens,
})
if err != nil {
return "", err
}
return unwrapSummary(out), nil
}
// unwrapSummary takes the summary out of the JSON object the grammar produced.
// Anything that does not parse is returned as-is: an operator running without a
// grammar, or a llama-server too old to honour one, gets the plain text it used
// to get rather than an empty meeting summary.
func unwrapSummary(raw string) string {
s := stripThink(strings.TrimSpace(raw))
start := strings.Index(s, "{")
end := strings.LastIndex(s, "}")
if start < 0 || end <= start {
return s
}
var parsed struct {
Summary string `json:"summary"`
}
if err := json.Unmarshal([]byte(s[start:end+1]), &parsed); err != nil {
return s
}
// An empty field is the model saying nothing, so hand back nothing. Returning
// the raw object here would write `{"summary":""}` into his notes.
return strings.TrimSpace(parsed.Summary)
}
// captureWiring — the recorder plus what it needs to write the result down.
+26
View File
@@ -116,3 +116,29 @@ func TestStopReturnsTranscriptAndNotesItWithoutASummary(t *testing.T) {
t.Fatalf("the meeting left no note behind: %+v", notes)
}
}
// The summary path is JSON-wrapped by summaryGrammar, and internal/capture must
// keep seeing plain prose. These cover the wrapper and every way it can be
// absent or broken, because a meeting summary is written once and not retried.
func TestUnwrapSummary(t *testing.T) {
cases := []struct {
name string
in string
want string
}{
{"grammar output", `{"summary": "решили купить насос"}`, "решили купить насос"},
{"multiline field", `{"summary": "- насос\n- бюджет"}`, "- насос\n- бюджет"},
{"empty marker survives", `{"summary": "пусто"}`, "пусто"},
{"empty field says nothing", `{"summary": ""}`, ""},
{"thinking prefix", "<think>hm</think>\n{\"summary\": \"итог\"}", "итог"},
{"no grammar, plain prose", "решили купить насос", "решили купить насос"},
{"broken json falls back", `{"summary": "обрыв`, `{"summary": "обрыв`},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
if got := unwrapSummary(c.in); got != c.want {
t.Errorf("unwrapSummary(%q) = %q, want %q", c.in, got, c.want)
}
})
}
}
+2
View File
@@ -8,6 +8,7 @@ import (
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/persona"
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/voice"
)
@@ -47,6 +48,7 @@ func (r *llmReplier) Reply(d router.Decision) string {
out, err := r.c.Complete(ctx, llm.Req{
System: persona.Prepend(r.block, replySystem),
User: replyContext(d),
Grammar: phraser.ResponseGrammar,
MaxTokens: 512,
RepeatPenalty: 1.3,
})
+18
View File
@@ -5,6 +5,7 @@ import (
"testing"
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/voice"
)
@@ -67,3 +68,20 @@ var errTestLLMDown = errTest("llm down")
type errTest string
func (e errTest) Error() string { return string(e) }
// grammarRecorder captures the request so the grammar can be asserted on.
type grammarRecorder struct{ req llm.Req }
func (g *grammarRecorder) Complete(_ context.Context, r llm.Req) (string, error) {
g.req = r
return `{"response":"записала","mood":"neutral"}`, nil
}
func TestLLMReplierCarriesTheResponseGrammar(t *testing.T) {
rec := &grammarRecorder{}
r := newLLMReplier(rec, nil)
r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
if rec.req.Grammar != phraser.ResponseGrammar {
t.Errorf("grammar = %q, want phraser.ResponseGrammar", rec.req.Grammar)
}
}
+4 -1
View File
@@ -328,7 +328,10 @@ func (s *scriptedLLM) Complete(_ context.Context, r llm.Req) (string, error) {
s.mu.Lock()
defer s.mu.Unlock()
s.calls = append(s.calls, r)
routing := r.Grammar != ""
// A grammar no longer separates the two contracts — the replier carries one
// too since phraser.ResponseGrammar was attached to it. Only the router's
// grammar names the intent enum, so that is what tells them apart.
routing := strings.Contains(r.Grammar, "intent")
for _, e := range s.entries {
if e.Match != "" && !strings.Contains(strings.ToLower(r.User), strings.ToLower(e.Match)) {
continue