grammar: the last two model calls that were still free text
The replier and the meeting summariser were the two call sites without a
GBNF. Both are exactly the shape that makes a Thinking variant answer with
its reasoning as prose, and neither had anything downstream that could
remove it.
The replier already parses {"response","mood"}, so it now sends the phraser's
grammar for that contract, exported once as phraser.ResponseGrammar so the
two definitions cannot drift.
The summariser stays text-in/text-out. The JSON wrapper is attached and
unwrapped in the daemon's Completer, so internal/capture is unchanged and a
Completer without a grammar still works.
The simulator told routing from phrasing by "has a grammar", which stopped
being true here; it now looks for the intent enum.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
+52
-1
@@ -31,9 +31,11 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -54,16 +56,65 @@ import (
|
||||
// asks Maven to stop recording gets the transcript back in seconds.
|
||||
const captureSummaryTimeout = 20 * time.Minute
|
||||
|
||||
// summaryGrammar — GBNF pinning a summarisation call to one JSON object holding
|
||||
// the summary and nothing else. Same reasoning as responseGrammar and memeval's
|
||||
// evalGrammar: the resident model is a Thinking variant, and a summarisation
|
||||
// prompt is exactly the shape that invites it to answer with its reasoning as
|
||||
// plain text. Demanding JSON leaves the reasoning nowhere to go.
|
||||
//
|
||||
// The bound is 2000 characters, twice the phraser's, because a reduce step over
|
||||
// a two-hour meeting is a paragraph and not a sentence. Newlines are escaped by
|
||||
// the escape rule, so the bullet list the prompt asks for survives the wrapper.
|
||||
const summaryGrammar = `
|
||||
root ::= "{" ws "\"summary\"" ws ":" ws string ws "}"
|
||||
string ::= "\"" ([^"\\] | "\\" ["\\/bfnrt]){0,2000} "\""
|
||||
ws ::= [ \t\n]*
|
||||
`
|
||||
|
||||
// llmCompleter adapts *llm.Client to capture.Completer. The pure package names
|
||||
// the two strings it needs and stays free of the llm request struct; the client
|
||||
// itself is the swap-aware one from llmClientFor, so a model swap re-points it.
|
||||
//
|
||||
// The JSON wrapper lives here, not in internal/capture: that package is
|
||||
// text-in/text-out by design, and the map/reduce steps still see plain prose.
|
||||
type llmCompleter struct {
|
||||
c *llm.Client
|
||||
maxTokens int
|
||||
}
|
||||
|
||||
func (l llmCompleter) Complete(ctx context.Context, system, user string) (string, error) {
|
||||
return l.c.Complete(ctx, llm.Req{System: system, User: user, MaxTokens: l.maxTokens})
|
||||
out, err := l.c.Complete(ctx, llm.Req{
|
||||
System: system,
|
||||
User: user,
|
||||
Grammar: summaryGrammar,
|
||||
MaxTokens: l.maxTokens,
|
||||
})
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return unwrapSummary(out), nil
|
||||
}
|
||||
|
||||
// unwrapSummary takes the summary out of the JSON object the grammar produced.
|
||||
// Anything that does not parse is returned as-is: an operator running without a
|
||||
// grammar, or a llama-server too old to honour one, gets the plain text it used
|
||||
// to get rather than an empty meeting summary.
|
||||
func unwrapSummary(raw string) string {
|
||||
s := stripThink(strings.TrimSpace(raw))
|
||||
start := strings.Index(s, "{")
|
||||
end := strings.LastIndex(s, "}")
|
||||
if start < 0 || end <= start {
|
||||
return s
|
||||
}
|
||||
var parsed struct {
|
||||
Summary string `json:"summary"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(s[start:end+1]), &parsed); err != nil {
|
||||
return s
|
||||
}
|
||||
// An empty field is the model saying nothing, so hand back nothing. Returning
|
||||
// the raw object here would write `{"summary":""}` into his notes.
|
||||
return strings.TrimSpace(parsed.Summary)
|
||||
}
|
||||
|
||||
// captureWiring — the recorder plus what it needs to write the result down.
|
||||
|
||||
@@ -116,3 +116,29 @@ func TestStopReturnsTranscriptAndNotesItWithoutASummary(t *testing.T) {
|
||||
t.Fatalf("the meeting left no note behind: %+v", notes)
|
||||
}
|
||||
}
|
||||
|
||||
// The summary path is JSON-wrapped by summaryGrammar, and internal/capture must
|
||||
// keep seeing plain prose. These cover the wrapper and every way it can be
|
||||
// absent or broken, because a meeting summary is written once and not retried.
|
||||
func TestUnwrapSummary(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in string
|
||||
want string
|
||||
}{
|
||||
{"grammar output", `{"summary": "решили купить насос"}`, "решили купить насос"},
|
||||
{"multiline field", `{"summary": "- насос\n- бюджет"}`, "- насос\n- бюджет"},
|
||||
{"empty marker survives", `{"summary": "пусто"}`, "пусто"},
|
||||
{"empty field says nothing", `{"summary": ""}`, ""},
|
||||
{"thinking prefix", "<think>hm</think>\n{\"summary\": \"итог\"}", "итог"},
|
||||
{"no grammar, plain prose", "решили купить насос", "решили купить насос"},
|
||||
{"broken json falls back", `{"summary": "обрыв`, `{"summary": "обрыв`},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
if got := unwrapSummary(c.in); got != c.want {
|
||||
t.Errorf("unwrapSummary(%q) = %q, want %q", c.in, got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/persona"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
@@ -47,6 +48,7 @@ func (r *llmReplier) Reply(d router.Decision) string {
|
||||
out, err := r.c.Complete(ctx, llm.Req{
|
||||
System: persona.Prepend(r.block, replySystem),
|
||||
User: replyContext(d),
|
||||
Grammar: phraser.ResponseGrammar,
|
||||
MaxTokens: 512,
|
||||
RepeatPenalty: 1.3,
|
||||
})
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
@@ -67,3 +68,20 @@ var errTestLLMDown = errTest("llm down")
|
||||
type errTest string
|
||||
|
||||
func (e errTest) Error() string { return string(e) }
|
||||
|
||||
// grammarRecorder captures the request so the grammar can be asserted on.
|
||||
type grammarRecorder struct{ req llm.Req }
|
||||
|
||||
func (g *grammarRecorder) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||
g.req = r
|
||||
return `{"response":"записала","mood":"neutral"}`, nil
|
||||
}
|
||||
|
||||
func TestLLMReplierCarriesTheResponseGrammar(t *testing.T) {
|
||||
rec := &grammarRecorder{}
|
||||
r := newLLMReplier(rec, nil)
|
||||
r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
||||
if rec.req.Grammar != phraser.ResponseGrammar {
|
||||
t.Errorf("grammar = %q, want phraser.ResponseGrammar", rec.req.Grammar)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -328,7 +328,10 @@ func (s *scriptedLLM) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.calls = append(s.calls, r)
|
||||
routing := r.Grammar != ""
|
||||
// A grammar no longer separates the two contracts — the replier carries one
|
||||
// too since phraser.ResponseGrammar was attached to it. Only the router's
|
||||
// grammar names the intent enum, so that is what tells them apart.
|
||||
routing := strings.Contains(r.Grammar, "intent")
|
||||
for _, e := range s.entries {
|
||||
if e.Match != "" && !strings.Contains(strings.ToLower(r.User), strings.ToLower(e.Match)) {
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user