diff --git a/internal/phraser/grammar_test.go b/internal/phraser/grammar_test.go new file mode 100644 index 0000000..fe262a6 --- /dev/null +++ b/internal/phraser/grammar_test.go @@ -0,0 +1,124 @@ +package phraser + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/kami/maven/internal/loop" +) + +// grammarSpy stands in for llama-server: it records the grammar field of every +// request and always answers with a contract-shaped reply. +type grammarSpy struct { + srv *httptest.Server + grammars []string +} + +func newGrammarSpy(t *testing.T) *grammarSpy { + t.Helper() + s := &grammarSpy{} + s.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var req chatReq + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + t.Errorf("spy: decode request: %v", err) + } + s.grammars = append(s.grammars, req.Grammar) + w.Header().Set("Content-Type", "application/json") + w.Write([]byte(`{"choices":[{"message":{"content":"{\"response\": \"ага\", \"mood\": \"neutral\"}"}}]}`)) + })) + t.Cleanup(s.srv.Close) + return s +} + +// callAllPhrasingPaths hits every path that expects the JSON contract. +func callAllPhrasingPaths(t *testing.T, p *LLMPhraser) { + t.Helper() + ctx := context.Background() + if _, err := p.PhraseNudge(ctx, loop.Candidate{Rule: loop.WaterRule(), Severity: loop.Sev1}); err != nil { + t.Fatalf("PhraseNudge: %v", err) + } + if _, err := p.PhraseChat(ctx, "привет", nil); err != nil { + t.Fatalf("PhraseChat: %v", err) + } + // Both branches: no notes (general knowledge) and with notes (grounded). + if _, err := p.PhraseQuery(ctx, "сколько воды я выпил", nil); err != nil { + t.Fatalf("PhraseQuery (no notes): %v", err) + } + if _, err := p.PhraseQuery(ctx, "сколько воды я выпил", []string{"два литра"}); err != nil { + t.Fatalf("PhraseQuery (notes): %v", err) + } +} + +func TestGrammarIsAttachedToEveryPhrasingRequest(t *testing.T) { + if strings.TrimSpace(responseGrammar) == "" { + t.Fatal("responseGrammar is empty") + } + spy := newGrammarSpy(t) + p := NewLLMPhraserAt(spy.srv.URL, Config{}) + + callAllPhrasingPaths(t, p) + + if len(spy.grammars) != 4 { + t.Fatalf("expected 4 requests, got %d", len(spy.grammars)) + } + for i, g := range spy.grammars { + if g != responseGrammar { + t.Errorf("request %d carries grammar %q, want responseGrammar", i, g) + } + } +} + +func TestNoGrammarConfigDisablesIt(t *testing.T) { + spy := newGrammarSpy(t) + p := NewLLMPhraserAt(spy.srv.URL, Config{NoGrammar: true}) + + callAllPhrasingPaths(t, p) + + for i, g := range spy.grammars { + if g != "" { + t.Errorf("request %d still carries a grammar with NoGrammar set: %q", i, g) + } + } +} + +// The grammar's string rule must accept any codepoint, not just ASCII. Replies +// are Russian: an ASCII-only class would constrain the model into empty replies. +func TestGrammarStringRuleIsNotASCIIOnly(t *testing.T) { + if !strings.Contains(responseGrammar, `([^"\\] | "\\" ["\\/bfnrt])`) { + t.Error("string rule is not the any-codepoint-except-quote-and-backslash class; Cyrillic replies would be impossible") + } +} + +// What the grammar describes must survive the parser that reads it back — a +// Russian body with an escaped quote inside, hand-built to test the contract. +func TestGrammarShapedJSONParses(t *testing.T) { + raw := `{"response": "он сказал \"привет\" и ушёл.\nвот так.", "mood": "confused"}` + text, mood := parseResponseMood(raw) + if want := "он сказал \"привет\" и ушёл.\nвот так."; text != want { + t.Errorf("response = %q, want %q", text, want) + } + if mood != "confused" { + t.Errorf("mood = %q, want confused", mood) + } +} + +// Every mood the grammar permits is one the contract knows, and all five are there. +func TestGrammarMoodEnumMatchesTheContract(t *testing.T) { + for _, m := range []string{"neutral", "happy", "thinking", "tired", "confused"} { + if !strings.Contains(responseGrammar, `"\"`+m+`\""`) { + t.Errorf("mood %q missing from the grammar", m) + } + } + // No sixth mood: the enum line lists exactly five alternatives. + for _, line := range strings.Split(responseGrammar, "\n") { + if strings.HasPrefix(line, "mood") { + if n := strings.Count(line, "|") + 1; n != 5 { + t.Errorf("mood rule lists %d alternatives, want 5: %s", n, line) + } + } + } +} diff --git a/internal/phraser/llmphraser.go b/internal/phraser/llmphraser.go index 48ba72a..2421ce9 100644 --- a/internal/phraser/llmphraser.go +++ b/internal/phraser/llmphraser.go @@ -45,6 +45,13 @@ type Config struct { // address him, the time) fresh for each turn. See internal/persona. // nil ⇒ no block, the prompts stand alone. ContextBlock func() string + + // NoGrammar turns the GBNF constraint off (zero value ⇒ grammar ON). + // The escape hatch exists because the target resident model — the + // locally CPT'd Qwen3-1.7B — does not exist yet: if its chat template + // ever fights the grammar, the fix should be a config flip on the + // deploy box, not a code change and a rebuild. + NoGrammar bool } func DefaultConfig(modelPath string) Config { @@ -290,6 +297,7 @@ func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTo Messages: msgs, Temperature: 0.7, MaxTokens: maxTokens, + Grammar: p.grammar(), } body, err := json.Marshal(req) if err != nil { @@ -369,6 +377,37 @@ type chatReq struct { Messages []chatMsg `json:"messages"` Temperature float64 `json:"temperature"` MaxTokens int `json:"max_tokens"` + // Grammar is llama-server's `grammar` field (GBNF). Same wiring as + // internal/llm.Req.Grammar. Empty ⇒ unconstrained sampling. + Grammar string `json:"grammar,omitempty"` +} + +// responseGrammar — GBNF constraining the model to the documented phrasing +// contract and nothing else: {"response": "", "mood": ""}. +// +// Without it a 0.8B answers roughly one chat turn in three with open reasoning +// as plain text ("Thinking Process:" …), which no tag-stripper can remove and +// which eats the token budget before the JSON closes. Modelled on +// routeGrammar in internal/router/llmrouter.go so the two read alike. +// +// text accepts ANY codepoint except the two JSON must escape — the replies are +// Russian, so an ASCII-only rule would make every reply empty. The escape rule +// is what lets the model close a string it opened with a quote inside. Length +// is bounded so a repetition loop truncates the field, not the JSON object. +const responseGrammar = ` +root ::= "{" ws "\"response\"" ws ":" ws string ws "," ws "\"mood\"" ws ":" ws mood ws "}" +mood ::= "\"neutral\"" | "\"happy\"" | "\"thinking\"" | "\"tired\"" | "\"confused\"" +string ::= "\"" ([^"\\] | "\\" ["\\/bfnrt]){0,400} "\"" +ws ::= [ \t\n]* +` + +// grammar returns the GBNF to attach to a phrasing request, or "" when the +// operator turned it off. +func (p *LLMPhraser) grammar() string { + if p.cfg.NoGrammar { + return "" + } + return responseGrammar } type chatResp struct { @@ -393,6 +432,7 @@ func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, ma }, Temperature: 0.7, MaxTokens: maxTokens, + Grammar: p.grammar(), } body, err := json.Marshal(req) if err != nil {