Files
Maven/internal/phraser/grammar_test.go
T
claude 32d5f68710 phrasing and routing: a raw newline is not JSON (V-537)
Sixty of the failures in the 2026-08-05 temperature sweep were one error,
`phraser: model output starts as JSON but does not parse`, all of them in the
reply family and two of them in all twelve runs. The write-up read that as
truncation. It is not: no run hit the token cap.

The string rule in both grammars was `[^"\\]`, which admits a literal
newline. A model that wants two lines writes one, the generation satisfies the
grammar, and json.Unmarshal then rejects it with "invalid character '\n' in
string literal". The object starts with "{", so it came back as errBrokenJSON
and the reply was an empty string. The router's rule also admitted `"\\" .`,
so \q satisfied it and failed to parse the same way.

Both string rules are now llama.cpp's own json.gbnf class: the control range is
out and the escape alternatives are exact. Verified against the resident model
on 8899 — llama-server accepts both grammars and both still emit what they did.

escapeRawControls is the second line, for NoGrammar and for a remote server that
ignores a grammar: a reply whose only fault is a raw newline is readable, so it
is read rather than dropped.
2026-08-05 11:24:35 +04:00

135 lines
4.6 KiB
Go

package phraser
import (
"context"
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"testing"
"github.com/kami/maven/internal/loop"
)
// grammarSpy stands in for llama-server: it records the grammar field of every
// request and always answers with a contract-shaped reply.
type grammarSpy struct {
srv *httptest.Server
grammars []string
}
func newGrammarSpy(t *testing.T) *grammarSpy {
t.Helper()
s := &grammarSpy{}
s.srv = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
var req chatReq
if err := json.NewDecoder(r.Body).Decode(&req); err != nil {
t.Errorf("spy: decode request: %v", err)
}
s.grammars = append(s.grammars, req.Grammar)
w.Header().Set("Content-Type", "application/json")
w.Write([]byte(`{"choices":[{"message":{"content":"{\"response\": \"ага\", \"mood\": \"neutral\"}"}}]}`))
}))
t.Cleanup(s.srv.Close)
return s
}
// callAllPhrasingPaths hits every path that expects the JSON contract.
// LLMNudges must be set on the phraser under test: nudges come from templates
// by default and never reach the model at all.
func callAllPhrasingPaths(t *testing.T, p *LLMPhraser) {
t.Helper()
ctx := context.Background()
if _, err := p.PhraseNudge(ctx, loop.Candidate{Rule: loop.WaterRule(), Severity: loop.Sev1}); err != nil {
t.Fatalf("PhraseNudge: %v", err)
}
if _, err := p.PhraseChat(ctx, "привет", nil); err != nil {
t.Fatalf("PhraseChat: %v", err)
}
// Both branches: no notes (general knowledge) and with notes (grounded).
if _, err := p.PhraseQuery(ctx, "сколько воды я выпил", nil); err != nil {
t.Fatalf("PhraseQuery (no notes): %v", err)
}
if _, err := p.PhraseQuery(ctx, "сколько воды я выпил", []string{"два литра"}); err != nil {
t.Fatalf("PhraseQuery (notes): %v", err)
}
}
func TestGrammarIsAttachedToEveryPhrasingRequest(t *testing.T) {
if strings.TrimSpace(responseGrammar) == "" {
t.Fatal("responseGrammar is empty")
}
spy := newGrammarSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{LLMNudges: true})
callAllPhrasingPaths(t, p)
if len(spy.grammars) != 4 {
t.Fatalf("expected 4 requests, got %d", len(spy.grammars))
}
for i, g := range spy.grammars {
if g != responseGrammar {
t.Errorf("request %d carries grammar %q, want responseGrammar", i, g)
}
}
}
func TestNoGrammarConfigDisablesIt(t *testing.T) {
spy := newGrammarSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{NoGrammar: true, LLMNudges: true})
callAllPhrasingPaths(t, p)
for i, g := range spy.grammars {
if g != "" {
t.Errorf("request %d still carries a grammar with NoGrammar set: %q", i, g)
}
}
}
// The grammar's string rule must accept any codepoint, not just ASCII. Replies
// are Russian: an ASCII-only class would constrain the model into empty replies.
func TestGrammarStringRuleIsNotASCIIOnly(t *testing.T) {
if !strings.Contains(responseGrammar, `[^"\\\x00-\x1F]`) {
t.Error("string rule is not the any-codepoint-except-quote-backslash-and-controls class; Cyrillic replies would be impossible")
}
// The control range must be out (Vikunja #537): a raw newline inside a JSON
// string is not JSON, and the model wrote one whenever it wanted two lines.
if strings.Contains(responseGrammar, `([^"\\] |`) {
t.Error("string rule still admits raw control characters; a multi-line reply will fail to parse")
}
}
// What the grammar describes must survive the parser that reads it back — a
// Russian body with an escaped quote inside, hand-built to test the contract.
func TestGrammarShapedJSONParses(t *testing.T) {
raw := `{"response": "он сказал \"привет\" и ушёл.\nвот так.", "mood": "confused"}`
text, mood, err := parseResponseMood(raw)
if err != nil {
t.Fatalf("grammar-shaped JSON did not parse: %v", err)
}
if want := "он сказал \"привет\" и ушёл.\nвот так."; text != want {
t.Errorf("response = %q, want %q", text, want)
}
if mood != "confused" {
t.Errorf("mood = %q, want confused", mood)
}
}
// Every mood the grammar permits is one the contract knows, and all five are there.
func TestGrammarMoodEnumMatchesTheContract(t *testing.T) {
for _, m := range []string{"neutral", "happy", "thinking", "tired", "confused"} {
if !strings.Contains(responseGrammar, `"\"`+m+`\""`) {
t.Errorf("mood %q missing from the grammar", m)
}
}
// No sixth mood: the enum line lists exactly five alternatives.
for _, line := range strings.Split(responseGrammar, "\n") {
if strings.HasPrefix(line, "mood") {
if n := strings.Count(line, "|") + 1; n != 5 {
t.Errorf("mood rule lists %d alternatives, want 5: %s", n, line)
}
}
}
}