Constrain the phrasing output with a GBNF grammar

The 0.8B answered about one chat turn in three with open reasoning as plain text, so no JSON ever closed and the fallback shipped "Thinking Process:" to the user. A grammar makes that output impossible.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
This commit is contained in:
kami
2026-07-31 17:18:06 +04:00
parent 0110e9bc8c
commit 1890ff5d5d
2 changed files with 164 additions and 0 deletions
+40
View File
@@ -45,6 +45,13 @@ type Config struct {
// address him, the time) fresh for each turn. See internal/persona.
// nil ⇒ no block, the prompts stand alone.
ContextBlock func() string
// NoGrammar turns the GBNF constraint off (zero value ⇒ grammar ON).
// The escape hatch exists because the target resident model — the
// locally CPT'd Qwen3-1.7B — does not exist yet: if its chat template
// ever fights the grammar, the fix should be a config flip on the
// deploy box, not a code change and a rebuild.
NoGrammar bool
}
func DefaultConfig(modelPath string) Config {
@@ -290,6 +297,7 @@ func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTo
Messages: msgs,
Temperature: 0.7,
MaxTokens: maxTokens,
Grammar: p.grammar(),
}
body, err := json.Marshal(req)
if err != nil {
@@ -369,6 +377,37 @@ type chatReq struct {
Messages []chatMsg `json:"messages"`
Temperature float64 `json:"temperature"`
MaxTokens int `json:"max_tokens"`
// Grammar is llama-server's `grammar` field (GBNF). Same wiring as
// internal/llm.Req.Grammar. Empty ⇒ unconstrained sampling.
Grammar string `json:"grammar,omitempty"`
}
// responseGrammar — GBNF constraining the model to the documented phrasing
// contract and nothing else: {"response": "<text>", "mood": "<enum>"}.
//
// Without it a 0.8B answers roughly one chat turn in three with open reasoning
// as plain text ("Thinking Process:" …), which no tag-stripper can remove and
// which eats the token budget before the JSON closes. Modelled on
// routeGrammar in internal/router/llmrouter.go so the two read alike.
//
// text accepts ANY codepoint except the two JSON must escape — the replies are
// Russian, so an ASCII-only rule would make every reply empty. The escape rule
// is what lets the model close a string it opened with a quote inside. Length
// is bounded so a repetition loop truncates the field, not the JSON object.
const responseGrammar = `
root ::= "{" ws "\"response\"" ws ":" ws string ws "," ws "\"mood\"" ws ":" ws mood ws "}"
mood ::= "\"neutral\"" | "\"happy\"" | "\"thinking\"" | "\"tired\"" | "\"confused\""
string ::= "\"" ([^"\\] | "\\" ["\\/bfnrt]){0,400} "\""
ws ::= [ \t\n]*
`
// grammar returns the GBNF to attach to a phrasing request, or "" when the
// operator turned it off.
func (p *LLMPhraser) grammar() string {
if p.cfg.NoGrammar {
return ""
}
return responseGrammar
}
type chatResp struct {
@@ -393,6 +432,7 @@ func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, ma
},
Temperature: 0.7,
MaxTokens: maxTokens,
Grammar: p.grammar(),
}
body, err := json.Marshal(req)
if err != nil {