package phraser import ( "bufio" "bytes" "context" "encoding/json" "errors" "fmt" "io" "log" "net/http" "os" "os/exec" "regexp" "strings" "sync" "syscall" "time" "github.com/kami/maven/internal/delivery" "github.com/kami/maven/internal/dialogue" "github.com/kami/maven/internal/loop" "github.com/kami/maven/internal/persona" "github.com/kami/maven/internal/router" ) var listenRE = regexp.MustCompile(`listening on (https?://\S+)`) // errEmptyResponse — the server answered and said nothing. Separate from a // transport failure: the model is up and produced no tokens, which is still not // an answer and must not score as one. var errEmptyResponse = errors.New("phraser: empty response from the model") type LLMPhraser struct { cfg Config client *http.Client // tmpl — the hand-written Russian nudges. Default path for nudges; see // Config.LLMNudges. nil only if the template file failed to load. tmpl *NudgeTemplates // spawnCtx — the parent of every llama-server this phraser starts, i.e. the // daemon's own context. Deliberately NOT the per-request context of the call // that asked for a model swap: that one is cancelled the moment the request // returns, which would kill the model it had just loaded. spawnCtx context.Context cancel context.CancelFunc // launch / probe — the two side effects of a swap, injectable so the swap // logic is testable without a real llama-server and a real model file. // launch is nil when this phraser does not own its server (NewLLMPhraserAt), // which is also what makes Swap refuse there. launch func(ctx context.Context, cfg Config) (backend, error) probe func(ctx context.Context, base string) (string, error) // remote — the workstation model, when one is configured. Set once at wiring // time by UseRemote and read on every phrasing call. nil ⇒ every call goes to // the resident llama-server this phraser owns, which is the whole deploy // before a `workstation` block exists. See world.go. remote Remote // swapMu — single-flight around Swap. Held for the whole swap, including the // model load, so two concurrent swap requests can never both be loading. swapMu sync.Mutex // mu guards everything below: the live backend, the swap gate and the // in-flight request count. See acquire/quiesce in swap.go. mu sync.Mutex be backend live liveModel swapping bool inflight int observers []func(baseURL string) } // liveModel — what is actually loaded right now. Distinct from Config, which // stays immutable after construction: a swap changes these three fields and // nothing else, so no reader of cfg (prompts, grammar, timeouts) races a swap. type liveModel struct { ModelPath string NGpuLayers int NCtx int } type Config struct { ModelPath string BinPath string Listen string NGpuLayers int NCtx int Timeout time.Duration // StartupTimeout bounds the wait for llama-server to print the address it // listens on. A config field and not a constant because the box may // legitimately need longer: a cold 1.7B loading off a spinning disk can // outrun a minute, and until this existed that returned "server did not // start within 60s" with no way to raise it. // // 0 ⇒ defaultStartupTimeout. StartupTimeout time.Duration // CacheRAMMiB bounds llama-server's prompt cache, which is what actually ate // this box. Measured on homesrv 2026-08-03: the server's own default limit is // 8192 MiB, it stores the full KV state of every idle slot it evicts (112 kiB // per token, so 166 MiB for one 1521-token prompt), and RSS climbed by that // much per distinct prompt until it hit 7.9 GB and half a gigabyte went to // swap. Weights are only 1.1 GB and mmapped, and -ngl 99 costs almost no RSS // because RADV keeps device memory outside the process. // // 0 ⇒ the flag is not passed and the server's own 8 GiB default applies. That // is the escape hatch for a llama-server too old to know --cache-ram, not a // recommendation. See docs/evals/2026-08-03-llama-prompt-cache.md. CacheRAMMiB int // ContextBlock renders the shared context block (who he is, how to // address him, the time) fresh for each turn. See internal/persona. // nil ⇒ no block, the prompts stand alone. ContextBlock func() string // LLMNudges puts the model back in charge of nudge wording. // // Off by default, and that is a deliberate deprecation of LLM-phrased // nudges: hand-written templates (nudges_ru_v1.json) word every nudge now. // A nudge has nothing to be creative about, and measured over many runs the // 0.8B broke the persona (formal "вы", plural imperatives, masculine // self-reference) and invented facts and units. Templates score 15/15 on the // nudge fixture, the model 11-13/15. // // The LLM path is kept, not deleted: flip this on to get it back. Chat, // query and reminder phrasing are untouched and still go through the model. LLMNudges bool // Temperature — what every phrasing call samples at. 0 ⇒ 0.7, which is // what this transport has always sent. // // A field rather than a constant so the talk fixture can sweep it // (Vikunja #402). Sampling is a dial, and a dial nobody can turn from // outside the package cannot be measured, only argued about. Temperature float64 // NoGrammar turns the GBNF constraint off (zero value ⇒ grammar ON). // The escape hatch exists because the target resident model — the // locally CPT'd Qwen3-1.7B — does not exist yet: if its chat template // ever fights the grammar, the fix should be a config flip on the // deploy box, not a code change and a rebuild. NoGrammar bool } func DefaultConfig(modelPath string) Config { return Config{ ModelPath: modelPath, BinPath: "llama-server", Listen: "127.0.0.1:0", NGpuLayers: -1, NCtx: 2048, // 512 MiB caps total RSS near 1 GB and still holds several recent prompts. CacheRAMMiB: 512, Timeout: 30 * time.Second, StartupTimeout: defaultStartupTimeout, } } func NewLLMPhraser(ctx context.Context, cfg Config) (*LLMPhraser, error) { ctx, cancel := context.WithCancel(ctx) p := &LLMPhraser{ cfg: cfg, client: &http.Client{Timeout: cfg.Timeout}, tmpl: loadNudgeTemplates(), spawnCtx: ctx, cancel: cancel, launch: spawnLlamaServer, probe: defaultProbe, live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx}, } be, err := p.launch(ctx, cfg) if err != nil { cancel() return nil, err } p.be = be return p, nil } // NewLLMPhraserAt wires a phraser to a llama-server that someone else started // and owns. It spawns nothing, so Close does not kill anything. // // This exists for the phrasing scorer (internal/phraser/eval), which must // measure the phrasing against a shared llama-server without taking the model // load hit per run or killing a server another process depends on. The daemon // still uses NewLLMPhraser and still owns its own child process. func NewLLMPhraserAt(baseURL string, cfg Config) *LLMPhraser { return &LLMPhraser{ cfg: cfg, client: &http.Client{Timeout: cfg.Timeout}, tmpl: loadNudgeTemplates(), spawnCtx: context.Background(), cancel: func() {}, probe: defaultProbe, // launch stays nil: we did not start this server, so we must not stop it. // Swap therefore refuses here (ErrSwapNotOwned) instead of killing a // server another process depends on. be: borrowedBackend(strings.TrimSuffix(baseURL, "/")), live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx}, } } // loadNudgeTemplates loads the Russian nudge templates. A broken template file // must not stop the daemon booting, so a failure logs and leaves the LLM path // in charge of nudges. func loadNudgeTemplates() *NudgeTemplates { nt, err := NewNudgeTemplates(nil) if err != nil { log.Printf("phraser: nudge templates unavailable, using the model: %v", err) return nil } return nt } // backend — one llama-server this phraser talks to. Two implementations: a // llamaProc we spawned and must reap, and a borrowedBackend someone else owns. type backend interface { BaseURL() string Close() error } // borrowedBackend — a server started and owned by someone else (the phrasing // scorer's shared llama-server). Closing it is a no-op by construction. type borrowedBackend string func (b borrowedBackend) BaseURL() string { return string(b) } func (b borrowedBackend) Close() error { return nil } // llamaProc — a llama-server child process plus the goroutine reading its // stderr. Close kills and reaps it; see the Pdeathsig note in spawnLlamaServer. type llamaProc struct { base string cmd *exec.Cmd cancel context.CancelFunc wg sync.WaitGroup } func (l *llamaProc) BaseURL() string { return l.base } func (l *llamaProc) Close() error { l.cancel() if l.cmd != nil && l.cmd.Process != nil { _ = l.cmd.Process.Kill() _ = l.cmd.Wait() // reap the process — without Wait, the child becomes a zombie } l.wg.Wait() return nil } // spawnLlamaServer starts one llama-server for cfg and waits until it says which // address it is listening on. ctx owns the process lifetime, so it must be the // daemon's context, not a request's. func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) { ctx, cancel := context.WithCancel(ctx) p, err := startLlamaProc(ctx, cfg) if err != nil { cancel() return nil, err } p.cancel = cancel return p, nil } // llamaArgs is the command line for one resident server. It is a function and // not an inline literal because kill-maven.sh's orphan sweep matches against // this exact line, and a test pins the two together. func llamaArgs(cfg Config) []string { args := []string{ "-m", cfg.ModelPath, "--host", "127.0.0.1", "--port", extractPort(cfg.Listen), "-c", fmt.Sprintf("%d", cfg.NCtx), "-ngl", fmt.Sprintf("%d", cfg.NGpuLayers), "--no-webui", } if cfg.CacheRAMMiB > 0 { args = append(args, "--cache-ram", fmt.Sprintf("%d", cfg.CacheRAMMiB)) } return args } // defaultStartupTimeout — the wait for llama-server's listen line when Config // does not set one. A cold model load off disk is the slow part. const defaultStartupTimeout = 60 * time.Second func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) { startupTimeout := cfg.StartupTimeout if startupTimeout <= 0 { startupTimeout = defaultStartupTimeout } p := &llamaProc{} cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...) // Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by // ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without // it a hard-killed mavend orphans its llama-server (reparented to init, keeps // eating GPU/RAM); repeated dev restarts pile up orphans until the box OOMs. // Setpgid isolates it in its own process group so a stray Ctrl-C on the // terminal group doesn't half-kill it out from under us. (Linux-only, like // the rest of the daemon.) cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true, Pdeathsig: syscall.SIGKILL} p.cmd = cmd // One pipe for both streams. llama.cpp writes its buffer sizes, KV-cache // layout and offload lines to stderr and its request log to stdout, and // stdout used to go nowhere at all — so nothing about the model's memory was // diagnosable from a running box. Both ends land in mavend's log now. pr, pw, err := os.Pipe() if err != nil { return nil, fmt.Errorf("llm: output pipe: %w", err) } cmd.Stdout = pw cmd.Stderr = pw if err := cmd.Start(); err != nil { pr.Close() pw.Close() return nil, fmt.Errorf("llm: start: %w", err) } // The child holds the only other reference to the write end. Dropping ours // is what makes the reader see EOF when the child dies. pw.Close() portCh := make(chan string, 1) errCh := make(chan error, 1) tail := &lineTail{} p.wg.Add(1) go func() { defer p.wg.Done() defer pr.Close() sc := bufio.NewScanner(pr) // llama.cpp prints one prompt per line and a prompt can be long. sc.Buffer(make([]byte, 0, 64*1024), 1024*1024) listening := false for sc.Scan() { line := sc.Bytes() log.Printf("llama: %s", line) if !listening { tail.add(string(line)) if m := listenRE.FindSubmatch(line); len(m) > 1 { listening = true portCh <- string(m[1]) close(portCh) } } } err := sc.Err() if err == nil { err = io.EOF } errCh <- err }() fail := func(err error) (*llamaProc, error) { _ = cmd.Process.Kill() _ = cmd.Wait() return nil, err } select { case addr := <-portCh: p.base = addr return p, nil case err := <-errCh: // The tail is the whole diagnosis when the server dies during load: bare // "EOF" never said which layer or which allocation it choked on. return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String())) case <-ctx.Done(): return fail(ctx.Err()) case <-time.After(startupTimeout): return fail(fmt.Errorf("llm: server did not start within %s; last output: %s", startupTimeout, tail.String())) } } // lineTail keeps the last few startup lines so a server that dies before it // listens can say why in the error, not just "EOF". Written by the reader // goroutine and read by whoever gives up on startup, so it takes a lock. type lineTail struct { mu sync.Mutex lines []string } const lineTailMax = 12 func (t *lineTail) add(line string) { t.mu.Lock() defer t.mu.Unlock() t.lines = append(t.lines, line) if len(t.lines) > lineTailMax { t.lines = t.lines[len(t.lines)-lineTailMax:] } } func (t *lineTail) String() string { t.mu.Lock() defer t.mu.Unlock() if len(t.lines) == 0 { return "(no output)" } return strings.Join(t.lines, " | ") } // BaseURL is the llama-server this phraser talks to right now. It changes when // the model is swapped, so callers that cache it must register an observer // (OnSwap) rather than keeping the string forever. func (p *LLMPhraser) BaseURL() string { p.mu.Lock() defer p.mu.Unlock() if p.be == nil { return "" } return p.be.BaseURL() } func (p *LLMPhraser) Close() error { p.cancel() p.mu.Lock() be := p.be p.be = nil p.mu.Unlock() if be != nil { return be.Close() } return nil } func (p *LLMPhraser) PhraseNudge(ctx context.Context, c loop.Candidate) (delivery.PhrasedNudge, error) { // Templates first — see Config.LLMNudges for why this is the default. if !p.cfg.LLMNudges && p.tmpl != nil { return p.tmpl.PhraseNudge(ctx, c) } prompt := buildNudgePrompt(c) resp, err := p.chat(ctx, prompt) if err != nil { return delivery.PhrasedNudge{}, err } body, mood, perr := parseResponseMood(resp) if perr != nil { // Truncated JSON. Not a nudge — use the plain Russian fallback. log.Printf("phraser: PhraseNudge: %v", perr) body, mood = "", "" } if body == "" { // fallback: try old body/summary format body, _ = parsePhrase(resp) } if body == "" { // The model said nothing usable. Say it in Russian anyway — this text // goes straight to a Russian piper voice, so the old "water — care" // fallback was unspeakable. body = fallbackNudge(c) } if mood == "" { mood = "neutral" } return delivery.PhrasedNudge{Candidate: c, Body: body, Summary: body, Mood: mood}, nil } // PhraseQuery prompts the LLM with the user's utterance and matching notes to // compose a natural answer. On any LLM error it returns the fallback text — // "вот что я нашла: ", or "не знаю." with no notes — and the error // together. The daemon uses the text and keeps the turn alive; a caller that is // measuring counts the failure. Until Vikunja #397 the error was dropped, so a // dead server scored as bad phrasing. func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error) { // Blank sources are no sources. A caller that hands over one empty string — // a page that fetched to nothing, a snippet trimmed away — used to take the // evidence branch and be told to answer from an empty list, which is the one // prompt guaranteed to make a small model fill the gap from memory. notes = nonEmpty(notes) if len(notes) == 0 { sys, prompt := p.knowledgePrompt(utterance) resp, err := p.chatWithSystem(ctx, sys, prompt, 768) if err != nil { return UnknownFallback(), fmt.Errorf("phrase query (knowledge): %w", err) } if resp == "" { return UnknownFallback(), errEmptyResponse } text, _, perr := parseResponseMood(resp) if perr != nil { return UnknownFallback(), fmt.Errorf("phrase query (knowledge): %w", perr) } if text != "" { return text, nil } return resp, nil } sys, prompt := p.evidencePrompt(utterance, notes) resp, err := p.chatWithSystem(ctx, sys, prompt, 768) text, _, perr := parseResponseMood(resp) if err != nil || perr != nil { // Read the notes out rather than ship a broken fragment. cause := err if cause == nil { cause = perr } return SourcesFallback(strings.Join(notes, "; ")), fmt.Errorf("phrase query (evidence): %w", cause) } if text != "" { return text, nil } return resp, nil } // PhraseChat uses the LLM to respond conversationally, building a multi-turn // message array from dialogue history + the current user utterance. On any LLM // error it returns both ChatFallback and the error, on the same rule as // PhraseQuery: the fallback keeps the turn alive, the error stays visible. func (p *LLMPhraser) PhraseChat(ctx context.Context, utterance string, history []dialogue.Turn) (string, error) { sys := chatSystemPrompt(p.cfg.ContextBlock) msgs := []chatMsg{ {Role: "system", Content: sys}, } // Combine history and current utterance into one user message. // Some model chat templates (Ministral, etc.) reject consecutive user turns. var combined string for _, t := range history { combined += t.Text + "\n" } combined += utterance msgs = append(msgs, chatMsg{Role: "user", Content: strings.TrimSpace(combined)}) resp, err := p.chatWithMessages(ctx, msgs, 768) if err != nil { return ChatFallback(), fmt.Errorf("phrase chat: %w", err) } text, _, perr := parseResponseMood(resp) if perr != nil { return ChatFallback(), fmt.Errorf("phrase chat: %w", perr) } if text != "" { return text, nil } // fallback: plain text without JSON if i := strings.IndexByte(resp, '\n'); i >= 0 { resp = resp[:i] } return strings.TrimSpace(resp), nil } // chatSystemPrompt returns the system prompt for conversational chat. // Prepends the shared context block when the phraser has one. func chatSystemPrompt(block func() string) string { // No self-introduction here: the persona block prepended one line above // already says who she is, same as router.KnowledgePrompt. // // The grammar examples used to be full clauses: ("я подумала", "я рада") // for her, ("ты сказал", "ты забыл") for him. A 1.7B copies those rather // than generalising from them. Observed on the box 2026-08-01: all three // chat replies in one session opened with "Я подумала, что ...", and one // ended "...немного тревожусь. ты сказал" — the second example pasted onto // the end of a finished sentence, which reads as a truncation but is not. // // So: contrastive pairs instead of usable openers. "рада, не рад" states // the rule as a correction, and short predicatives do not hand the model a // sentence frame to start with. The him-examples are gone entirely; the // "ты" instruction carries that on its own and those two produced the // worst output. The last line says outright not to echo the instructions, // because a small model will otherwise treat any quoted string as licence. // // Amended the same day: with the openers gone the tic went with them, but // "не забыл ли я" appeared — masculine, about herself. The old "я подумала" // had been suppressing that by accident, being a feminine past tense the // model could copy. Two short predicatives are not enough signal on their // own, so the rule is now stated as morphology (-ла) rather than as a pair // of words. A suffix rule generalises where an example only gets copied. base := `Ты разговариваешь с хозяином. О себе — в женском роде: "рада", не "рад"; "поняла", не "понял". Все свои глаголы в прошедшем времени оканчивай на -ла: сделала, забыла, записала, подумала. Он мужчина: обращайся к нему на "ты", в мужском роде. Никогда не "вы"/"ваш" и никогда "он"/"его" — ты говоришь ему, а не о нём. Отвечай по-русски, коротко: одна-три фразы, живым языком. Ты доброжелательная, тебе интересно, но чувства не изображай. Не повторяй формулировки из этой инструкции — отвечай своими словами. Отвечай ТОЛЬКО одним объектом JSON: {"response": "...", "mood": "neutral"}. В "response" — твой ответ. В "mood" — ровно одно из: neutral, happy, thinking, tired, confused.` return persona.Prepend(block, base) } // chatWithMessages sends a full message array (system + history + current) to // the LLM completion endpoint. Like chatWithSystem but for an arbitrary message // slice — the caller owns the system prompt placement. func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTokens int) (string, error) { // Same silent preference as chatWithSystem, when the array is the shape // llm.Req can carry: one system turn and one user turn. PhraseChat already // folds the history into a single user message (some chat templates reject // consecutive user turns), so today that is every call. A longer array goes // to the resident model rather than get flattened here, because flattening a // conversation is a decision its owner should make. if len(msgs) == 2 && msgs[0].Role == "system" && msgs[1].Role == "user" { if out, ok := p.remoteChat(ctx, msgs[0].Content, msgs[1].Content, maxTokens); ok { return out, nil } } base, release, err := p.acquire() if err != nil { return "", err } defer release() req := chatReq{ Messages: msgs, Temperature: p.temperature(), MaxTokens: maxTokens, Grammar: p.grammar(), RepeatPenalty: phraseRepeatPenalty, } body, err := json.Marshal(req) if err != nil { return "", fmt.Errorf("llm: marshal: %w", err) } httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body)) if err != nil { return "", fmt.Errorf("llm: request: %w", err) } httpReq.Header.Set("Content-Type", "application/json") resp, err := p.client.Do(httpReq) if err != nil { return "", fmt.Errorf("llm: post: %w", err) } defer resp.Body.Close() raw, err := io.ReadAll(resp.Body) if err != nil { return "", fmt.Errorf("llm: read: %w", err) } if resp.StatusCode != 200 { return "", fmt.Errorf("llm: status %d: %s", resp.StatusCode, strings.TrimSpace(string(raw))) } var cr chatResp if err := json.Unmarshal(raw, &cr); err != nil { return "", fmt.Errorf("llm: parse: %w", err) } if len(cr.Choices) == 0 { return "", fmt.Errorf("llm: no choices in response") } logIfTruncated("chat", cr.Choices[0].FinishReason, maxTokens) content := cr.Choices[0].Message.Content if content == "" { content = cr.Choices[0].Message.ReasoningContent } log.Printf("llm raw content: %q", content) return stripThink(content), nil } func (p *LLMPhraser) PhraseReminder(ctx context.Context, d loop.ReminderDecision) (delivery.PhrasedReminder, error) { text := extractReminderText(d.Reminder.Payload) if text == "" { text = "reminder" } // Russian, like the other two prompts (Vikunja #404). Asking a model for a // Russian reply in English is asking it to switch languages mid-prompt, // and a 1.7B sometimes answers in the language it was asked in. The // persona rules and the JSON contract are not repeated here: this call // goes through chat(), so nudgeSystem already states both, and a second // statement of the same contract is one more thing that can drift. prompt := fmt.Sprintf( `Он поставил напоминание: "%s". Скажи это своими словами, коротко и мягко — одно предложение.`, text, ) resp, err := p.chat(ctx, prompt) if err != nil { return delivery.PhrasedReminder{}, err } body, mood, perr := parseResponseMood(resp) if perr != nil { // Truncated JSON. Fall through to the reminder's own text. log.Printf("phraser: PhraseReminder: %v", perr) body, mood = "", "" } if body == "" { // fallback: try old body/summary format body, _ = parsePhrase(resp) } if body == "" { body = text } if mood == "" { mood = "neutral" } summary := body if len(summary) > 60 { summary = summary[:57] + "..." } return delivery.PhrasedReminder{Decision: d, Body: body, Summary: summary, Mood: mood}, nil } type chatMsg struct { Role string `json:"role"` Content string `json:"content"` } type chatReq struct { Model string `json:"model"` Messages []chatMsg `json:"messages"` Temperature float64 `json:"temperature"` MaxTokens int `json:"max_tokens"` // Grammar is llama-server's `grammar` field (GBNF). Same wiring as // internal/llm.Req.Grammar. Empty ⇒ unconstrained sampling. Grammar string `json:"grammar,omitempty"` // RepeatPenalty — defence in depth behind the bounded grammar, not the fix // for #531. This struct had no such field, so every caller through // chatWithSystem ran at the server default of 1.0 while Replier.PhraseReply // sent 1.3 through internal/llm and was protected by accident. Two wire // structs that disagree about the sampler is the condition that let one // path run away and the other not, and it should not survive as a // difference nobody chose. RepeatPenalty float64 `json:"repeat_penalty,omitempty"` } // phraseRepeatPenalty — matches Replier.PhraseReply, which has sent 1.3 since // it was written. The value is not tuned here and is not what stops the // whitespace loop; the bounded ws rule is. It is here so the two phrasing // paths sample alike. const phraseRepeatPenalty = 1.3 // responseGrammar — GBNF constraining the model to the documented phrasing // contract and nothing else: {"response": "", "mood": ""}. // // Without it a 0.8B answers roughly one chat turn in three with open reasoning // as plain text ("Thinking Process:" …), which no tag-stripper can remove and // which eats the token budget before the JSON closes. Modelled on // routeGrammar in internal/router/llmrouter.go so the two read alike. // // text accepts ANY codepoint except the two JSON must escape and the control // range — the replies are Russian, so an ASCII-only rule would make every reply // empty. The escape rule is what lets the model close a string it opened with a // quote inside. Length is bounded so a repetition loop truncates the field, not // the JSON object. // // The control range is excluded because a raw newline inside a JSON string is // not JSON (Vikunja #537). The class used to be `[^"\\]`, which let the model // write a multi-line reply that satisfied the grammar and then failed // json.Unmarshal with "invalid character '\n' in string literal" — the object // starts with "{", so it came back as errBrokenJSON and the case answered with // an empty string. Sixty of the failures in the 2026-08-05 temperature sweep // were that one error, and it never once hit the token cap, which is why the // truncation reading was wrong. The escape alternatives are llama.cpp's own // json.gbnf: a model that wants a line break must write \n, which parses. // // That bound was 400 and 400 was too tight. Measured against Qwen3.5-0.8B: on // "почему гром слышно позже молнии?" the reply came back exactly 400 characters // long, cut mid-word ("Нужно записать и,"), at every token cap from 256 to 2048. // So the token cap was never what stopped it — this rule was. 1000 characters is // roughly six Russian sentences, still short enough to stop a repetition loop. // // ws is bounded for the same reason and it is the more expensive of the two. // `*` let the model open the object and then satisfy ws with whitespace until // max_tokens, which is 512 here: both interactive turns measured on 2026-08-04 // decoded exactly 512 tokens and spent 24-30 seconds doing it, all of it // whitespace (Vikunja #531). Nothing on this path sends a repeat penalty — // chatReq had no field for one — so the sampler never broke the loop. {0,4} // was measured: three runs, three clean stops at 33 tokens, no penalty needed. const responseGrammar = ` root ::= "{" ws "\"response\"" ws ":" ws string ws "," ws "\"mood\"" ws ":" ws mood ws "}" mood ::= "\"neutral\"" | "\"happy\"" | "\"thinking\"" | "\"tired\"" | "\"confused\"" string ::= "\"" ([^"\\\x00-\x1F] | "\\" ["\\/bfnrt] | "\\u" [0-9a-fA-F]{4}){0,1000} "\"" ws ::= [ \t\n]{0,4} ` // ResponseGrammar exposes responseGrammar to the other callers that emit the // same {"response","mood"} contract — cmd/mavend's reactive replier, which is // parsed by the same two fields. One definition, so the two cannot drift. const ResponseGrammar = responseGrammar // grammar returns the GBNF to attach to a phrasing request, or "" when the // operator turned it off. func (p *LLMPhraser) grammar() string { if p.cfg.NoGrammar { return "" } return responseGrammar } type chatResp struct { Choices []struct { Message struct { Content string `json:"content"` Reasoning string `json:"reasoning"` ReasoningContent string `json:"reasoning_content"` } `json:"message"` // FinishReason — "stop" when the model chose to end, "length" when the // token cap cut it off. Parsed since #531, where two turns ran to the // 512-token cap and both happened to parse anyway: the grammar had // already closed the JSON, so a truncated generation was indistinguishable // from a good one at every layer above this struct. FinishReason string `json:"finish_reason"` } `json:"choices"` } // logIfTruncated says so when a generation stopped at the token cap. // // A cap hit is never routine. Either the model was looping, which is the #531 // shape, or the reply was genuinely longer than maxTokens, which means she cut // herself off mid-sentence. Both are worth a line, and neither produced one // before: the caller sees a parsed string and cannot tell. func logIfTruncated(where, reason string, maxTokens int) { if reason == "length" { log.Printf("phraser: %s hit the %d-token cap (finish_reason=length) — the reply is truncated, or the model was looping", where, maxTokens) } } func (p *LLMPhraser) chat(ctx context.Context, userPrompt string) (string, error) { return p.chatWithSystem(ctx, p.systemPrompt(), userPrompt, 512) } func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, maxTokens int) (string, error) { // The workstation model first when it will take work, and silently: every // caller of this helper is on the silent half of the degradation rule. It // answering is not news, and it being asleep is not news either. if out, ok := p.remoteChat(ctx, system, user, maxTokens); ok { return out, nil } base, release, err := p.acquire() if err != nil { return "", err } defer release() req := chatReq{ Messages: []chatMsg{ {Role: "system", Content: system}, {Role: "user", Content: user}, }, Temperature: p.temperature(), MaxTokens: maxTokens, Grammar: p.grammar(), RepeatPenalty: phraseRepeatPenalty, } body, err := json.Marshal(req) if err != nil { return "", fmt.Errorf("llm: marshal: %w", err) } httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body)) if err != nil { return "", fmt.Errorf("llm: request: %w", err) } httpReq.Header.Set("Content-Type", "application/json") resp, err := p.client.Do(httpReq) if err != nil { return "", fmt.Errorf("llm: post: %w", err) } defer resp.Body.Close() raw, err := io.ReadAll(resp.Body) if err != nil { return "", fmt.Errorf("llm: read: %w", err) } if resp.StatusCode != 200 { return "", fmt.Errorf("llm: status %d: %s", resp.StatusCode, strings.TrimSpace(string(raw))) } var cr chatResp if err := json.Unmarshal(raw, &cr); err != nil { return "", fmt.Errorf("llm: parse: %w", err) } if len(cr.Choices) == 0 { return "", fmt.Errorf("llm: no choices in response") } logIfTruncated("chatWithSystem", cr.Choices[0].FinishReason, maxTokens) content := cr.Choices[0].Message.Content if content == "" { content = cr.Choices[0].Message.ReasoningContent } return stripThink(content), nil } // nudgeSystem — the phrasing contract for nudges. // // Written as filled-in examples, not as a schema with "..." in it. A 0.8B // copies whatever sits in the response slot, so a literal placeholder there // teaches it to answer with the placeholder. Measured: 7/15 nudges came back // as "..." before this. See docs/evals/2026-07-31-phrasing.md. // // Russian only, feminine self-reference, second person masculine (the owner is // a man). She talks TO him, informally, singular — never "вы", never "он". // One short sentence — the nudge is spoken aloud. // // What the ban on обращения forbids is pet names ("дорогой", "милый"), not his // name: "Ками, ноутбук на трёх процентах" is exactly how she talks, and the // unqualified word read as forbidding that too. Hence "ласковые обращения". // // The examples also never claim a physical act. She has no hands and no smart // plug — she can tell him the battery is at three percent, she cannot put the // laptop on charge. An example that says she did teaches the model to invent // actions Maven never took, which is worse than a missing nudge. const nudgeSystem = `Ты — Maven, домашняя ассистентка. О себе говоришь в женском роде ("я проверила", "я записала"). Владелец — мужчина, обращайся к нему в мужском роде ("ты пил", "ты забыл"). Говоришь с ним на "ты", в единственном числе ("выпей", "встань"). Никогда не "вы"/"вас"/"ваш" и никогда "он"/"его" — ты говоришь ему, а не о нём. Пиши ОДНО короткое напоминание по-русски: не больше 120 символов и не больше 16 слов. Только по делу. Запрещено: ласковые обращения ("дорогой", "милый"), эмодзи, извинения ("прости", "извини"), вопросы о самочувствии, похвала, больше одного восклицательного знака, английские слова кроме имён сервисов. Отвечай ТОЛЬКО одним объектом JSON с полями "response" и "mood". "response" — сам текст напоминания. "mood" — ровно одно из: neutral, happy, thinking, tired, confused. Так выглядит правильный ответ по форме. Темы здесь посторонние — их в запросе не будет: {"response": "Стиральная машина закончила. Развесь бельё.", "mood": "neutral"} {"response": "Ками, ноутбук на трёх процентах. Поставь его на зарядку.", "mood": "confused"} Это примеры ФОРМЫ, а не темы. Пиши только про ту ситуацию, которую тебе дали в запросе. Не копируй примеры и никогда не пиши "..." в поле response.` func (p *LLMPhraser) systemPrompt() string { return persona.Prepend(p.cfg.ContextBlock, nudgeSystem) } // knowledgePrompt — the no-sources branch: a world question, answered from // weights alone. The system prompt is the single tested source in // router.KnowledgePrompt. // // Split out of PhraseQuery so PhraseWorld sends the workstation model the same // bytes the resident model gets. Prompt parity across two models is a stated // constraint (CLAUDE.md), and two copies of a prompt is how it stops holding. func (p *LLMPhraser) knowledgePrompt(utterance string) (sys, user string) { return persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt()), fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance) } // evidencePrompt — the sources branch: read these, add nothing. Shared with // PhraseWorld for the same reason as knowledgePrompt. func (p *LLMPhraser) evidencePrompt(utterance string, notes []string) (sys, user string) { return p.querySystemPrompt(), fmt.Sprintf( "Он спрашивает: \"%s\"\n\nИсточники:\n%s\nОтветь ему коротко и своими словами, опираясь только на эти источники. Если ответа в них нет — так и скажи.", utterance, evidenceBlock(notes), ) } // querySystemPrompt returns the system prompt for the evidence branch of // PhraseQuery. Prepends the configured persona when set. // // Evidence-first, and that is the whole point of this prompt. Every source that // reaches PhraseQuery with something in hand — his notes, a stored fact, a page, // a live search, a ZIM article — arrives as numbered sources, and the model's // job here is to READ them, not to recall. A 1.7B asked a world question // answers from its weights with total confidence and no signal that it is // guessing; that is how "Война и мир" got Левитан as its author. The rule that // prevents it is stated three ways, because one way did not hold: answer from // the sources, say plainly when they do not answer, add nothing of your own. // // It no longer says "заметки". The sources are not always his notes, and // calling a Wikipedia paragraph his note both misleads him and licenses the // model to blur where an answer came from. // // No self-introduction here: the persona block prepended one line above already // says who she is, same as router.KnowledgePrompt. // // The opener is deliberate and stays: the fixed prefix is what marks the answer // as a lookup rather than as something she knows. The grammar examples are not // deliberate — same defect chatSystemPrompt had, where a 1.7B copies a quoted // word instead of generalising from it. Stated as morphology instead. func (p *LLMPhraser) querySystemPrompt() string { base := "Ты отвечаешь ему по источникам, которые тебе дали. Отвечай ТОЛЬКО по ним: всё, что ты говоришь, должно быть написано в источниках. " + "Если ответа в них нет — так и скажи и на этом остановись; не добавляй ничего из своих знаний и не догадывайся. " + "Не приплетай прошлые реплики разговора. " + "Отвечай по-русски, коротко и своими словами, начинай с \"вот что я нашла: \". О себе — в женском роде, глаголы в прошедшем времени с окончанием -ла. Он мужчина, обращайся к нему на \"ты\". Отвечай ТОЛЬКО одним объектом JSON: {\"response\": \"...\", \"mood\": \"neutral\"}." return persona.Prepend(p.cfg.ContextBlock, base) } // evidenceBlock renders the sources for the evidence branch of PhraseQuery. // // Numbered lines, one source each, rather than the quoted semicolon-joined // string this used to build. Two reasons, both measured on small models: a // numbered list survives being long, where a run-on quoted string blurs into // one claim the model then merges; and the numbering gives it something to // answer FROM, which is what makes "этого в источниках нет" reachable at all. func evidenceBlock(sources []string) string { var b strings.Builder for i, s := range sources { fmt.Fprintf(&b, "[%d] %s\n", i+1, s) } return b.String() } // nonEmpty drops blank sources and trims the rest, without touching the // caller's slice. func nonEmpty(sources []string) []string { out := make([]string, 0, len(sources)) for _, s := range sources { if s = strings.TrimSpace(s); s != "" { out = append(out, s) } } return out } // ruleTopics — Russian gloss for each built-in rule name. The rule names are // English identifiers; a 0.8B asked to nudge about "netdata_critical" writes // about nothing. The daemon knows what its own rules mean, so it says so. var ruleTopics = map[string]string{ "water": "он давно не пил воду", "meal": "он давно не ел", "break": "он давно без перерыва, пора встать и размяться", "service_down": "сервис не отвечает, лежит", "netdata_critical": "критический алярм в netdata, проблема с диском или местом", } // ruleKeywords — the word the message must contain. The 0.8B drifts to // whatever topic it saw last unless the required word is named outright. var ruleKeywords = map[string]string{ "water": "воду", "meal": "поешь", "break": "перерыв", "service_down": "сервис", "netdata_critical": "диск", } // ruleTopic turns a rule name into a Russian description of the situation. // "routine:зарядка" and "morning:утро" carry their own Russian suffix. func ruleTopic(rule string) string { if t, ok := ruleTopics[rule]; ok { return t } if i := strings.IndexByte(rule, ':'); i > 0 && i+1 < len(rule) { switch rule[:i] { case "morning": return "утро, пора начать день: " + rule[i+1:] default: return "пора сделать по распорядку: " + rule[i+1:] } } return rule } // ruleKeyword — the word the nudge must contain, or "" when the rule name's // own Russian suffix already is that word. func ruleKeyword(rule string) string { if k, ok := ruleKeywords[rule]; ok { return k } if i := strings.IndexByte(rule, ':'); i > 0 && i+1 < len(rule) { return rule[i+1:] } return "" } // ruDur — duration in Russian. humanDur is English and its output was landing // verbatim in the message. func ruDur(d time.Duration) string { if d < 0 { d = 0 } h, m := int(d.Hours()), int(d.Minutes())%60 switch { case h >= 2: return fmt.Sprintf("%d ч", h) case h == 1 && m >= 30: return "полтора часа" case h == 1: return "час" default: return fmt.Sprintf("%d мин", m) } } // fallbackNudge — plain Russian for when the model returns nothing parseable. var fallbackNudges = map[string]string{ "water": "Ты давно не пил воду.", "meal": "Ты давно не ел, поешь.", "break": "Пора сделать перерыв.", "service_down": "Сервис не отвечает.", "netdata_critical": "Критический алярм: проверь диск.", } func fallbackNudge(c loop.Candidate) string { if down := loop.DownServices(c.State); len(down) > 0 { return "Не отвечает: " + strings.Join(down, ", ") + "." } if s, ok := fallbackNudges[c.Rule.Name]; ok { return s } if kw := ruleKeyword(c.Rule.Name); kw != "" { return "Напоминаю: " + kw + "." } return "Напоминаю о деле." } func buildNudgePrompt(c loop.Candidate) string { var ctxParts []string ctxParts = append(ctxParts, "Ситуация: "+ruleTopic(c.Rule.Name)) if f, ok := c.State.Facts[c.Rule.Name]; ok && f.Key != "" && f.Key != c.Rule.Name { ctxParts = append(ctxParts, "Что именно: "+f.Key) } if down := loop.DownServices(c.State); len(down) > 0 { // The names come from the same helper the rule fired on, so the model // is never handed a service that is actually up. ctxParts = append(ctxParts, "Какие сервисы лежат: "+strings.Join(down, ", ")) } if d, ok := c.State.Since(c.Rule.Name); ok { ctxParts = append(ctxParts, "Прошло: "+ruDur(d)) } switch sevLabel(c.Severity) { case "alarm": ctxParts = append(ctxParts, "Срочно, скажи прямо.") case "ops": ctxParts = append(ctxParts, "Это про сервер, не про здоровье.") } tail := "Напиши напоминание про эту ситуацию. Одно предложение, по-русски, в JSON." if kw := ruleKeyword(c.Rule.Name); kw != "" { // Last line on purpose: a 0.8B weights the end of the prompt hardest, // and without the required word it drifts back to the examples. tail += " Ответ ДОЛЖЕН содержать слово «" + kw + "»." } return strings.Join(ctxParts, "\n") + "\n\n" + tail } type responseMood struct { Response string `json:"response"` Mood string `json:"mood"` } // errBrokenJSON — the model started a JSON object and never finished it. // That is a failed generation, not a reply. Callers must use their fallback. var errBrokenJSON = fmt.Errorf("phraser: model output starts as JSON but does not parse") // parseResponseMood extracts {"response","mood"} from LLM output, tolerant // of thinking tokens and extra text before/after the JSON block. // // Three outcomes: // - parsed fine → the fields, nil error. // - output never looked like JSON → ("", "", nil). The caller may ship it // as-is; small models sometimes answer in bare prose and that is fine. // - output starts with "{" but does not parse → errBrokenJSON. The grammar // guarantees a valid *prefix*, so a generation that hits the token cap // mid-object comes back as a fragment like `{` or `{\n "`. Shipping that // as a reply is the bug this error exists to stop. func parseResponseMood(raw string) (response, mood string, err error) { cleaned := strings.TrimSpace(raw) start := strings.Index(cleaned, "{") end := strings.LastIndex(cleaned, "}") if start < 0 || end < 0 || end <= start { if strings.HasPrefix(cleaned, "{") { return "", "", errBrokenJSON } return "", "", nil } var parsed responseMood if e := json.Unmarshal([]byte(escapeRawControls(cleaned[start:end+1])), &parsed); e != nil { if strings.HasPrefix(cleaned, "{") { return "", "", errBrokenJSON } return "", "", nil } return parsed.Response, parsed.Mood, nil } // escapeRawControls escapes the control characters a model writes literally // inside a JSON string, so a reply that is otherwise fine still parses. // // The grammar is what stops these being generated (Vikunja #537). This is the // second line, for the paths that send no grammar at all — NoGrammar, and any // remote model whose server ignores one. A raw newline is the shape that was // measured; the rest of the range is here because the same argument covers it. // // Inside a string only. The first version escaped the whole object on the // argument that JSON permits no control character outside a string either, so // rewriting one could not do harm. That argument is wrong: JSON permits a // newline, a tab and a return BETWEEN tokens, which is what pretty-printing is. // Qwen3-1.7B pretty-prints — it opens `{` and writes three newlines before the // first key — and escaping those into a literal backslash-n broke every reply // it wrote. Measured 2026-08-05 on the talk fixture: 31 of 36 conversational // cases came back as errBrokenJSON and answered from the stub (Vikunja #44). func escapeRawControls(s string) string { if !strings.ContainsFunc(s, func(r rune) bool { return r < 0x20 }) { return s } var b strings.Builder b.Grow(len(s) + 8) inString := false escaped := false for _, r := range s { switch { case escaped: // The character after a backslash is the model's own escape and is // already whatever it meant to write. escaped = false b.WriteRune(r) continue case inString && r == '\\': escaped = true b.WriteRune(r) continue case r == '"': inString = !inString b.WriteRune(r) continue } switch { case !inString || r >= 0x20: b.WriteRune(r) case r == '\n': b.WriteString(`\n`) case r == '\r': b.WriteString(`\r`) case r == '\t': b.WriteString(`\t`) default: fmt.Fprintf(&b, `\u%04x`, r) } } return b.String() } func parsePhrase(raw string) (body, summary string) { cleaned := strings.TrimSpace(raw) start := strings.Index(cleaned, "{") end := strings.LastIndex(cleaned, "}") if start < 0 || end < 0 || end <= start { return "", "" } var parsed struct { Body string `json:"body"` Summary string `json:"summary"` } if err := json.Unmarshal([]byte(cleaned[start:end+1]), &parsed); err != nil { return "", "" } return parsed.Body, parsed.Summary } func extractPort(listen string) string { _, port, _ := strings.Cut(listen, ":") if port == "" { return "0" } return port } // stripThink removes the block that Thinking-variant models emit // before the actual response. No-op when no think block is present. func stripThink(s string) string { if i := strings.LastIndex(s, ""); i >= 0 { s = strings.TrimSpace(s[i+8:]) } return s }