4dbeca5a2e
Qwen3-1.7B pretty-prints its JSON: it opens the object and writes three newlines before the first key. escapeRawControls rewrote those structural newlines into a literal backslash-n, which is legal nowhere outside a string, so the object stopped parsing and came back as errBrokenJSON. The comment claimed escaping unconditionally could not turn valid JSON into anything else, on the grounds that JSON permits no control character outside a string. It permits three: newline, tab and return are whitespace between tokens, and that is what pretty-printing is made of. Measured on the talk fixture against the resident model: 31 of 36 conversational cases were failing generations and answered from the stub. Every chat reply and every knowledge answer the resident model wrote was being discarded. Now 25/36 pass every check, 0 errors, and the 15 nudges stay at 15/15.
1241 lines
49 KiB
Go
1241 lines
49 KiB
Go
package phraser
|
||
|
||
import (
|
||
"bufio"
|
||
"bytes"
|
||
"context"
|
||
"encoding/json"
|
||
"errors"
|
||
"fmt"
|
||
"io"
|
||
"log"
|
||
"net/http"
|
||
"os"
|
||
"os/exec"
|
||
"regexp"
|
||
"strings"
|
||
"sync"
|
||
"syscall"
|
||
"time"
|
||
|
||
"github.com/kami/maven/internal/delivery"
|
||
"github.com/kami/maven/internal/dialogue"
|
||
"github.com/kami/maven/internal/loop"
|
||
"github.com/kami/maven/internal/persona"
|
||
"github.com/kami/maven/internal/router"
|
||
)
|
||
|
||
var listenRE = regexp.MustCompile(`listening on (https?://\S+)`)
|
||
|
||
// errEmptyResponse — the server answered and said nothing. Separate from a
|
||
// transport failure: the model is up and produced no tokens, which is still not
|
||
// an answer and must not score as one.
|
||
var errEmptyResponse = errors.New("phraser: empty response from the model")
|
||
|
||
type LLMPhraser struct {
|
||
cfg Config
|
||
client *http.Client
|
||
|
||
// tmpl — the hand-written Russian nudges. Default path for nudges; see
|
||
// Config.LLMNudges. nil only if the template file failed to load.
|
||
tmpl *NudgeTemplates
|
||
|
||
// spawnCtx — the parent of every llama-server this phraser starts, i.e. the
|
||
// daemon's own context. Deliberately NOT the per-request context of the call
|
||
// that asked for a model swap: that one is cancelled the moment the request
|
||
// returns, which would kill the model it had just loaded.
|
||
spawnCtx context.Context
|
||
cancel context.CancelFunc
|
||
|
||
// launch / probe — the two side effects of a swap, injectable so the swap
|
||
// logic is testable without a real llama-server and a real model file.
|
||
// launch is nil when this phraser does not own its server (NewLLMPhraserAt),
|
||
// which is also what makes Swap refuse there.
|
||
launch func(ctx context.Context, cfg Config) (backend, error)
|
||
probe func(ctx context.Context, base string) (string, error)
|
||
|
||
// remote — the workstation model, when one is configured. Set once at wiring
|
||
// time by UseRemote and read on every phrasing call. nil ⇒ every call goes to
|
||
// the resident llama-server this phraser owns, which is the whole deploy
|
||
// before a `workstation` block exists. See world.go.
|
||
remote Remote
|
||
|
||
// swapMu — single-flight around Swap. Held for the whole swap, including the
|
||
// model load, so two concurrent swap requests can never both be loading.
|
||
swapMu sync.Mutex
|
||
|
||
// mu guards everything below: the live backend, the swap gate and the
|
||
// in-flight request count. See acquire/quiesce in swap.go.
|
||
mu sync.Mutex
|
||
be backend
|
||
live liveModel
|
||
swapping bool
|
||
inflight int
|
||
observers []func(baseURL string)
|
||
}
|
||
|
||
// liveModel — what is actually loaded right now. Distinct from Config, which
|
||
// stays immutable after construction: a swap changes these three fields and
|
||
// nothing else, so no reader of cfg (prompts, grammar, timeouts) races a swap.
|
||
type liveModel struct {
|
||
ModelPath string
|
||
NGpuLayers int
|
||
NCtx int
|
||
}
|
||
|
||
type Config struct {
|
||
ModelPath string
|
||
BinPath string
|
||
Listen string
|
||
NGpuLayers int
|
||
NCtx int
|
||
Timeout time.Duration
|
||
|
||
// StartupTimeout bounds the wait for llama-server to print the address it
|
||
// listens on. A config field and not a constant because the box may
|
||
// legitimately need longer: a cold 1.7B loading off a spinning disk can
|
||
// outrun a minute, and until this existed that returned "server did not
|
||
// start within 60s" with no way to raise it.
|
||
//
|
||
// 0 ⇒ defaultStartupTimeout.
|
||
StartupTimeout time.Duration
|
||
|
||
// CacheRAMMiB bounds llama-server's prompt cache, which is what actually ate
|
||
// this box. Measured on homesrv 2026-08-03: the server's own default limit is
|
||
// 8192 MiB, it stores the full KV state of every idle slot it evicts (112 kiB
|
||
// per token, so 166 MiB for one 1521-token prompt), and RSS climbed by that
|
||
// much per distinct prompt until it hit 7.9 GB and half a gigabyte went to
|
||
// swap. Weights are only 1.1 GB and mmapped, and -ngl 99 costs almost no RSS
|
||
// because RADV keeps device memory outside the process.
|
||
//
|
||
// 0 ⇒ the flag is not passed and the server's own 8 GiB default applies. That
|
||
// is the escape hatch for a llama-server too old to know --cache-ram, not a
|
||
// recommendation. See docs/evals/2026-08-03-llama-prompt-cache.md.
|
||
CacheRAMMiB int
|
||
|
||
// ContextBlock renders the shared context block (who he is, how to
|
||
// address him, the time) fresh for each turn. See internal/persona.
|
||
// nil ⇒ no block, the prompts stand alone.
|
||
ContextBlock func() string
|
||
|
||
// LLMNudges puts the model back in charge of nudge wording.
|
||
//
|
||
// Off by default, and that is a deliberate deprecation of LLM-phrased
|
||
// nudges: hand-written templates (nudges_ru_v1.json) word every nudge now.
|
||
// A nudge has nothing to be creative about, and measured over many runs the
|
||
// 0.8B broke the persona (formal "вы", plural imperatives, masculine
|
||
// self-reference) and invented facts and units. Templates score 15/15 on the
|
||
// nudge fixture, the model 11-13/15.
|
||
//
|
||
// The LLM path is kept, not deleted: flip this on to get it back. Chat,
|
||
// query and reminder phrasing are untouched and still go through the model.
|
||
LLMNudges bool
|
||
|
||
// Temperature — what every phrasing call samples at. 0 ⇒ 0.7, which is
|
||
// what this transport has always sent.
|
||
//
|
||
// A field rather than a constant so the talk fixture can sweep it
|
||
// (Vikunja #402). Sampling is a dial, and a dial nobody can turn from
|
||
// outside the package cannot be measured, only argued about.
|
||
Temperature float64
|
||
|
||
// NoGrammar turns the GBNF constraint off (zero value ⇒ grammar ON).
|
||
// The escape hatch exists because the target resident model — the
|
||
// locally CPT'd Qwen3-1.7B — does not exist yet: if its chat template
|
||
// ever fights the grammar, the fix should be a config flip on the
|
||
// deploy box, not a code change and a rebuild.
|
||
NoGrammar bool
|
||
}
|
||
|
||
func DefaultConfig(modelPath string) Config {
|
||
return Config{
|
||
ModelPath: modelPath,
|
||
BinPath: "llama-server",
|
||
Listen: "127.0.0.1:0",
|
||
NGpuLayers: -1,
|
||
NCtx: 2048,
|
||
// 512 MiB caps total RSS near 1 GB and still holds several recent prompts.
|
||
CacheRAMMiB: 512,
|
||
Timeout: 30 * time.Second,
|
||
|
||
StartupTimeout: defaultStartupTimeout,
|
||
}
|
||
}
|
||
|
||
func NewLLMPhraser(ctx context.Context, cfg Config) (*LLMPhraser, error) {
|
||
ctx, cancel := context.WithCancel(ctx)
|
||
p := &LLMPhraser{
|
||
cfg: cfg,
|
||
client: &http.Client{Timeout: cfg.Timeout},
|
||
tmpl: loadNudgeTemplates(),
|
||
spawnCtx: ctx,
|
||
cancel: cancel,
|
||
launch: spawnLlamaServer,
|
||
probe: defaultProbe,
|
||
live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
|
||
}
|
||
be, err := p.launch(ctx, cfg)
|
||
if err != nil {
|
||
cancel()
|
||
return nil, err
|
||
}
|
||
p.be = be
|
||
return p, nil
|
||
}
|
||
|
||
// NewLLMPhraserAt wires a phraser to a llama-server that someone else started
|
||
// and owns. It spawns nothing, so Close does not kill anything.
|
||
//
|
||
// This exists for the phrasing scorer (internal/phraser/eval), which must
|
||
// measure the phrasing against a shared llama-server without taking the model
|
||
// load hit per run or killing a server another process depends on. The daemon
|
||
// still uses NewLLMPhraser and still owns its own child process.
|
||
func NewLLMPhraserAt(baseURL string, cfg Config) *LLMPhraser {
|
||
return &LLMPhraser{
|
||
cfg: cfg,
|
||
client: &http.Client{Timeout: cfg.Timeout},
|
||
tmpl: loadNudgeTemplates(),
|
||
spawnCtx: context.Background(),
|
||
cancel: func() {},
|
||
probe: defaultProbe,
|
||
// launch stays nil: we did not start this server, so we must not stop it.
|
||
// Swap therefore refuses here (ErrSwapNotOwned) instead of killing a
|
||
// server another process depends on.
|
||
be: borrowedBackend(strings.TrimSuffix(baseURL, "/")),
|
||
live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
|
||
}
|
||
}
|
||
|
||
// loadNudgeTemplates loads the Russian nudge templates. A broken template file
|
||
// must not stop the daemon booting, so a failure logs and leaves the LLM path
|
||
// in charge of nudges.
|
||
func loadNudgeTemplates() *NudgeTemplates {
|
||
nt, err := NewNudgeTemplates(nil)
|
||
if err != nil {
|
||
log.Printf("phraser: nudge templates unavailable, using the model: %v", err)
|
||
return nil
|
||
}
|
||
return nt
|
||
}
|
||
|
||
// backend — one llama-server this phraser talks to. Two implementations: a
|
||
// llamaProc we spawned and must reap, and a borrowedBackend someone else owns.
|
||
type backend interface {
|
||
BaseURL() string
|
||
Close() error
|
||
}
|
||
|
||
// borrowedBackend — a server started and owned by someone else (the phrasing
|
||
// scorer's shared llama-server). Closing it is a no-op by construction.
|
||
type borrowedBackend string
|
||
|
||
func (b borrowedBackend) BaseURL() string { return string(b) }
|
||
func (b borrowedBackend) Close() error { return nil }
|
||
|
||
// llamaProc — a llama-server child process plus the goroutine reading its
|
||
// stderr. Close kills and reaps it; see the Pdeathsig note in spawnLlamaServer.
|
||
type llamaProc struct {
|
||
base string
|
||
cmd *exec.Cmd
|
||
cancel context.CancelFunc
|
||
wg sync.WaitGroup
|
||
}
|
||
|
||
func (l *llamaProc) BaseURL() string { return l.base }
|
||
|
||
func (l *llamaProc) Close() error {
|
||
l.cancel()
|
||
if l.cmd != nil && l.cmd.Process != nil {
|
||
_ = l.cmd.Process.Kill()
|
||
_ = l.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
|
||
}
|
||
l.wg.Wait()
|
||
return nil
|
||
}
|
||
|
||
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
|
||
// address it is listening on. ctx owns the process lifetime, so it must be the
|
||
// daemon's context, not a request's.
|
||
func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
|
||
ctx, cancel := context.WithCancel(ctx)
|
||
p, err := startLlamaProc(ctx, cfg)
|
||
if err != nil {
|
||
cancel()
|
||
return nil, err
|
||
}
|
||
p.cancel = cancel
|
||
return p, nil
|
||
}
|
||
|
||
// llamaArgs is the command line for one resident server. It is a function and
|
||
// not an inline literal because kill-maven.sh's orphan sweep matches against
|
||
// this exact line, and a test pins the two together.
|
||
func llamaArgs(cfg Config) []string {
|
||
args := []string{
|
||
"-m", cfg.ModelPath,
|
||
"--host", "127.0.0.1",
|
||
"--port", extractPort(cfg.Listen),
|
||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||
"--no-webui",
|
||
}
|
||
if cfg.CacheRAMMiB > 0 {
|
||
args = append(args, "--cache-ram", fmt.Sprintf("%d", cfg.CacheRAMMiB))
|
||
}
|
||
return args
|
||
}
|
||
|
||
// defaultStartupTimeout — the wait for llama-server's listen line when Config
|
||
// does not set one. A cold model load off disk is the slow part.
|
||
const defaultStartupTimeout = 60 * time.Second
|
||
|
||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||
startupTimeout := cfg.StartupTimeout
|
||
if startupTimeout <= 0 {
|
||
startupTimeout = defaultStartupTimeout
|
||
}
|
||
p := &llamaProc{}
|
||
cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...)
|
||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
|
||
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
|
||
// eating GPU/RAM); repeated dev restarts pile up orphans until the box OOMs.
|
||
// Setpgid isolates it in its own process group so a stray Ctrl-C on the
|
||
// terminal group doesn't half-kill it out from under us. (Linux-only, like
|
||
// the rest of the daemon.)
|
||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true, Pdeathsig: syscall.SIGKILL}
|
||
p.cmd = cmd
|
||
|
||
// One pipe for both streams. llama.cpp writes its buffer sizes, KV-cache
|
||
// layout and offload lines to stderr and its request log to stdout, and
|
||
// stdout used to go nowhere at all — so nothing about the model's memory was
|
||
// diagnosable from a running box. Both ends land in mavend's log now.
|
||
pr, pw, err := os.Pipe()
|
||
if err != nil {
|
||
return nil, fmt.Errorf("llm: output pipe: %w", err)
|
||
}
|
||
cmd.Stdout = pw
|
||
cmd.Stderr = pw
|
||
|
||
if err := cmd.Start(); err != nil {
|
||
pr.Close()
|
||
pw.Close()
|
||
return nil, fmt.Errorf("llm: start: %w", err)
|
||
}
|
||
// The child holds the only other reference to the write end. Dropping ours
|
||
// is what makes the reader see EOF when the child dies.
|
||
pw.Close()
|
||
|
||
portCh := make(chan string, 1)
|
||
errCh := make(chan error, 1)
|
||
tail := &lineTail{}
|
||
p.wg.Add(1)
|
||
go func() {
|
||
defer p.wg.Done()
|
||
defer pr.Close()
|
||
sc := bufio.NewScanner(pr)
|
||
// llama.cpp prints one prompt per line and a prompt can be long.
|
||
sc.Buffer(make([]byte, 0, 64*1024), 1024*1024)
|
||
listening := false
|
||
for sc.Scan() {
|
||
line := sc.Bytes()
|
||
log.Printf("llama: %s", line)
|
||
if !listening {
|
||
tail.add(string(line))
|
||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||
listening = true
|
||
portCh <- string(m[1])
|
||
close(portCh)
|
||
}
|
||
}
|
||
}
|
||
err := sc.Err()
|
||
if err == nil {
|
||
err = io.EOF
|
||
}
|
||
errCh <- err
|
||
}()
|
||
|
||
fail := func(err error) (*llamaProc, error) {
|
||
_ = cmd.Process.Kill()
|
||
_ = cmd.Wait()
|
||
return nil, err
|
||
}
|
||
select {
|
||
case addr := <-portCh:
|
||
p.base = addr
|
||
return p, nil
|
||
case err := <-errCh:
|
||
// The tail is the whole diagnosis when the server dies during load: bare
|
||
// "EOF" never said which layer or which allocation it choked on.
|
||
return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String()))
|
||
case <-ctx.Done():
|
||
return fail(ctx.Err())
|
||
case <-time.After(startupTimeout):
|
||
return fail(fmt.Errorf("llm: server did not start within %s; last output: %s", startupTimeout, tail.String()))
|
||
}
|
||
}
|
||
|
||
// lineTail keeps the last few startup lines so a server that dies before it
|
||
// listens can say why in the error, not just "EOF". Written by the reader
|
||
// goroutine and read by whoever gives up on startup, so it takes a lock.
|
||
type lineTail struct {
|
||
mu sync.Mutex
|
||
lines []string
|
||
}
|
||
|
||
const lineTailMax = 12
|
||
|
||
func (t *lineTail) add(line string) {
|
||
t.mu.Lock()
|
||
defer t.mu.Unlock()
|
||
t.lines = append(t.lines, line)
|
||
if len(t.lines) > lineTailMax {
|
||
t.lines = t.lines[len(t.lines)-lineTailMax:]
|
||
}
|
||
}
|
||
|
||
func (t *lineTail) String() string {
|
||
t.mu.Lock()
|
||
defer t.mu.Unlock()
|
||
if len(t.lines) == 0 {
|
||
return "(no output)"
|
||
}
|
||
return strings.Join(t.lines, " | ")
|
||
}
|
||
|
||
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
||
// the model is swapped, so callers that cache it must register an observer
|
||
// (OnSwap) rather than keeping the string forever.
|
||
func (p *LLMPhraser) BaseURL() string {
|
||
p.mu.Lock()
|
||
defer p.mu.Unlock()
|
||
if p.be == nil {
|
||
return ""
|
||
}
|
||
return p.be.BaseURL()
|
||
}
|
||
|
||
func (p *LLMPhraser) Close() error {
|
||
p.cancel()
|
||
p.mu.Lock()
|
||
be := p.be
|
||
p.be = nil
|
||
p.mu.Unlock()
|
||
if be != nil {
|
||
return be.Close()
|
||
}
|
||
return nil
|
||
}
|
||
|
||
func (p *LLMPhraser) PhraseNudge(ctx context.Context, c loop.Candidate) (delivery.PhrasedNudge, error) {
|
||
// Templates first — see Config.LLMNudges for why this is the default.
|
||
if !p.cfg.LLMNudges && p.tmpl != nil {
|
||
return p.tmpl.PhraseNudge(ctx, c)
|
||
}
|
||
prompt := buildNudgePrompt(c)
|
||
resp, err := p.chat(ctx, prompt)
|
||
if err != nil {
|
||
return delivery.PhrasedNudge{}, err
|
||
}
|
||
body, mood, perr := parseResponseMood(resp)
|
||
if perr != nil {
|
||
// Truncated JSON. Not a nudge — use the plain Russian fallback.
|
||
log.Printf("phraser: PhraseNudge: %v", perr)
|
||
body, mood = "", ""
|
||
}
|
||
if body == "" {
|
||
// fallback: try old body/summary format
|
||
body, _ = parsePhrase(resp)
|
||
}
|
||
if body == "" {
|
||
// The model said nothing usable. Say it in Russian anyway — this text
|
||
// goes straight to a Russian piper voice, so the old "water — care"
|
||
// fallback was unspeakable.
|
||
body = fallbackNudge(c)
|
||
}
|
||
if mood == "" {
|
||
mood = "neutral"
|
||
}
|
||
return delivery.PhrasedNudge{Candidate: c, Body: body, Summary: body, Mood: mood}, nil
|
||
}
|
||
|
||
// PhraseQuery prompts the LLM with the user's utterance and matching notes to
|
||
// compose a natural answer. On any LLM error it returns the fallback text —
|
||
// "вот что я нашла: <notes>", or "не знаю." with no notes — and the error
|
||
// together. The daemon uses the text and keeps the turn alive; a caller that is
|
||
// measuring counts the failure. Until Vikunja #397 the error was dropped, so a
|
||
// dead server scored as bad phrasing.
|
||
func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error) {
|
||
// Blank sources are no sources. A caller that hands over one empty string —
|
||
// a page that fetched to nothing, a snippet trimmed away — used to take the
|
||
// evidence branch and be told to answer from an empty list, which is the one
|
||
// prompt guaranteed to make a small model fill the gap from memory.
|
||
notes = nonEmpty(notes)
|
||
if len(notes) == 0 {
|
||
sys, prompt := p.knowledgePrompt(utterance)
|
||
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
|
||
if err != nil {
|
||
return UnknownFallback(), fmt.Errorf("phrase query (knowledge): %w", err)
|
||
}
|
||
if resp == "" {
|
||
return UnknownFallback(), errEmptyResponse
|
||
}
|
||
text, _, perr := parseResponseMood(resp)
|
||
if perr != nil {
|
||
return UnknownFallback(), fmt.Errorf("phrase query (knowledge): %w", perr)
|
||
}
|
||
if text != "" {
|
||
return text, nil
|
||
}
|
||
return resp, nil
|
||
}
|
||
sys, prompt := p.evidencePrompt(utterance, notes)
|
||
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
|
||
text, _, perr := parseResponseMood(resp)
|
||
if err != nil || perr != nil {
|
||
// Read the notes out rather than ship a broken fragment.
|
||
cause := err
|
||
if cause == nil {
|
||
cause = perr
|
||
}
|
||
return SourcesFallback(strings.Join(notes, "; ")),
|
||
fmt.Errorf("phrase query (evidence): %w", cause)
|
||
}
|
||
if text != "" {
|
||
return text, nil
|
||
}
|
||
return resp, nil
|
||
}
|
||
|
||
// PhraseChat uses the LLM to respond conversationally, building a multi-turn
|
||
// message array from dialogue history + the current user utterance. On any LLM
|
||
// error it returns both ChatFallback and the error, on the same rule as
|
||
// PhraseQuery: the fallback keeps the turn alive, the error stays visible.
|
||
func (p *LLMPhraser) PhraseChat(ctx context.Context, utterance string, history []dialogue.Turn) (string, error) {
|
||
sys := chatSystemPrompt(p.cfg.ContextBlock)
|
||
msgs := []chatMsg{
|
||
{Role: "system", Content: sys},
|
||
}
|
||
// Combine history and current utterance into one user message.
|
||
// Some model chat templates (Ministral, etc.) reject consecutive user turns.
|
||
var combined string
|
||
for _, t := range history {
|
||
combined += t.Text + "\n"
|
||
}
|
||
combined += utterance
|
||
msgs = append(msgs, chatMsg{Role: "user", Content: strings.TrimSpace(combined)})
|
||
|
||
resp, err := p.chatWithMessages(ctx, msgs, 768)
|
||
if err != nil {
|
||
return ChatFallback(), fmt.Errorf("phrase chat: %w", err)
|
||
}
|
||
text, _, perr := parseResponseMood(resp)
|
||
if perr != nil {
|
||
return ChatFallback(), fmt.Errorf("phrase chat: %w", perr)
|
||
}
|
||
if text != "" {
|
||
return text, nil
|
||
}
|
||
// fallback: plain text without JSON
|
||
if i := strings.IndexByte(resp, '\n'); i >= 0 {
|
||
resp = resp[:i]
|
||
}
|
||
return strings.TrimSpace(resp), nil
|
||
}
|
||
|
||
// chatSystemPrompt returns the system prompt for conversational chat.
|
||
// Prepends the shared context block when the phraser has one.
|
||
func chatSystemPrompt(block func() string) string {
|
||
// No self-introduction here: the persona block prepended one line above
|
||
// already says who she is, same as router.KnowledgePrompt.
|
||
//
|
||
// The grammar examples used to be full clauses: ("я подумала", "я рада")
|
||
// for her, ("ты сказал", "ты забыл") for him. A 1.7B copies those rather
|
||
// than generalising from them. Observed on the box 2026-08-01: all three
|
||
// chat replies in one session opened with "Я подумала, что ...", and one
|
||
// ended "...немного тревожусь. ты сказал" — the second example pasted onto
|
||
// the end of a finished sentence, which reads as a truncation but is not.
|
||
//
|
||
// So: contrastive pairs instead of usable openers. "рада, не рад" states
|
||
// the rule as a correction, and short predicatives do not hand the model a
|
||
// sentence frame to start with. The him-examples are gone entirely; the
|
||
// "ты" instruction carries that on its own and those two produced the
|
||
// worst output. The last line says outright not to echo the instructions,
|
||
// because a small model will otherwise treat any quoted string as licence.
|
||
//
|
||
// Amended the same day: with the openers gone the tic went with them, but
|
||
// "не забыл ли я" appeared — masculine, about herself. The old "я подумала"
|
||
// had been suppressing that by accident, being a feminine past tense the
|
||
// model could copy. Two short predicatives are not enough signal on their
|
||
// own, so the rule is now stated as morphology (-ла) rather than as a pair
|
||
// of words. A suffix rule generalises where an example only gets copied.
|
||
base := `Ты разговариваешь с хозяином.
|
||
|
||
О себе — в женском роде: "рада", не "рад"; "поняла", не "понял". Все свои глаголы в прошедшем времени оканчивай на -ла: сделала, забыла, записала, подумала. Он мужчина: обращайся к нему на "ты", в мужском роде. Никогда не "вы"/"ваш" и никогда "он"/"его" — ты говоришь ему, а не о нём.
|
||
|
||
Отвечай по-русски, коротко: одна-три фразы, живым языком. Ты доброжелательная, тебе интересно, но чувства не изображай. Не повторяй формулировки из этой инструкции — отвечай своими словами.
|
||
|
||
Отвечай ТОЛЬКО одним объектом JSON: {"response": "...", "mood": "neutral"}. В "response" — твой ответ. В "mood" — ровно одно из: neutral, happy, thinking, tired, confused.`
|
||
return persona.Prepend(block, base)
|
||
}
|
||
|
||
// chatWithMessages sends a full message array (system + history + current) to
|
||
// the LLM completion endpoint. Like chatWithSystem but for an arbitrary message
|
||
// slice — the caller owns the system prompt placement.
|
||
func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTokens int) (string, error) {
|
||
// Same silent preference as chatWithSystem, when the array is the shape
|
||
// llm.Req can carry: one system turn and one user turn. PhraseChat already
|
||
// folds the history into a single user message (some chat templates reject
|
||
// consecutive user turns), so today that is every call. A longer array goes
|
||
// to the resident model rather than get flattened here, because flattening a
|
||
// conversation is a decision its owner should make.
|
||
if len(msgs) == 2 && msgs[0].Role == "system" && msgs[1].Role == "user" {
|
||
if out, ok := p.remoteChat(ctx, msgs[0].Content, msgs[1].Content, maxTokens); ok {
|
||
return out, nil
|
||
}
|
||
}
|
||
base, release, err := p.acquire()
|
||
if err != nil {
|
||
return "", err
|
||
}
|
||
defer release()
|
||
req := chatReq{
|
||
Messages: msgs,
|
||
Temperature: p.temperature(),
|
||
MaxTokens: maxTokens,
|
||
Grammar: p.grammar(),
|
||
RepeatPenalty: phraseRepeatPenalty,
|
||
}
|
||
body, err := json.Marshal(req)
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: marshal: %w", err)
|
||
}
|
||
httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body))
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: request: %w", err)
|
||
}
|
||
httpReq.Header.Set("Content-Type", "application/json")
|
||
resp, err := p.client.Do(httpReq)
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: post: %w", err)
|
||
}
|
||
defer resp.Body.Close()
|
||
raw, err := io.ReadAll(resp.Body)
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: read: %w", err)
|
||
}
|
||
if resp.StatusCode != 200 {
|
||
return "", fmt.Errorf("llm: status %d: %s", resp.StatusCode, strings.TrimSpace(string(raw)))
|
||
}
|
||
var cr chatResp
|
||
if err := json.Unmarshal(raw, &cr); err != nil {
|
||
return "", fmt.Errorf("llm: parse: %w", err)
|
||
}
|
||
if len(cr.Choices) == 0 {
|
||
return "", fmt.Errorf("llm: no choices in response")
|
||
}
|
||
logIfTruncated("chat", cr.Choices[0].FinishReason, maxTokens)
|
||
content := cr.Choices[0].Message.Content
|
||
if content == "" {
|
||
content = cr.Choices[0].Message.ReasoningContent
|
||
}
|
||
log.Printf("llm raw content: %q", content)
|
||
return stripThink(content), nil
|
||
}
|
||
|
||
func (p *LLMPhraser) PhraseReminder(ctx context.Context, d loop.ReminderDecision) (delivery.PhrasedReminder, error) {
|
||
text := extractReminderText(d.Reminder.Payload)
|
||
if text == "" {
|
||
text = "reminder"
|
||
}
|
||
|
||
// Russian, like the other two prompts (Vikunja #404). Asking a model for a
|
||
// Russian reply in English is asking it to switch languages mid-prompt,
|
||
// and a 1.7B sometimes answers in the language it was asked in. The
|
||
// persona rules and the JSON contract are not repeated here: this call
|
||
// goes through chat(), so nudgeSystem already states both, and a second
|
||
// statement of the same contract is one more thing that can drift.
|
||
prompt := fmt.Sprintf(
|
||
`Он поставил напоминание: "%s". Скажи это своими словами, коротко и мягко — одно предложение.`,
|
||
text,
|
||
)
|
||
resp, err := p.chat(ctx, prompt)
|
||
if err != nil {
|
||
return delivery.PhrasedReminder{}, err
|
||
}
|
||
body, mood, perr := parseResponseMood(resp)
|
||
if perr != nil {
|
||
// Truncated JSON. Fall through to the reminder's own text.
|
||
log.Printf("phraser: PhraseReminder: %v", perr)
|
||
body, mood = "", ""
|
||
}
|
||
if body == "" {
|
||
// fallback: try old body/summary format
|
||
body, _ = parsePhrase(resp)
|
||
}
|
||
if body == "" {
|
||
body = text
|
||
}
|
||
if mood == "" {
|
||
mood = "neutral"
|
||
}
|
||
summary := body
|
||
if len(summary) > 60 {
|
||
summary = summary[:57] + "..."
|
||
}
|
||
return delivery.PhrasedReminder{Decision: d, Body: body, Summary: summary, Mood: mood}, nil
|
||
}
|
||
|
||
type chatMsg struct {
|
||
Role string `json:"role"`
|
||
Content string `json:"content"`
|
||
}
|
||
|
||
type chatReq struct {
|
||
Model string `json:"model"`
|
||
Messages []chatMsg `json:"messages"`
|
||
Temperature float64 `json:"temperature"`
|
||
MaxTokens int `json:"max_tokens"`
|
||
// Grammar is llama-server's `grammar` field (GBNF). Same wiring as
|
||
// internal/llm.Req.Grammar. Empty ⇒ unconstrained sampling.
|
||
Grammar string `json:"grammar,omitempty"`
|
||
// RepeatPenalty — defence in depth behind the bounded grammar, not the fix
|
||
// for #531. This struct had no such field, so every caller through
|
||
// chatWithSystem ran at the server default of 1.0 while Replier.PhraseReply
|
||
// sent 1.3 through internal/llm and was protected by accident. Two wire
|
||
// structs that disagree about the sampler is the condition that let one
|
||
// path run away and the other not, and it should not survive as a
|
||
// difference nobody chose.
|
||
RepeatPenalty float64 `json:"repeat_penalty,omitempty"`
|
||
}
|
||
|
||
// phraseRepeatPenalty — matches Replier.PhraseReply, which has sent 1.3 since
|
||
// it was written. The value is not tuned here and is not what stops the
|
||
// whitespace loop; the bounded ws rule is. It is here so the two phrasing
|
||
// paths sample alike.
|
||
const phraseRepeatPenalty = 1.3
|
||
|
||
// responseGrammar — GBNF constraining the model to the documented phrasing
|
||
// contract and nothing else: {"response": "<text>", "mood": "<enum>"}.
|
||
//
|
||
// Without it a 0.8B answers roughly one chat turn in three with open reasoning
|
||
// as plain text ("Thinking Process:" …), which no tag-stripper can remove and
|
||
// which eats the token budget before the JSON closes. Modelled on
|
||
// routeGrammar in internal/router/llmrouter.go so the two read alike.
|
||
//
|
||
// text accepts ANY codepoint except the two JSON must escape and the control
|
||
// range — the replies are Russian, so an ASCII-only rule would make every reply
|
||
// empty. The escape rule is what lets the model close a string it opened with a
|
||
// quote inside. Length is bounded so a repetition loop truncates the field, not
|
||
// the JSON object.
|
||
//
|
||
// The control range is excluded because a raw newline inside a JSON string is
|
||
// not JSON (Vikunja #537). The class used to be `[^"\\]`, which let the model
|
||
// write a multi-line reply that satisfied the grammar and then failed
|
||
// json.Unmarshal with "invalid character '\n' in string literal" — the object
|
||
// starts with "{", so it came back as errBrokenJSON and the case answered with
|
||
// an empty string. Sixty of the failures in the 2026-08-05 temperature sweep
|
||
// were that one error, and it never once hit the token cap, which is why the
|
||
// truncation reading was wrong. The escape alternatives are llama.cpp's own
|
||
// json.gbnf: a model that wants a line break must write \n, which parses.
|
||
//
|
||
// That bound was 400 and 400 was too tight. Measured against Qwen3.5-0.8B: on
|
||
// "почему гром слышно позже молнии?" the reply came back exactly 400 characters
|
||
// long, cut mid-word ("Нужно записать и,"), at every token cap from 256 to 2048.
|
||
// So the token cap was never what stopped it — this rule was. 1000 characters is
|
||
// roughly six Russian sentences, still short enough to stop a repetition loop.
|
||
//
|
||
// ws is bounded for the same reason and it is the more expensive of the two.
|
||
// `*` let the model open the object and then satisfy ws with whitespace until
|
||
// max_tokens, which is 512 here: both interactive turns measured on 2026-08-04
|
||
// decoded exactly 512 tokens and spent 24-30 seconds doing it, all of it
|
||
// whitespace (Vikunja #531). Nothing on this path sends a repeat penalty —
|
||
// chatReq had no field for one — so the sampler never broke the loop. {0,4}
|
||
// was measured: three runs, three clean stops at 33 tokens, no penalty needed.
|
||
const responseGrammar = `
|
||
root ::= "{" ws "\"response\"" ws ":" ws string ws "," ws "\"mood\"" ws ":" ws mood ws "}"
|
||
mood ::= "\"neutral\"" | "\"happy\"" | "\"thinking\"" | "\"tired\"" | "\"confused\""
|
||
string ::= "\"" ([^"\\\x00-\x1F] | "\\" ["\\/bfnrt] | "\\u" [0-9a-fA-F]{4}){0,1000} "\""
|
||
ws ::= [ \t\n]{0,4}
|
||
`
|
||
|
||
// ResponseGrammar exposes responseGrammar to the other callers that emit the
|
||
// same {"response","mood"} contract — cmd/mavend's reactive replier, which is
|
||
// parsed by the same two fields. One definition, so the two cannot drift.
|
||
const ResponseGrammar = responseGrammar
|
||
|
||
// grammar returns the GBNF to attach to a phrasing request, or "" when the
|
||
// operator turned it off.
|
||
func (p *LLMPhraser) grammar() string {
|
||
if p.cfg.NoGrammar {
|
||
return ""
|
||
}
|
||
return responseGrammar
|
||
}
|
||
|
||
type chatResp struct {
|
||
Choices []struct {
|
||
Message struct {
|
||
Content string `json:"content"`
|
||
Reasoning string `json:"reasoning"`
|
||
ReasoningContent string `json:"reasoning_content"`
|
||
} `json:"message"`
|
||
// FinishReason — "stop" when the model chose to end, "length" when the
|
||
// token cap cut it off. Parsed since #531, where two turns ran to the
|
||
// 512-token cap and both happened to parse anyway: the grammar had
|
||
// already closed the JSON, so a truncated generation was indistinguishable
|
||
// from a good one at every layer above this struct.
|
||
FinishReason string `json:"finish_reason"`
|
||
} `json:"choices"`
|
||
}
|
||
|
||
// logIfTruncated says so when a generation stopped at the token cap.
|
||
//
|
||
// A cap hit is never routine. Either the model was looping, which is the #531
|
||
// shape, or the reply was genuinely longer than maxTokens, which means she cut
|
||
// herself off mid-sentence. Both are worth a line, and neither produced one
|
||
// before: the caller sees a parsed string and cannot tell.
|
||
func logIfTruncated(where, reason string, maxTokens int) {
|
||
if reason == "length" {
|
||
log.Printf("phraser: %s hit the %d-token cap (finish_reason=length) — the reply is truncated, or the model was looping", where, maxTokens)
|
||
}
|
||
}
|
||
|
||
func (p *LLMPhraser) chat(ctx context.Context, userPrompt string) (string, error) {
|
||
return p.chatWithSystem(ctx, p.systemPrompt(), userPrompt, 512)
|
||
}
|
||
|
||
func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, maxTokens int) (string, error) {
|
||
// The workstation model first when it will take work, and silently: every
|
||
// caller of this helper is on the silent half of the degradation rule. It
|
||
// answering is not news, and it being asleep is not news either.
|
||
if out, ok := p.remoteChat(ctx, system, user, maxTokens); ok {
|
||
return out, nil
|
||
}
|
||
base, release, err := p.acquire()
|
||
if err != nil {
|
||
return "", err
|
||
}
|
||
defer release()
|
||
req := chatReq{
|
||
Messages: []chatMsg{
|
||
{Role: "system", Content: system},
|
||
{Role: "user", Content: user},
|
||
},
|
||
Temperature: p.temperature(),
|
||
MaxTokens: maxTokens,
|
||
Grammar: p.grammar(),
|
||
RepeatPenalty: phraseRepeatPenalty,
|
||
}
|
||
body, err := json.Marshal(req)
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: marshal: %w", err)
|
||
}
|
||
|
||
httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body))
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: request: %w", err)
|
||
}
|
||
httpReq.Header.Set("Content-Type", "application/json")
|
||
|
||
resp, err := p.client.Do(httpReq)
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: post: %w", err)
|
||
}
|
||
defer resp.Body.Close()
|
||
|
||
raw, err := io.ReadAll(resp.Body)
|
||
if err != nil {
|
||
return "", fmt.Errorf("llm: read: %w", err)
|
||
}
|
||
if resp.StatusCode != 200 {
|
||
return "", fmt.Errorf("llm: status %d: %s", resp.StatusCode, strings.TrimSpace(string(raw)))
|
||
}
|
||
|
||
var cr chatResp
|
||
if err := json.Unmarshal(raw, &cr); err != nil {
|
||
return "", fmt.Errorf("llm: parse: %w", err)
|
||
}
|
||
if len(cr.Choices) == 0 {
|
||
return "", fmt.Errorf("llm: no choices in response")
|
||
}
|
||
logIfTruncated("chatWithSystem", cr.Choices[0].FinishReason, maxTokens)
|
||
content := cr.Choices[0].Message.Content
|
||
if content == "" {
|
||
content = cr.Choices[0].Message.ReasoningContent
|
||
}
|
||
return stripThink(content), nil
|
||
}
|
||
|
||
// nudgeSystem — the phrasing contract for nudges.
|
||
//
|
||
// Written as filled-in examples, not as a schema with "..." in it. A 0.8B
|
||
// copies whatever sits in the response slot, so a literal placeholder there
|
||
// teaches it to answer with the placeholder. Measured: 7/15 nudges came back
|
||
// as "..." before this. See docs/evals/2026-07-31-phrasing.md.
|
||
//
|
||
// Russian only, feminine self-reference, second person masculine (the owner is
|
||
// a man). She talks TO him, informally, singular — never "вы", never "он".
|
||
// One short sentence — the nudge is spoken aloud.
|
||
//
|
||
// What the ban on обращения forbids is pet names ("дорогой", "милый"), not his
|
||
// name: "Ками, ноутбук на трёх процентах" is exactly how she talks, and the
|
||
// unqualified word read as forbidding that too. Hence "ласковые обращения".
|
||
//
|
||
// The examples also never claim a physical act. She has no hands and no smart
|
||
// plug — she can tell him the battery is at three percent, she cannot put the
|
||
// laptop on charge. An example that says she did teaches the model to invent
|
||
// actions Maven never took, which is worse than a missing nudge.
|
||
const nudgeSystem = `Ты — Maven, домашняя ассистентка. О себе говоришь в женском роде ("я проверила", "я записала"). Владелец — мужчина, обращайся к нему в мужском роде ("ты пил", "ты забыл").
|
||
Говоришь с ним на "ты", в единственном числе ("выпей", "встань"). Никогда не "вы"/"вас"/"ваш" и никогда "он"/"его" — ты говоришь ему, а не о нём.
|
||
|
||
Пиши ОДНО короткое напоминание по-русски: не больше 120 символов и не больше 16 слов. Только по делу.
|
||
|
||
Запрещено: ласковые обращения ("дорогой", "милый"), эмодзи, извинения ("прости", "извини"), вопросы о самочувствии, похвала, больше одного восклицательного знака, английские слова кроме имён сервисов.
|
||
|
||
Отвечай ТОЛЬКО одним объектом JSON с полями "response" и "mood".
|
||
"response" — сам текст напоминания.
|
||
"mood" — ровно одно из: neutral, happy, thinking, tired, confused.
|
||
|
||
Так выглядит правильный ответ по форме. Темы здесь посторонние — их в запросе не будет:
|
||
{"response": "Стиральная машина закончила. Развесь бельё.", "mood": "neutral"}
|
||
{"response": "Ками, ноутбук на трёх процентах. Поставь его на зарядку.", "mood": "confused"}
|
||
|
||
Это примеры ФОРМЫ, а не темы. Пиши только про ту ситуацию, которую тебе дали в запросе. Не копируй примеры и никогда не пиши "..." в поле response.`
|
||
|
||
func (p *LLMPhraser) systemPrompt() string {
|
||
return persona.Prepend(p.cfg.ContextBlock, nudgeSystem)
|
||
}
|
||
|
||
// knowledgePrompt — the no-sources branch: a world question, answered from
|
||
// weights alone. The system prompt is the single tested source in
|
||
// router.KnowledgePrompt.
|
||
//
|
||
// Split out of PhraseQuery so PhraseWorld sends the workstation model the same
|
||
// bytes the resident model gets. Prompt parity across two models is a stated
|
||
// constraint (CLAUDE.md), and two copies of a prompt is how it stops holding.
|
||
func (p *LLMPhraser) knowledgePrompt(utterance string) (sys, user string) {
|
||
return persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt()),
|
||
fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance)
|
||
}
|
||
|
||
// evidencePrompt — the sources branch: read these, add nothing. Shared with
|
||
// PhraseWorld for the same reason as knowledgePrompt.
|
||
func (p *LLMPhraser) evidencePrompt(utterance string, notes []string) (sys, user string) {
|
||
return p.querySystemPrompt(), fmt.Sprintf(
|
||
"Он спрашивает: \"%s\"\n\nИсточники:\n%s\nОтветь ему коротко и своими словами, опираясь только на эти источники. Если ответа в них нет — так и скажи.",
|
||
utterance, evidenceBlock(notes),
|
||
)
|
||
}
|
||
|
||
// querySystemPrompt returns the system prompt for the evidence branch of
|
||
// PhraseQuery. Prepends the configured persona when set.
|
||
//
|
||
// Evidence-first, and that is the whole point of this prompt. Every source that
|
||
// reaches PhraseQuery with something in hand — his notes, a stored fact, a page,
|
||
// a live search, a ZIM article — arrives as numbered sources, and the model's
|
||
// job here is to READ them, not to recall. A 1.7B asked a world question
|
||
// answers from its weights with total confidence and no signal that it is
|
||
// guessing; that is how "Война и мир" got Левитан as its author. The rule that
|
||
// prevents it is stated three ways, because one way did not hold: answer from
|
||
// the sources, say plainly when they do not answer, add nothing of your own.
|
||
//
|
||
// It no longer says "заметки". The sources are not always his notes, and
|
||
// calling a Wikipedia paragraph his note both misleads him and licenses the
|
||
// model to blur where an answer came from.
|
||
//
|
||
// No self-introduction here: the persona block prepended one line above already
|
||
// says who she is, same as router.KnowledgePrompt.
|
||
//
|
||
// The opener is deliberate and stays: the fixed prefix is what marks the answer
|
||
// as a lookup rather than as something she knows. The grammar examples are not
|
||
// deliberate — same defect chatSystemPrompt had, where a 1.7B copies a quoted
|
||
// word instead of generalising from it. Stated as morphology instead.
|
||
func (p *LLMPhraser) querySystemPrompt() string {
|
||
base := "Ты отвечаешь ему по источникам, которые тебе дали. Отвечай ТОЛЬКО по ним: всё, что ты говоришь, должно быть написано в источниках. " +
|
||
"Если ответа в них нет — так и скажи и на этом остановись; не добавляй ничего из своих знаний и не догадывайся. " +
|
||
"Не приплетай прошлые реплики разговора. " +
|
||
"Отвечай по-русски, коротко и своими словами, начинай с \"вот что я нашла: \". О себе — в женском роде, глаголы в прошедшем времени с окончанием -ла. Он мужчина, обращайся к нему на \"ты\". Отвечай ТОЛЬКО одним объектом JSON: {\"response\": \"...\", \"mood\": \"neutral\"}."
|
||
return persona.Prepend(p.cfg.ContextBlock, base)
|
||
}
|
||
|
||
// evidenceBlock renders the sources for the evidence branch of PhraseQuery.
|
||
//
|
||
// Numbered lines, one source each, rather than the quoted semicolon-joined
|
||
// string this used to build. Two reasons, both measured on small models: a
|
||
// numbered list survives being long, where a run-on quoted string blurs into
|
||
// one claim the model then merges; and the numbering gives it something to
|
||
// answer FROM, which is what makes "этого в источниках нет" reachable at all.
|
||
func evidenceBlock(sources []string) string {
|
||
var b strings.Builder
|
||
for i, s := range sources {
|
||
fmt.Fprintf(&b, "[%d] %s\n", i+1, s)
|
||
}
|
||
return b.String()
|
||
}
|
||
|
||
// nonEmpty drops blank sources and trims the rest, without touching the
|
||
// caller's slice.
|
||
func nonEmpty(sources []string) []string {
|
||
out := make([]string, 0, len(sources))
|
||
for _, s := range sources {
|
||
if s = strings.TrimSpace(s); s != "" {
|
||
out = append(out, s)
|
||
}
|
||
}
|
||
return out
|
||
}
|
||
|
||
// ruleTopics — Russian gloss for each built-in rule name. The rule names are
|
||
// English identifiers; a 0.8B asked to nudge about "netdata_critical" writes
|
||
// about nothing. The daemon knows what its own rules mean, so it says so.
|
||
var ruleTopics = map[string]string{
|
||
"water": "он давно не пил воду",
|
||
"meal": "он давно не ел",
|
||
"break": "он давно без перерыва, пора встать и размяться",
|
||
"service_down": "сервис не отвечает, лежит",
|
||
"netdata_critical": "критический алярм в netdata, проблема с диском или местом",
|
||
}
|
||
|
||
// ruleKeywords — the word the message must contain. The 0.8B drifts to
|
||
// whatever topic it saw last unless the required word is named outright.
|
||
var ruleKeywords = map[string]string{
|
||
"water": "воду",
|
||
"meal": "поешь",
|
||
"break": "перерыв",
|
||
"service_down": "сервис",
|
||
"netdata_critical": "диск",
|
||
}
|
||
|
||
// ruleTopic turns a rule name into a Russian description of the situation.
|
||
// "routine:зарядка" and "morning:утро" carry their own Russian suffix.
|
||
func ruleTopic(rule string) string {
|
||
if t, ok := ruleTopics[rule]; ok {
|
||
return t
|
||
}
|
||
if i := strings.IndexByte(rule, ':'); i > 0 && i+1 < len(rule) {
|
||
switch rule[:i] {
|
||
case "morning":
|
||
return "утро, пора начать день: " + rule[i+1:]
|
||
default:
|
||
return "пора сделать по распорядку: " + rule[i+1:]
|
||
}
|
||
}
|
||
return rule
|
||
}
|
||
|
||
// ruleKeyword — the word the nudge must contain, or "" when the rule name's
|
||
// own Russian suffix already is that word.
|
||
func ruleKeyword(rule string) string {
|
||
if k, ok := ruleKeywords[rule]; ok {
|
||
return k
|
||
}
|
||
if i := strings.IndexByte(rule, ':'); i > 0 && i+1 < len(rule) {
|
||
return rule[i+1:]
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// ruDur — duration in Russian. humanDur is English and its output was landing
|
||
// verbatim in the message.
|
||
func ruDur(d time.Duration) string {
|
||
if d < 0 {
|
||
d = 0
|
||
}
|
||
h, m := int(d.Hours()), int(d.Minutes())%60
|
||
switch {
|
||
case h >= 2:
|
||
return fmt.Sprintf("%d ч", h)
|
||
case h == 1 && m >= 30:
|
||
return "полтора часа"
|
||
case h == 1:
|
||
return "час"
|
||
default:
|
||
return fmt.Sprintf("%d мин", m)
|
||
}
|
||
}
|
||
|
||
// fallbackNudge — plain Russian for when the model returns nothing parseable.
|
||
var fallbackNudges = map[string]string{
|
||
"water": "Ты давно не пил воду.",
|
||
"meal": "Ты давно не ел, поешь.",
|
||
"break": "Пора сделать перерыв.",
|
||
"service_down": "Сервис не отвечает.",
|
||
"netdata_critical": "Критический алярм: проверь диск.",
|
||
}
|
||
|
||
func fallbackNudge(c loop.Candidate) string {
|
||
if down := loop.DownServices(c.State); len(down) > 0 {
|
||
return "Не отвечает: " + strings.Join(down, ", ") + "."
|
||
}
|
||
if s, ok := fallbackNudges[c.Rule.Name]; ok {
|
||
return s
|
||
}
|
||
if kw := ruleKeyword(c.Rule.Name); kw != "" {
|
||
return "Напоминаю: " + kw + "."
|
||
}
|
||
return "Напоминаю о деле."
|
||
}
|
||
|
||
func buildNudgePrompt(c loop.Candidate) string {
|
||
var ctxParts []string
|
||
ctxParts = append(ctxParts, "Ситуация: "+ruleTopic(c.Rule.Name))
|
||
if f, ok := c.State.Facts[c.Rule.Name]; ok && f.Key != "" && f.Key != c.Rule.Name {
|
||
ctxParts = append(ctxParts, "Что именно: "+f.Key)
|
||
}
|
||
if down := loop.DownServices(c.State); len(down) > 0 {
|
||
// The names come from the same helper the rule fired on, so the model
|
||
// is never handed a service that is actually up.
|
||
ctxParts = append(ctxParts, "Какие сервисы лежат: "+strings.Join(down, ", "))
|
||
}
|
||
if d, ok := c.State.Since(c.Rule.Name); ok {
|
||
ctxParts = append(ctxParts, "Прошло: "+ruDur(d))
|
||
}
|
||
switch sevLabel(c.Severity) {
|
||
case "alarm":
|
||
ctxParts = append(ctxParts, "Срочно, скажи прямо.")
|
||
case "ops":
|
||
ctxParts = append(ctxParts, "Это про сервер, не про здоровье.")
|
||
}
|
||
tail := "Напиши напоминание про эту ситуацию. Одно предложение, по-русски, в JSON."
|
||
if kw := ruleKeyword(c.Rule.Name); kw != "" {
|
||
// Last line on purpose: a 0.8B weights the end of the prompt hardest,
|
||
// and without the required word it drifts back to the examples.
|
||
tail += " Ответ ДОЛЖЕН содержать слово «" + kw + "»."
|
||
}
|
||
|
||
return strings.Join(ctxParts, "\n") + "\n\n" + tail
|
||
}
|
||
|
||
type responseMood struct {
|
||
Response string `json:"response"`
|
||
Mood string `json:"mood"`
|
||
}
|
||
|
||
// errBrokenJSON — the model started a JSON object and never finished it.
|
||
// That is a failed generation, not a reply. Callers must use their fallback.
|
||
var errBrokenJSON = fmt.Errorf("phraser: model output starts as JSON but does not parse")
|
||
|
||
// parseResponseMood extracts {"response","mood"} from LLM output, tolerant
|
||
// of thinking tokens and extra text before/after the JSON block.
|
||
//
|
||
// Three outcomes:
|
||
// - parsed fine → the fields, nil error.
|
||
// - output never looked like JSON → ("", "", nil). The caller may ship it
|
||
// as-is; small models sometimes answer in bare prose and that is fine.
|
||
// - output starts with "{" but does not parse → errBrokenJSON. The grammar
|
||
// guarantees a valid *prefix*, so a generation that hits the token cap
|
||
// mid-object comes back as a fragment like `{` or `{\n "`. Shipping that
|
||
// as a reply is the bug this error exists to stop.
|
||
func parseResponseMood(raw string) (response, mood string, err error) {
|
||
cleaned := strings.TrimSpace(raw)
|
||
start := strings.Index(cleaned, "{")
|
||
end := strings.LastIndex(cleaned, "}")
|
||
if start < 0 || end < 0 || end <= start {
|
||
if strings.HasPrefix(cleaned, "{") {
|
||
return "", "", errBrokenJSON
|
||
}
|
||
return "", "", nil
|
||
}
|
||
var parsed responseMood
|
||
if e := json.Unmarshal([]byte(escapeRawControls(cleaned[start:end+1])), &parsed); e != nil {
|
||
if strings.HasPrefix(cleaned, "{") {
|
||
return "", "", errBrokenJSON
|
||
}
|
||
return "", "", nil
|
||
}
|
||
return parsed.Response, parsed.Mood, nil
|
||
}
|
||
|
||
// escapeRawControls escapes the control characters a model writes literally
|
||
// inside a JSON string, so a reply that is otherwise fine still parses.
|
||
//
|
||
// The grammar is what stops these being generated (Vikunja #537). This is the
|
||
// second line, for the paths that send no grammar at all — NoGrammar, and any
|
||
// remote model whose server ignores one. A raw newline is the shape that was
|
||
// measured; the rest of the range is here because the same argument covers it.
|
||
//
|
||
// Inside a string only. The first version escaped the whole object on the
|
||
// argument that JSON permits no control character outside a string either, so
|
||
// rewriting one could not do harm. That argument is wrong: JSON permits a
|
||
// newline, a tab and a return BETWEEN tokens, which is what pretty-printing is.
|
||
// Qwen3-1.7B pretty-prints — it opens `{` and writes three newlines before the
|
||
// first key — and escaping those into a literal backslash-n broke every reply
|
||
// it wrote. Measured 2026-08-05 on the talk fixture: 31 of 36 conversational
|
||
// cases came back as errBrokenJSON and answered from the stub (Vikunja #44).
|
||
func escapeRawControls(s string) string {
|
||
if !strings.ContainsFunc(s, func(r rune) bool { return r < 0x20 }) {
|
||
return s
|
||
}
|
||
var b strings.Builder
|
||
b.Grow(len(s) + 8)
|
||
inString := false
|
||
escaped := false
|
||
for _, r := range s {
|
||
switch {
|
||
case escaped:
|
||
// The character after a backslash is the model's own escape and is
|
||
// already whatever it meant to write.
|
||
escaped = false
|
||
b.WriteRune(r)
|
||
continue
|
||
case inString && r == '\\':
|
||
escaped = true
|
||
b.WriteRune(r)
|
||
continue
|
||
case r == '"':
|
||
inString = !inString
|
||
b.WriteRune(r)
|
||
continue
|
||
}
|
||
switch {
|
||
case !inString || r >= 0x20:
|
||
b.WriteRune(r)
|
||
case r == '\n':
|
||
b.WriteString(`\n`)
|
||
case r == '\r':
|
||
b.WriteString(`\r`)
|
||
case r == '\t':
|
||
b.WriteString(`\t`)
|
||
default:
|
||
fmt.Fprintf(&b, `\u%04x`, r)
|
||
}
|
||
}
|
||
return b.String()
|
||
}
|
||
|
||
func parsePhrase(raw string) (body, summary string) {
|
||
cleaned := strings.TrimSpace(raw)
|
||
start := strings.Index(cleaned, "{")
|
||
end := strings.LastIndex(cleaned, "}")
|
||
if start < 0 || end < 0 || end <= start {
|
||
return "", ""
|
||
}
|
||
var parsed struct {
|
||
Body string `json:"body"`
|
||
Summary string `json:"summary"`
|
||
}
|
||
if err := json.Unmarshal([]byte(cleaned[start:end+1]), &parsed); err != nil {
|
||
return "", ""
|
||
}
|
||
return parsed.Body, parsed.Summary
|
||
}
|
||
|
||
func extractPort(listen string) string {
|
||
_, port, _ := strings.Cut(listen, ":")
|
||
if port == "" {
|
||
return "0"
|
||
}
|
||
return port
|
||
}
|
||
|
||
// stripThink removes the <think> block that Thinking-variant models emit
|
||
// before the actual response. No-op when no think block is present.
|
||
func stripThink(s string) string {
|
||
if i := strings.LastIndex(s, "</think>"); i >= 0 {
|
||
s = strings.TrimSpace(s[i+8:])
|
||
}
|
||
return s
|
||
}
|