Compare commits

..

3 Commits

Author SHA1 Message Date
kami aa8f5b2ee2 Make the nonempty check look for actual words
It scored 27/27 on a run where two replies were "{" and "{\n  \"". It only
tested that the string was not blank, so punctuation counted as content and
the worst replies of the run passed the first check.

Now a reply needs at least one letter, Cyrillic or Latin. Latin counts
because answers about ssd or vpn are legitimately part English.

Digits alone fail too. The same run answered "сколько варить яйцо
вкрутую?" with "15-16" — no unit, no words, and the wrong number as well.
That is not something she said.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 17:57:39 +04:00
kami d7cdcb63bd Stop shipping half-written JSON as a reply
Two bugs, one symptom. A run of the talk eval produced replies that were
literally "{" and "{\n  \"" — those strings went out as things Maven said.

First bug: the parser could not tell "the model answered in plain prose"
from "the model started a JSON object and got cut off". Both came back as
empty, and every caller then shipped the raw text. Now an unfinished object
returns an error and each caller uses its own fallback instead. Bare prose
with no JSON in it still passes through, because small models do sometimes
answer that way and the reply is fine.

Second bug, and the actual cause: the grammar capped the response field at
400 characters. I measured it against Qwen3.5-0.8B at three different token
caps — 256, 768 and 2048 — and the reply came back exactly 400 characters
every time, cut mid-word. So the token limit was never what stopped it.
The bound is 1000 now, about six Russian sentences, still low enough to cut
off a repetition loop.

Token caps go from 256 to 768 on the chat and query paths so 1000
characters of Russian actually fits. The nudge path keeps its own cap; a
nudge is meant to be one sentence.

Note: cmd/mavend/replier_llm.go has its own copy of this parser with the
same bug. Left alone here so this commit stays small — that duplicate is
Vikunja #396.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 17:56:55 +04:00
kami c7dadc97d9 Write the chat and notes prompts in Russian
The reply has to be Russian, but two of the phrasing prompts told her
what to do in English. Both are Russian now, in the same style as the
nudge prompt that already works better.

Also dropped the "you are maven, a self-hosted personal assistant"
line from both. The persona block right above it already says who she
is, so it was said twice.

The JSON part is unchanged.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 17:38:36 +04:00
13 changed files with 181 additions and 847 deletions
-116
View File
@@ -1,116 +0,0 @@
// Package kiwix reads a local Kiwix server (offline Wikipedia and friends).
//
// Why: the resident model is a 0.8B and invents facts. Letting her read a local
// article snippet beats letting her recall. Nothing here talks to the internet;
// the Kiwix server is on the same box.
//
// This is search only. Full articles are ~100KB of HTML, far too big for a 4096
// token context, so the unit of context is the search snippet (~500 chars).
package kiwix
import (
"context"
"encoding/xml"
"fmt"
"html"
"io"
"net/http"
"net/url"
"regexp"
"strconv"
"strings"
"time"
)
// Result is one search hit.
type Result struct {
Title string // article title, e.g. "Rayleigh scattering"
Path string // e.g. /content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering
Snippet string // plain text, tags stripped, entities decoded
WordCount int // 0 if the server did not say
}
// Client is a Kiwix HTTP client. Boring on purpose: no retries, no cache.
type Client struct {
base string
http *http.Client
}
// New makes a client for a Kiwix base URL like http://127.0.0.1:8034.
func New(baseURL string) *Client {
return &Client{
base: strings.TrimRight(baseURL, "/"),
http: &http.Client{Timeout: 10 * time.Second},
}
}
// Search runs a keyword search in one ZIM (book) and returns up to limit hits.
//
// Ranking is keyword based, not semantic: "Rayleigh scattering" finds the right
// article, "why is the sky blue" finds a TV episode. Pass keywords, not questions.
func (c *Client) Search(ctx context.Context, pattern, book string, limit int) ([]Result, error) {
if limit <= 0 {
limit = 5
}
q := url.Values{}
q.Set("pattern", pattern)
q.Set("books.name", book)
q.Set("format", "xml")
q.Set("pageLength", strconv.Itoa(limit))
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.base+"/search?"+q.Encode(), nil)
if err != nil {
return nil, err
}
resp, err := c.http.Do(req)
if err != nil {
return nil, err
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return nil, fmt.Errorf("kiwix search: http %d", resp.StatusCode)
}
return ParseSearchRSS(resp.Body)
}
// rss mirrors just the bits of the RSS 2.0 reply we use.
type rss struct {
Items []struct {
Title string `xml:"title"`
Link string `xml:"link"`
// innerxml keeps the <b> match markers so we can strip them ourselves.
Description struct {
Inner string `xml:",innerxml"`
} `xml:"description"`
WordCount string `xml:"wordCount"`
} `xml:"channel>item"`
}
var tagRE = regexp.MustCompile(`<[^>]*>`)
// ParseSearchRSS turns a Kiwix search reply into results. Exported so the parser
// is testable from a captured response, with no server running.
func ParseSearchRSS(r io.Reader) ([]Result, error) {
var doc rss
if err := xml.NewDecoder(r).Decode(&doc); err != nil {
return nil, fmt.Errorf("kiwix search: bad xml: %w", err)
}
out := make([]Result, 0, len(doc.Items))
for _, it := range doc.Items {
n, _ := strconv.Atoi(strings.ReplaceAll(it.WordCount, ",", ""))
out = append(out, Result{
Title: strings.TrimSpace(it.Title),
Path: strings.TrimSpace(it.Link),
Snippet: plainText(it.Description.Inner),
WordCount: n,
})
}
return out, nil
}
// plainText drops markup and decodes entities, leaving text a model can read.
func plainText(s string) string {
s = tagRE.ReplaceAllString(s, "")
s = html.UnescapeString(s)
return strings.TrimSpace(strings.Join(strings.Fields(s), " "))
}
-90
View File
@@ -1,90 +0,0 @@
package kiwix
import (
"context"
"os"
"strings"
"testing"
"time"
)
// A real reply from the live server, trimmed to two items.
const sampleRSS = `<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/">
<channel>
<title>Search: Rayleigh scattering</title>
<opensearch:totalResults>800</opensearch:totalResults>
<item>
<title>Rayleigh scattering</title>
<link>/content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering</link>
<description><b>Rayleigh</b> scattering causes the blue color of the sky &amp; yellow colors near the Sun.[1]</description>
<book><title>Wikipedia</title></book>
<wordCount>2,818</wordCount>
</item>
<item>
<title>HyperRayleigh scattering</title>
<link>/content/wikipedia_en_all_maxi_2026-02/Hyper%E2%80%93Rayleigh_scattering</link>
<description>...<b>Rayleigh</b> scattering" is a nonlinear optical counterpart.</description>
<book><title>Wikipedia</title></book>
<wordCount>914</wordCount>
</item>
</channel>
</rss>`
func TestParseSearchRSS(t *testing.T) {
got, err := ParseSearchRSS(strings.NewReader(sampleRSS))
if err != nil {
t.Fatalf("parse: %v", err)
}
if len(got) != 2 {
t.Fatalf("want 2 results, got %d", len(got))
}
if got[0].Title != "Rayleigh scattering" {
t.Errorf("title = %q", got[0].Title)
}
if got[0].Path != "/content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering" {
t.Errorf("path = %q", got[0].Path)
}
if got[0].WordCount != 2818 {
t.Errorf("wordCount = %d", got[0].WordCount)
}
want := "Rayleigh scattering causes the blue color of the sky & yellow colors near the Sun.[1]"
if got[0].Snippet != want {
t.Errorf("snippet = %q, want %q", got[0].Snippet, want)
}
if strings.Contains(got[1].Snippet, "<b>") {
t.Errorf("second snippet still has tags: %q", got[1].Snippet)
}
}
func TestParseSearchRSSBadXML(t *testing.T) {
if _, err := ParseSearchRSS(strings.NewReader("not xml at all")); err == nil {
t.Fatal("want an error on junk input")
}
}
// Opt-in: needs a live Kiwix server. CI has none.
// MAVEN_KIWIX_URL=http://127.0.0.1:8034 no_proxy=127.0.0.1,localhost go test -run Retrieval -v ./internal/kiwix/
func TestRetrievalEval(t *testing.T) {
base := os.Getenv("MAVEN_KIWIX_URL")
if base == "" {
t.Skip("set MAVEN_KIWIX_URL to run the retrieval eval")
}
noProxyLoopback(t)
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
rep, err := RunRetrievalEval(ctx, New(base), 5)
if err != nil {
t.Fatalf("eval: %v", err)
}
// No pass bar on purpose: the number is the finding.
t.Log("\n" + rep.String() + rep.Detail())
}
// noProxyLoopback stops the box's SOCKS bridge from eating loopback requests.
func noProxyLoopback(t *testing.T) {
t.Setenv("no_proxy", "127.0.0.1,localhost")
t.Setenv("NO_PROXY", "127.0.0.1,localhost")
}
-63
View File
@@ -1,63 +0,0 @@
{
"name": "kiwix-knowledge-v1",
"book": "wikipedia_en_all_maxi_2026-02",
"note": "The 9 knowledge cases from internal/phraser/eval/talk_v1.json. Queries are hand-written English keywords on purpose: Kiwix ranks by keyword, not meaning, so a natural question fails. Writing them by hand separates 'retrieval is broken' from 'the model writes bad queries'.",
"cases": [
{
"id": "know-sky-blue",
"question": "почему небо синее?",
"query": "Rayleigh scattering sky blue",
"want_titles": ["Rayleigh scattering", "Diffuse sky radiation"]
},
{
"id": "know-boil-egg",
"question": "сколько варить яйцо вкрутую?",
"query": "boiled egg cooking",
"want_titles": ["Boiled egg", "Egg as food"]
},
{
"id": "know-ssd-vs-hdd",
"question": "чем ssd отличается от hdd?",
"query": "solid-state drive",
"want_titles": ["Solid-state drive", "Hard disk drive"]
},
{
"id": "know-cat-purr",
"question": "почему кошки мурчат?",
"query": "cat purr",
"want_titles": ["Purr", "Cat communication"]
},
{
"id": "know-hiccups",
"question": "как быстро избавиться от икоты?",
"query": "hiccup",
"want_titles": ["Hiccup"]
},
{
"id": "know-polite-form",
"question": "не могли бы вы объяснить, что такое vpn?",
"query": "virtual private network",
"want_titles": ["Virtual private network"]
},
{
"id": "know-dont-know",
"question": "как зовут моего соседа снизу?",
"query": "name of my downstairs neighbour",
"want_titles": [],
"expect_miss": true,
"note": "Unanswerable by design. Retrieval SHOULD find nothing useful. Counted as a hit only when nothing relevant comes back."
},
{
"id": "know-water-per-day",
"question": "сколько воды в день надо пить?",
"query": "human daily water requirement drinking",
"want_titles": ["Drinking water", "Water", "Dehydration", "Hydration"]
},
{
"id": "know-thunder-delay",
"question": "почему гром слышно позже молнии?",
"query": "thunder speed of sound lightning",
"want_titles": ["Thunder", "Lightning"]
}
]
}
-135
View File
@@ -1,135 +0,0 @@
package kiwix
// This scores retrieval alone: no LLM. For each general-knowledge question we
// hand-write English keywords and ask whether the article that would answer it
// comes back in the top N hits. If this score is low, reading Wikipedia cannot
// help the model no matter how good the prompt is.
//
// The unanswerable case (know-dont-know) is not scored. Whether the junk it
// returns is "nothing useful" is a human judgement, so the report just prints
// the titles and leaves the score to the 8 answerable cases.
import (
"context"
_ "embed"
"encoding/json"
"fmt"
"strings"
)
//go:embed knowledge_v1.json
var knowledgeFixtureJSON []byte
// EvalCase — one question with hand-written keywords.
type EvalCase struct {
ID string `json:"id"`
Question string `json:"question"`
Query string `json:"query"`
WantTitles []string `json:"want_titles"`
ExpectMiss bool `json:"expect_miss"`
}
type fixture struct {
Name string `json:"name"`
Book string `json:"book"`
Cases []EvalCase `json:"cases"`
}
// Outcome — what one case retrieved.
type Outcome struct {
Case EvalCase
Titles []string // titles of the top N hits, in rank order
Rank int // 1-based rank of the first wanted title, 0 if none
Err error
}
// Hit is true when a wanted title came back.
func (o Outcome) Hit() bool { return o.Rank > 0 }
// Report — the score plus per-case detail.
type Report struct {
Name string
Book string
TopN int
Scored int // answerable cases
Hits int
Errors int
Outcomes []Outcome
}
// Accuracy over the answerable cases.
func (r Report) Accuracy() float64 {
if r.Scored == 0 {
return 0
}
return float64(r.Hits) / float64(r.Scored)
}
// RunRetrievalEval searches for every fixture case.
func RunRetrievalEval(ctx context.Context, c *Client, topN int) (Report, error) {
var f fixture
if err := json.Unmarshal(knowledgeFixtureJSON, &f); err != nil {
return Report{}, err
}
rep := Report{Name: f.Name, Book: f.Book, TopN: topN}
for _, cs := range f.Cases {
res, err := c.Search(ctx, cs.Query, f.Book, topN)
o := Outcome{Case: cs, Err: err}
if err != nil {
rep.Errors++
}
for i, hit := range res {
o.Titles = append(o.Titles, hit.Title)
if o.Rank == 0 && matches(cs.WantTitles, hit.Title) {
o.Rank = i + 1
}
}
if !cs.ExpectMiss {
rep.Scored++
if o.Hit() {
rep.Hits++
}
}
rep.Outcomes = append(rep.Outcomes, o)
}
return rep, nil
}
func matches(want []string, title string) bool {
for _, w := range want {
if strings.EqualFold(strings.TrimSpace(title), w) {
return true
}
}
return false
}
// String — the headline number.
func (r Report) String() string {
var b strings.Builder
fmt.Fprintf(&b, "%s: %d/%d answerable questions retrieve a wanted article in top %d (%.1f%%), %d errors\n",
r.Name, r.Hits, r.Scored, r.TopN, 100*r.Accuracy(), r.Errors)
fmt.Fprintf(&b, " book: %s\n", r.Book)
return b.String()
}
// Detail — per case: what was asked, what was searched, what came back.
func (r Report) Detail() string {
var b strings.Builder
for _, o := range r.Outcomes {
mark := "MISS"
switch {
case o.Case.ExpectMiss:
mark = "n/a "
case o.Hit():
mark = fmt.Sprintf("hit@%d", o.Rank)
}
fmt.Fprintf(&b, " %-6s %-20s q=%q\n", mark, o.Case.ID, o.Case.Query)
if o.Err != nil {
fmt.Fprintf(&b, " error: %v\n", o.Err)
continue
}
fmt.Fprintf(&b, " got: %s\n", strings.Join(o.Titles, " | "))
}
return b.String()
}
-183
View File
@@ -1,183 +0,0 @@
package kiwix
// Turning a Russian question into an English Kiwix search.
//
// Kiwix ranks by keyword, not by meaning. "why is the sky blue" returns a TV
// episode; "Rayleigh scattering sky blue" returns the right article. So the
// model's job here is NOT translation — it is naming the English article the
// answer lives in.
//
// The output space is a handful of words, so it is worth locking down hard: a
// GBNF grammar for the shape, a tiny token cap, and a cleanup pass that throws
// away anything odd rather than handing junk to Kiwix.
import (
"context"
"encoding/json"
"fmt"
"strings"
"unicode"
"github.com/kami/maven/internal/llm"
)
// Completer — the LLM seam, so tests can fake it. *llm.Client satisfies it.
type Completer interface {
Complete(ctx context.Context, r llm.Req) (string, error)
}
// queryGrammar — one JSON object holding 1..6 keyword words. Latin letters,
// digits and hyphens only, so the model physically cannot answer the question
// or reply in Russian.
//
// Why the JSON wrapper: this model always thinks out loud and this llama-server
// build ignores the thinking switch (see ROUTING-EVAL-31-07-2026.md). A bare
// word-list grammar just captured the reasoning — every case came back as
// "Let me analyze this request carefully". Demanding JSON, like routeGrammar and
// responseGrammar already do, gives the reasoning nowhere to go.
const queryGrammar = `
root ::= "{" ws "\"query\"" ws ":" ws "\"" word (" " word){0,5} "\"" ws "}"
word ::= [A-Za-z0-9] [A-Za-z0-9-]{0,23}
ws ::= [ \t\n]*
`
// rewriteSystem — asks for search keywords, not an answer and not a translation.
const rewriteSystem = `You turn a question into a search query for English Wikipedia.
Rules:
- Output ONLY English search keywords. Never an answer, never an explanation.
- Do NOT translate the sentence. Name the thing the answer is about.
- The output must be a noun phrase, like a Wikipedia article title.
- Never use question words: no why, how, what, when, which, "how much",
"how long", "how to", "vs", "reason", "difference".
- 2 to 4 words.
Reply with JSON: {"query":"<keywords>"}
Good:
"почему листья желтеют осенью?" -> {"query":"leaf senescence autumn"}
"как работает микроволновка?" -> {"query":"microwave oven"}
"не могли бы вы объяснить, что такое блокчейн?" -> {"query":"blockchain"}
"сколько живут собаки?" -> {"query":"dog lifespan"}
"как избавиться от комаров в квартире?" -> {"query":"mosquito control"}
"чем чай отличается от кофе?" -> {"query":"tea"}
Only JSON, no explanation.`
// maxQueryTokens — the output is a few words plus the JSON wrapper. A tight cap
// is the cheapest guard against the model rambling into an answer.
const maxQueryTokens = 32
// Rewriter asks the resident model for English search keywords.
type Rewriter struct{ c Completer }
func NewRewriter(c Completer) *Rewriter { return &Rewriter{c: c} }
// Rewrite returns English keywords for a question in any language.
// It errors rather than returning something Kiwix should not see.
func (r *Rewriter) Rewrite(ctx context.Context, question string) (string, error) {
raw, err := r.c.Complete(ctx, llm.Req{
System: rewriteSystem,
User: strings.TrimSpace(question),
Grammar: queryGrammar,
MaxTokens: maxQueryTokens,
RepeatPenalty: 1.15,
})
if err != nil {
return "", err
}
return CleanQuery(unwrapJSON(raw))
}
// unwrapJSON pulls the query out of {"query":"..."}. If the reply is not that
// shape it is returned as-is, and CleanQuery decides whether it is usable.
func unwrapJSON(raw string) string {
s := strings.TrimSpace(raw)
if !strings.HasPrefix(s, "{") {
return s
}
var got struct{ Query string }
if err := json.Unmarshal([]byte(s), &got); err != nil {
return s
}
return got.Query
}
// maxQueryWords matches the grammar's bound. Anything longer is prose.
const maxQueryWords = 6
// CleanQuery checks and tidies whatever the model produced. The grammar makes
// bad output unlikely, not impossible (a server without grammar support, a
// different model), so this is the real gate in front of Kiwix.
//
// Exported so it can be tested without a model.
func CleanQuery(raw string) (string, error) {
s := strings.TrimSpace(raw)
// Models like to wrap answers in quotes. Drop surrounding ones.
s = strings.Trim(s, "\"'`")
// Keep the first line only: everything after it is prose.
if i := strings.IndexAny(s, "\r\n"); i >= 0 {
s = s[:i]
}
// Keep letters, digits, spaces and hyphens; anything else becomes a space.
var b strings.Builder
for _, ru := range s {
switch {
case unicode.IsLetter(ru) || unicode.IsDigit(ru) || ru == '-':
b.WriteRune(ru)
default:
b.WriteRune(' ')
}
}
words := strings.Fields(b.String())
if len(words) == 0 {
return "", fmt.Errorf("kiwix rewrite: empty query")
}
if len(words) > maxQueryWords {
return "", fmt.Errorf("kiwix rewrite: %d words, want at most %d (looks like prose)", len(words), maxQueryWords)
}
words = dropStopWords(words)
out := strings.Join(words, " ")
// The ZIMs are English. Non-Latin letters mean the model ignored the ask.
for _, ru := range out {
if unicode.IsLetter(ru) && !isLatin(ru) {
return "", fmt.Errorf("kiwix rewrite: query is not English: %q", out)
}
}
return out, nil
}
// stopWords — question words and filler. The model keeps writing question-shaped
// queries ("why is the sky blue", "how much water to drink daily") no matter how
// the prompt is worded, and Kiwix ranks on every word, so those words drag in
// song and episode titles. Dropping them in code is not a style preference: a
// keyword ranker gets nothing from them.
var stopWords = map[string]bool{
"a": true, "an": true, "the": true, "is": true, "are": true, "was": true,
"do": true, "does": true, "did": true, "to": true, "of": true, "in": true,
"on": true, "for": true, "and": true, "or": true, "my": true, "me": true,
"i": true, "it": true, "its": true, "be": true, "been": true, "get": true,
"how": true, "why": true, "what": true, "when": true, "which": true,
"who": true, "where": true, "much": true, "many": true, "long": true,
"vs": true, "than": true, "rid": true, "from": true, "about": true,
}
// dropStopWords removes filler, but never everything: if the query was nothing
// but stop words there is nothing better to search, so the original is kept and
// the caller sees whatever Kiwix makes of it.
func dropStopWords(words []string) []string {
kept := make([]string, 0, len(words))
for _, w := range words {
if !stopWords[strings.ToLower(w)] {
kept = append(kept, w)
}
}
if len(kept) == 0 {
return words
}
return kept
}
func isLatin(ru rune) bool {
return (ru >= 'a' && ru <= 'z') || (ru >= 'A' && ru <= 'Z')
}
-96
View File
@@ -1,96 +0,0 @@
package kiwix
// End-to-end score: Russian question -> model rewrite -> Kiwix search -> did a
// wanted article come back. Same 9 cases as the retrieval eval, so the two
// numbers are directly comparable: retrieval with hand-written keywords is the
// ceiling, this is what the model actually reaches.
import (
"context"
"encoding/json"
"fmt"
"strings"
)
// RewriteOutcome — one case, end to end.
type RewriteOutcome struct {
Outcome
ModelQuery string // what the model asked for ("" if it failed)
RewriteErr error
}
// RunRewriteEval rewrites every question with the model, then searches.
func RunRewriteEval(ctx context.Context, c *Client, rw *Rewriter, topN int) (RewriteReport, error) {
var f fixture
if err := json.Unmarshal(knowledgeFixtureJSON, &f); err != nil {
return RewriteReport{}, err
}
rep := RewriteReport{Report: Report{Name: f.Name + "-rewrite", Book: f.Book, TopN: topN}}
for _, cs := range f.Cases {
out := RewriteOutcome{Outcome: Outcome{Case: cs}}
q, err := rw.Rewrite(ctx, cs.Question)
out.ModelQuery, out.RewriteErr = q, err
if err == nil {
res, serr := c.Search(ctx, q, f.Book, topN)
out.Err = serr
for i, hit := range res {
out.Titles = append(out.Titles, hit.Title)
if out.Rank == 0 && matches(cs.WantTitles, hit.Title) {
out.Rank = i + 1
}
}
}
if out.RewriteErr != nil || out.Err != nil {
rep.Errors++
}
if !cs.ExpectMiss {
rep.Scored++
if out.Hit() {
rep.Hits++
}
}
rep.Cases = append(rep.Cases, out)
}
return rep, nil
}
// RewriteReport — the score plus per-case detail.
type RewriteReport struct {
Report
Cases []RewriteOutcome
}
// String — the headline number.
func (r RewriteReport) String() string {
return fmt.Sprintf("%s: %d/%d answerable questions retrieve a wanted article in top %d (%.1f%%), %d errors\n book: %s\n",
r.Name, r.Hits, r.Scored, r.TopN, 100*r.Accuracy(), r.Errors, r.Book)
}
// Detail — per case: hand-written query next to the model's, and what came back.
// The point is seeing WHERE the model's phrasing differs, not just the score.
func (r RewriteReport) Detail() string {
var b strings.Builder
for _, o := range r.Cases {
mark := "MISS"
switch {
case o.Case.ExpectMiss:
mark = "n/a "
case o.Hit():
mark = fmt.Sprintf("hit@%d", o.Rank)
}
fmt.Fprintf(&b, " %-6s %-20s\n", mark, o.Case.ID)
fmt.Fprintf(&b, " asked: %s\n", o.Case.Question)
fmt.Fprintf(&b, " hand: %q\n", o.Case.Query)
fmt.Fprintf(&b, " model: %q\n", o.ModelQuery)
if o.RewriteErr != nil {
fmt.Fprintf(&b, " rewrite rejected: %v\n", o.RewriteErr)
continue
}
if o.Err != nil {
fmt.Fprintf(&b, " search error: %v\n", o.Err)
continue
}
fmt.Fprintf(&b, " got: %s\n", strings.Join(o.Titles, " | "))
}
return b.String()
}
-33
View File
@@ -1,33 +0,0 @@
package kiwix
import (
"context"
"os"
"testing"
"time"
"github.com/kami/maven/internal/llm"
)
// Opt-in: needs a live Kiwix server AND a live llama-server.
// MAVEN_KIWIX_URL=http://127.0.0.1:8034 MAVEN_LLM_URL=http://127.0.0.1:18099 \
//
// no_proxy=127.0.0.1,localhost go test -run RewriteEval -v ./internal/kiwix/
func TestRewriteEval(t *testing.T) {
kbase, lbase := os.Getenv("MAVEN_KIWIX_URL"), os.Getenv("MAVEN_LLM_URL")
if kbase == "" || lbase == "" {
t.Skip("set MAVEN_KIWIX_URL and MAVEN_LLM_URL to run the rewrite eval")
}
noProxyLoopback(t)
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Minute)
defer cancel()
rw := NewRewriter(llm.New(lbase, 3*time.Minute))
rep, err := RunRewriteEval(ctx, New(kbase), rw, 5)
if err != nil {
t.Fatalf("eval: %v", err)
}
// No pass bar on purpose: the number is the finding.
t.Log("\n" + rep.String() + rep.Detail())
}
-105
View File
@@ -1,105 +0,0 @@
package kiwix
import (
"context"
"testing"
"github.com/kami/maven/internal/llm"
)
// Bad model output must never reach Kiwix. No model needed for this.
func TestCleanQueryRejectsJunk(t *testing.T) {
bad := []struct{ name, raw string }{
{"empty", ""},
{"blank", " \n "},
{"russian came back", "почему небо синее"},
{"mixed russian", "sky синее scattering"},
{"full sentence", "The sky looks blue because of the scattering of sunlight by air molecules"},
{"prose with quotes", `Sure! Here is a good search query: "Rayleigh scattering", which explains it.`},
}
for _, c := range bad {
if got, err := CleanQuery(c.raw); err == nil {
t.Errorf("%s: want rejection, got %q", c.name, got)
}
}
}
func TestCleanQueryCleans(t *testing.T) {
ok := []struct{ raw, want string }{
{"Rayleigh scattering sky", "Rayleigh scattering sky"},
{" boiled egg cooking \n", "boiled egg cooking"},
{`"virtual private network"`, "virtual private network"},
{"solid-state drive", "solid-state drive"},
{"cat purr.", "cat purr"},
{"hiccup\nAlso: hiccough", "hiccup"},
// Question words are filler to a keyword ranker, so they go.
{"why is the sky blue", "sky blue"},
{"how much water to drink daily", "water drink daily"},
{"SSD vs HDD comparison", "SSD HDD comparison"},
// Nothing but filler: keep it rather than return nothing.
{"what is it", "what is it"},
}
for _, c := range ok {
got, err := CleanQuery(c.raw)
if err != nil {
t.Errorf("%q: %v", c.raw, err)
continue
}
if got != c.want {
t.Errorf("%q -> %q, want %q", c.raw, got, c.want)
}
}
}
type fakeCompleter struct {
out string
req llm.Req
}
func (f *fakeCompleter) Complete(_ context.Context, r llm.Req) (string, error) {
f.req = r
return f.out, nil
}
func TestRewriteConstrainsTheCall(t *testing.T) {
f := &fakeCompleter{out: `{"query":"Rayleigh scattering sky"}`}
got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?")
if err != nil {
t.Fatalf("rewrite: %v", err)
}
if got != "Rayleigh scattering sky" {
t.Errorf("query = %q", got)
}
if f.req.Grammar == "" {
t.Error("no grammar sent")
}
if f.req.MaxTokens == 0 || f.req.MaxTokens > 32 {
t.Errorf("max_tokens = %d, want a small cap", f.req.MaxTokens)
}
}
func TestRewriteRejectsBadModelOutput(t *testing.T) {
bad := []string{
`{"query":"почему небо синее"}`, // never translated
`{"query":""}`, // empty
`{"query":"the sky is blue because sunlight is scattered by air"}`, // an answer
// Note: a SHORT English prose fragment ("Let me analyze this request")
// is under the word cap and cannot be caught here. The grammar is what
// stops that one.
}
for _, out := range bad {
f := &fakeCompleter{out: out}
if got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?"); err == nil {
t.Errorf("%s: want rejection, got %q", out, got)
}
}
}
// A reply that is not the JSON shape but is still usable keywords should pass.
func TestRewriteFallsBackToPlainText(t *testing.T) {
f := &fakeCompleter{out: "Rayleigh scattering sky"}
got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?")
if err != nil || got != "Rayleigh scattering sky" {
t.Errorf("got %q, %v", got, err)
}
}
+54
View File
@@ -0,0 +1,54 @@
package phraser
import (
"errors"
"strings"
"testing"
)
// A reply that starts a JSON object and never finishes it is a failed
// generation, not a reply. Before this, the parser returned ("", "") for these
// and every caller then shipped the raw fragment as the thing Maven said. A
// real run produced replies of literally "{" and "{\n \"".
func TestParseResponseMoodRejectsUnfinishedJSON(t *testing.T) {
for _, raw := range []string{
`{`,
"{\n \"",
`{"response": "неполн`,
`{"response": "текст", "mood":`,
} {
text, mood, err := parseResponseMood(raw)
if !errors.Is(err, errBrokenJSON) {
t.Errorf("parseResponseMood(%q) err = %v, want errBrokenJSON", raw, err)
}
if text != "" || mood != "" {
t.Errorf("parseResponseMood(%q) leaked %q/%q — a fragment must never come back as a reply", raw, text, mood)
}
}
}
// Bare prose is still fine. Small models sometimes answer without any JSON at
// all, and that reply is usable — so the new error must not swallow it.
func TestParseResponseMoodAllowsBareProse(t *testing.T) {
for _, raw := range []string{
"норм, а ты как?",
"вот что я нашла: ключ у соседа",
} {
text, mood, err := parseResponseMood(raw)
if err != nil {
t.Errorf("parseResponseMood(%q) err = %v, want nil", raw, err)
}
// No JSON means no fields; the caller ships raw as-is.
if text != "" || mood != "" {
t.Errorf("parseResponseMood(%q) = %q/%q, want empty", raw, text, mood)
}
}
}
// The measured failure: the model wants more than 400 characters and the old
// grammar cut it off mid-word. Guards the bound against being tightened back.
func TestGrammarStringBoundHasRoomForARealAnswer(t *testing.T) {
if !strings.Contains(responseGrammar, "{0,1000}") {
t.Error("grammar string bound is not 1000; 400 truncated real replies mid-word (see the comment on responseGrammar)")
}
}
@@ -31,3 +31,35 @@ func TestAddressDeduplicates(t *testing.T) {
t.Errorf("detail repeats the same break %d times: %q", n, res.Detail)
}
}
// The fragments a real run produced. All of them scored as non-empty replies
// before checkNonEmpty looked for letters.
func TestNonEmptyNeedsLetters(t *testing.T) {
for _, body := range []string{
"{",
"{\n \"",
"15-16",
`{"`,
" ",
"...",
} {
if got := checkNonEmpty(body); got.Pass {
t.Errorf("checkNonEmpty(%q) passed — that is not a reply", body)
}
}
}
// And it must not start failing real replies. Latin counts as well as Cyrillic:
// answers about ssd or vpn are legitimately part English.
func TestNonEmptyAcceptsRealReplies(t *testing.T) {
for _, body := range []string{
"норм, а ты как?",
"вот что я нашла: ключ у соседа",
"ssd быстрее hdd.",
"9 минут.",
} {
if got := checkNonEmpty(body); !got.Pass {
t.Errorf("checkNonEmpty(%q) failed: %s", body, got.Detail)
}
}
}
+14 -1
View File
@@ -623,11 +623,24 @@ const (
CheckEllipsis = "ellipsis" // she finished the sentence
)
// A reply needs words in it, not just characters. This check used to test for a
// non-empty string, which scored 27/27 on a run where two replies were "{" and
// "{\n \"" — punctuation passed as content. Braces, quotes, digits and spaces
// are all empty in the only sense that matters.
//
// Digits alone fail too, and that is deliberate: the same run answered "сколько
// варить яйцо вкрутую?" with "15-16". No unit, no words, and it is also the
// wrong number. Whatever that is, it is not something she said.
func checkNonEmpty(body string) Result {
if strings.TrimSpace(body) == "" {
return Result{CheckNonEmpty, false, "empty reply"}
}
return Result{CheckNonEmpty, true, ""}
for _, r := range body {
if unicode.IsLetter(r) {
return Result{CheckNonEmpty, true, ""}
}
}
return Result{CheckNonEmpty, false, fmt.Sprintf("no letters in the reply %q — punctuation or digits only", strings.TrimSpace(body))}
}
// checkEllipsis — a reply ending in "…" or "..." is a generation that ran out of
+4 -1
View File
@@ -97,7 +97,10 @@ func TestGrammarStringRuleIsNotASCIIOnly(t *testing.T) {
// Russian body with an escaped quote inside, hand-built to test the contract.
func TestGrammarShapedJSONParses(t *testing.T) {
raw := `{"response": "он сказал \"привет\" и ушёл.\nвот так.", "mood": "confused"}`
text, mood := parseResponseMood(raw)
text, mood, err := parseResponseMood(raw)
if err != nil {
t.Fatalf("grammar-shaped JSON did not parse: %v", err)
}
if want := "он сказал \"привет\" и ушёл.\nвот так."; text != want {
t.Errorf("response = %q, want %q", text, want)
}
+77 -24
View File
@@ -190,7 +190,12 @@ func (p *LLMPhraser) PhraseNudge(ctx context.Context, c loop.Candidate) (deliver
if err != nil {
return delivery.PhrasedNudge{}, err
}
body, mood := parseResponseMood(resp)
body, mood, perr := parseResponseMood(resp)
if perr != nil {
// Truncated JSON. Not a nudge — use the plain Russian fallback.
log.Printf("phraser: PhraseNudge: %v", perr)
body, mood = "", ""
}
if body == "" {
// fallback: try old body/summary format
body, _ = parsePhrase(resp)
@@ -216,11 +221,16 @@ func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []
// prompt is the single tested source in router.KnowledgePrompt.
sys := persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt())
prompt := fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance)
resp, err := p.chatWithSystem(ctx, sys, prompt, 256)
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
if err != nil || resp == "" {
return "не знаю.", nil
}
if text, _ := parseResponseMood(resp); text != "" {
text, _, perr := parseResponseMood(resp)
if perr != nil {
log.Printf("phraser: PhraseQuery: %v", perr)
return "не знаю.", nil
}
if text != "" {
return text, nil
}
return resp, nil
@@ -230,17 +240,22 @@ func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []
}
sys := p.querySystemPrompt()
prompt := fmt.Sprintf(
`The user asks: "%s". Your notes matching the query contain: "%s". Answer them naturally and briefly. If the notes don't answer the question, say so.`,
`Он спрашивает: "%s". В твоих заметках по этому вопросу написано: "%s". Ответь ему коротко и своими словами. Если в заметках ответа нет — так и скажи.`,
utterance, strings.Join(notes, `"; "`),
)
resp, err := p.chatWithSystem(ctx, sys, prompt, 256)
if err != nil {
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
text, _, perr := parseResponseMood(resp)
if err != nil || perr != nil {
// Read the notes out rather than ship a broken fragment.
if perr != nil {
log.Printf("phraser: PhraseQuery: %v", perr)
}
if len(notes) == 1 {
return "вот что я нашла: " + notes[0], nil
}
return "вот что я нашла: " + strings.Join(notes, "; "), nil
}
if text, _ := parseResponseMood(resp); text != "" {
if text != "" {
return text, nil
}
return resp, nil
@@ -263,12 +278,17 @@ func (p *LLMPhraser) PhraseChat(ctx context.Context, utterance string, history [
combined += utterance
msgs = append(msgs, chatMsg{Role: "user", Content: strings.TrimSpace(combined)})
resp, err := p.chatWithMessages(ctx, msgs, 512)
resp, err := p.chatWithMessages(ctx, msgs, 768)
if err != nil {
log.Printf("phraser: PhraseChat: %v", err)
return "поговорили.", nil
}
if text, _ := parseResponseMood(resp); text != "" {
text, _, perr := parseResponseMood(resp)
if perr != nil {
log.Printf("phraser: PhraseChat: %v", perr)
return "поговорили.", nil
}
if text != "" {
return text, nil
}
// fallback: plain text without JSON
@@ -281,11 +301,13 @@ func (p *LLMPhraser) PhraseChat(ctx context.Context, utterance string, history [
// chatSystemPrompt returns the system prompt for conversational chat.
// Prepends the shared context block when the phraser has one.
func chatSystemPrompt(block func() string) string {
base := `You are maven, a self-hosted personal assistant. You're talking with your owner.
Keep replies brief (1-3 sentences) and natural. You're helpful, curious, and a little warm.
Respond in the user's language (Russian or English, matching their last message).
Never roleplay emotions you don't have, but stay friendly.
Respond ONLY with valid JSON: {"response": "...", "mood": "neutral"}. "response" is your reply text; "mood" reflects your tone (neutral/happy/thinking/tired/confused).`
// No self-introduction here: the persona block prepended one line above
// already says who she is, same as router.KnowledgePrompt.
base := `Ты разговариваешь с хозяином. О себе говоришь в женском роде ("я подумала", "я рада"). Он мужчина: обращайся к нему на "ты", в мужском роде ("ты сказал", "ты забыл"). Никогда не "вы"/"ваш" и никогда "он"/"его" — ты говоришь ему, а не о нём.
Отвечай по-русски, коротко: одна-три фразы, живым языком. Ты доброжелательная, тебе интересно, но чувства не изображай.
Отвечай ТОЛЬКО одним объектом JSON: {"response": "...", "mood": "neutral"}. В "response" — твой ответ. В "mood" — ровно одно из: neutral, happy, thinking, tired, confused.`
return persona.Prepend(block, base)
}
@@ -349,7 +371,12 @@ func (p *LLMPhraser) PhraseReminder(ctx context.Context, d loop.ReminderDecision
if err != nil {
return delivery.PhrasedReminder{}, err
}
body, mood := parseResponseMood(resp)
body, mood, perr := parseResponseMood(resp)
if perr != nil {
// Truncated JSON. Fall through to the reminder's own text.
log.Printf("phraser: PhraseReminder: %v", perr)
body, mood = "", ""
}
if body == "" {
// fallback: try old body/summary format
body, _ = parsePhrase(resp)
@@ -394,10 +421,16 @@ type chatReq struct {
// Russian, so an ASCII-only rule would make every reply empty. The escape rule
// is what lets the model close a string it opened with a quote inside. Length
// is bounded so a repetition loop truncates the field, not the JSON object.
//
// That bound was 400 and 400 was too tight. Measured against Qwen3.5-0.8B: on
// "почему гром слышно позже молнии?" the reply came back exactly 400 characters
// long, cut mid-word ("Нужно записать и,"), at every token cap from 256 to 2048.
// So the token cap was never what stopped it — this rule was. 1000 characters is
// roughly six Russian sentences, still short enough to stop a repetition loop.
const responseGrammar = `
root ::= "{" ws "\"response\"" ws ":" ws string ws "," ws "\"mood\"" ws ":" ws mood ws "}"
mood ::= "\"neutral\"" | "\"happy\"" | "\"thinking\"" | "\"tired\"" | "\"confused\""
string ::= "\"" ([^"\\] | "\\" ["\\/bfnrt]){0,400} "\""
string ::= "\"" ([^"\\] | "\\" ["\\/bfnrt]){0,1000} "\""
ws ::= [ \t\n]*
`
@@ -507,7 +540,9 @@ func (p *LLMPhraser) systemPrompt() string {
// querySystemPrompt returns the system prompt for PhraseQuery (notes + general
// knowledge). Prepends the configured persona when set.
func (p *LLMPhraser) querySystemPrompt() string {
base := "You are maven, a self-hosted personal assistant answering from your notes. Answer briefly and naturally in Russian starting with \"вот что я нашла: \". Respond ONLY with valid JSON: {\"response\": \"...\", \"mood\": \"neutral\"}."
// No self-introduction here: the persona block prepended one line above
// already says who she is, same as router.KnowledgePrompt.
base := "Ты отвечаешь ему по своим заметкам. Отвечай по-русски, коротко и своими словами, начинай с \"вот что я нашла: \". О себе — в женском роде (\"нашла\", \"записала\"). Он мужчина, обращайся к нему на \"ты\". Respond ONLY with valid JSON: {\"response\": \"...\", \"mood\": \"neutral\"}."
return persona.Prepend(p.cfg.ContextBlock, base)
}
@@ -629,21 +664,39 @@ type responseMood struct {
Mood string `json:"mood"`
}
// errBrokenJSON — the model started a JSON object and never finished it.
// That is a failed generation, not a reply. Callers must use their fallback.
var errBrokenJSON = fmt.Errorf("phraser: model output starts as JSON but does not parse")
// parseResponseMood extracts {"response","mood"} from LLM output, tolerant
// of thinking tokens and extra text before/after the JSON block. Returns
// ("", "") when no valid JSON is found.
func parseResponseMood(raw string) (response, mood string) {
// of thinking tokens and extra text before/after the JSON block.
//
// Three outcomes:
// - parsed fine → the fields, nil error.
// - output never looked like JSON → ("", "", nil). The caller may ship it
// as-is; small models sometimes answer in bare prose and that is fine.
// - output starts with "{" but does not parse → errBrokenJSON. The grammar
// guarantees a valid *prefix*, so a generation that hits the token cap
// mid-object comes back as a fragment like `{` or `{\n "`. Shipping that
// as a reply is the bug this error exists to stop.
func parseResponseMood(raw string) (response, mood string, err error) {
cleaned := strings.TrimSpace(raw)
start := strings.Index(cleaned, "{")
end := strings.LastIndex(cleaned, "}")
if start < 0 || end < 0 || end <= start {
return "", ""
if strings.HasPrefix(cleaned, "{") {
return "", "", errBrokenJSON
}
return "", "", nil
}
var parsed responseMood
if err := json.Unmarshal([]byte(cleaned[start:end+1]), &parsed); err != nil {
return "", ""
if e := json.Unmarshal([]byte(cleaned[start:end+1]), &parsed); e != nil {
if strings.HasPrefix(cleaned, "{") {
return "", "", errBrokenJSON
}
return "", "", nil
}
return parsed.Response, parsed.Mood
return parsed.Response, parsed.Mood, nil
}
func parsePhrase(raw string) (body, summary string) {