Compare commits

..

2 Commits

Author SHA1 Message Date
kami 742b2ad1d7 Score the Russian-to-keywords rewrite end to end (#403)
Same 9 cases as the retrieval eval, so the numbers compare directly:
hand-written keywords hit 8 of 8, this is what the model reaches on its
own. Reports the hand-written query next to the model's for every case,
because where the phrasing differs is the useful part.

Opt-in on MAVEN_KIWIX_URL + MAVEN_LLM_URL, like the other evals.

Result on Qwen3.5-0.8B: 3 of 8, identical on all three runs.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 18:19:56 +04:00
kami b300ac5c70 Rewrite a Russian question into English Kiwix keywords (#403)
Kiwix ranks by keyword, not meaning, so a translated question finds song
and TV titles. This asks the resident model for the TOPIC instead: a short
English noun phrase, like a Wikipedia article title.

Locked down three ways, because a wrong query is silently wrong:
- A GBNF grammar, same idea as routeGrammar and responseGrammar. The
  reply must be {"query":"..."} with Latin words only. The JSON wrapper
  matters: this model always thinks out loud and this llama-server build
  ignores the thinking switch, so a bare word-list grammar just captured
  "Let me analyze this request carefully" for every question.
- max_tokens 32, since the answer is a few words.
- CleanQuery, which throws away empty, Russian and prose replies rather
  than passing them to Kiwix, and drops question words like "why" and
  "how much" that a keyword ranker cannot use anyway.

Client side only. Nothing is wired into the daemon or any config.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 18:19:42 +04:00
4 changed files with 417 additions and 0 deletions
+183
View File
@@ -0,0 +1,183 @@
package kiwix
// Turning a Russian question into an English Kiwix search.
//
// Kiwix ranks by keyword, not by meaning. "why is the sky blue" returns a TV
// episode; "Rayleigh scattering sky blue" returns the right article. So the
// model's job here is NOT translation — it is naming the English article the
// answer lives in.
//
// The output space is a handful of words, so it is worth locking down hard: a
// GBNF grammar for the shape, a tiny token cap, and a cleanup pass that throws
// away anything odd rather than handing junk to Kiwix.
import (
"context"
"encoding/json"
"fmt"
"strings"
"unicode"
"github.com/kami/maven/internal/llm"
)
// Completer — the LLM seam, so tests can fake it. *llm.Client satisfies it.
type Completer interface {
Complete(ctx context.Context, r llm.Req) (string, error)
}
// queryGrammar — one JSON object holding 1..6 keyword words. Latin letters,
// digits and hyphens only, so the model physically cannot answer the question
// or reply in Russian.
//
// Why the JSON wrapper: this model always thinks out loud and this llama-server
// build ignores the thinking switch (see ROUTING-EVAL-31-07-2026.md). A bare
// word-list grammar just captured the reasoning — every case came back as
// "Let me analyze this request carefully". Demanding JSON, like routeGrammar and
// responseGrammar already do, gives the reasoning nowhere to go.
const queryGrammar = `
root ::= "{" ws "\"query\"" ws ":" ws "\"" word (" " word){0,5} "\"" ws "}"
word ::= [A-Za-z0-9] [A-Za-z0-9-]{0,23}
ws ::= [ \t\n]*
`
// rewriteSystem — asks for search keywords, not an answer and not a translation.
const rewriteSystem = `You turn a question into a search query for English Wikipedia.
Rules:
- Output ONLY English search keywords. Never an answer, never an explanation.
- Do NOT translate the sentence. Name the thing the answer is about.
- The output must be a noun phrase, like a Wikipedia article title.
- Never use question words: no why, how, what, when, which, "how much",
"how long", "how to", "vs", "reason", "difference".
- 2 to 4 words.
Reply with JSON: {"query":"<keywords>"}
Good:
"почему листья желтеют осенью?" -> {"query":"leaf senescence autumn"}
"как работает микроволновка?" -> {"query":"microwave oven"}
"не могли бы вы объяснить, что такое блокчейн?" -> {"query":"blockchain"}
"сколько живут собаки?" -> {"query":"dog lifespan"}
"как избавиться от комаров в квартире?" -> {"query":"mosquito control"}
"чем чай отличается от кофе?" -> {"query":"tea"}
Only JSON, no explanation.`
// maxQueryTokens — the output is a few words plus the JSON wrapper. A tight cap
// is the cheapest guard against the model rambling into an answer.
const maxQueryTokens = 32
// Rewriter asks the resident model for English search keywords.
type Rewriter struct{ c Completer }
func NewRewriter(c Completer) *Rewriter { return &Rewriter{c: c} }
// Rewrite returns English keywords for a question in any language.
// It errors rather than returning something Kiwix should not see.
func (r *Rewriter) Rewrite(ctx context.Context, question string) (string, error) {
raw, err := r.c.Complete(ctx, llm.Req{
System: rewriteSystem,
User: strings.TrimSpace(question),
Grammar: queryGrammar,
MaxTokens: maxQueryTokens,
RepeatPenalty: 1.15,
})
if err != nil {
return "", err
}
return CleanQuery(unwrapJSON(raw))
}
// unwrapJSON pulls the query out of {"query":"..."}. If the reply is not that
// shape it is returned as-is, and CleanQuery decides whether it is usable.
func unwrapJSON(raw string) string {
s := strings.TrimSpace(raw)
if !strings.HasPrefix(s, "{") {
return s
}
var got struct{ Query string }
if err := json.Unmarshal([]byte(s), &got); err != nil {
return s
}
return got.Query
}
// maxQueryWords matches the grammar's bound. Anything longer is prose.
const maxQueryWords = 6
// CleanQuery checks and tidies whatever the model produced. The grammar makes
// bad output unlikely, not impossible (a server without grammar support, a
// different model), so this is the real gate in front of Kiwix.
//
// Exported so it can be tested without a model.
func CleanQuery(raw string) (string, error) {
s := strings.TrimSpace(raw)
// Models like to wrap answers in quotes. Drop surrounding ones.
s = strings.Trim(s, "\"'`")
// Keep the first line only: everything after it is prose.
if i := strings.IndexAny(s, "\r\n"); i >= 0 {
s = s[:i]
}
// Keep letters, digits, spaces and hyphens; anything else becomes a space.
var b strings.Builder
for _, ru := range s {
switch {
case unicode.IsLetter(ru) || unicode.IsDigit(ru) || ru == '-':
b.WriteRune(ru)
default:
b.WriteRune(' ')
}
}
words := strings.Fields(b.String())
if len(words) == 0 {
return "", fmt.Errorf("kiwix rewrite: empty query")
}
if len(words) > maxQueryWords {
return "", fmt.Errorf("kiwix rewrite: %d words, want at most %d (looks like prose)", len(words), maxQueryWords)
}
words = dropStopWords(words)
out := strings.Join(words, " ")
// The ZIMs are English. Non-Latin letters mean the model ignored the ask.
for _, ru := range out {
if unicode.IsLetter(ru) && !isLatin(ru) {
return "", fmt.Errorf("kiwix rewrite: query is not English: %q", out)
}
}
return out, nil
}
// stopWords — question words and filler. The model keeps writing question-shaped
// queries ("why is the sky blue", "how much water to drink daily") no matter how
// the prompt is worded, and Kiwix ranks on every word, so those words drag in
// song and episode titles. Dropping them in code is not a style preference: a
// keyword ranker gets nothing from them.
var stopWords = map[string]bool{
"a": true, "an": true, "the": true, "is": true, "are": true, "was": true,
"do": true, "does": true, "did": true, "to": true, "of": true, "in": true,
"on": true, "for": true, "and": true, "or": true, "my": true, "me": true,
"i": true, "it": true, "its": true, "be": true, "been": true, "get": true,
"how": true, "why": true, "what": true, "when": true, "which": true,
"who": true, "where": true, "much": true, "many": true, "long": true,
"vs": true, "than": true, "rid": true, "from": true, "about": true,
}
// dropStopWords removes filler, but never everything: if the query was nothing
// but stop words there is nothing better to search, so the original is kept and
// the caller sees whatever Kiwix makes of it.
func dropStopWords(words []string) []string {
kept := make([]string, 0, len(words))
for _, w := range words {
if !stopWords[strings.ToLower(w)] {
kept = append(kept, w)
}
}
if len(kept) == 0 {
return words
}
return kept
}
func isLatin(ru rune) bool {
return (ru >= 'a' && ru <= 'z') || (ru >= 'A' && ru <= 'Z')
}
+96
View File
@@ -0,0 +1,96 @@
package kiwix
// End-to-end score: Russian question -> model rewrite -> Kiwix search -> did a
// wanted article come back. Same 9 cases as the retrieval eval, so the two
// numbers are directly comparable: retrieval with hand-written keywords is the
// ceiling, this is what the model actually reaches.
import (
"context"
"encoding/json"
"fmt"
"strings"
)
// RewriteOutcome — one case, end to end.
type RewriteOutcome struct {
Outcome
ModelQuery string // what the model asked for ("" if it failed)
RewriteErr error
}
// RunRewriteEval rewrites every question with the model, then searches.
func RunRewriteEval(ctx context.Context, c *Client, rw *Rewriter, topN int) (RewriteReport, error) {
var f fixture
if err := json.Unmarshal(knowledgeFixtureJSON, &f); err != nil {
return RewriteReport{}, err
}
rep := RewriteReport{Report: Report{Name: f.Name + "-rewrite", Book: f.Book, TopN: topN}}
for _, cs := range f.Cases {
out := RewriteOutcome{Outcome: Outcome{Case: cs}}
q, err := rw.Rewrite(ctx, cs.Question)
out.ModelQuery, out.RewriteErr = q, err
if err == nil {
res, serr := c.Search(ctx, q, f.Book, topN)
out.Err = serr
for i, hit := range res {
out.Titles = append(out.Titles, hit.Title)
if out.Rank == 0 && matches(cs.WantTitles, hit.Title) {
out.Rank = i + 1
}
}
}
if out.RewriteErr != nil || out.Err != nil {
rep.Errors++
}
if !cs.ExpectMiss {
rep.Scored++
if out.Hit() {
rep.Hits++
}
}
rep.Cases = append(rep.Cases, out)
}
return rep, nil
}
// RewriteReport — the score plus per-case detail.
type RewriteReport struct {
Report
Cases []RewriteOutcome
}
// String — the headline number.
func (r RewriteReport) String() string {
return fmt.Sprintf("%s: %d/%d answerable questions retrieve a wanted article in top %d (%.1f%%), %d errors\n book: %s\n",
r.Name, r.Hits, r.Scored, r.TopN, 100*r.Accuracy(), r.Errors, r.Book)
}
// Detail — per case: hand-written query next to the model's, and what came back.
// The point is seeing WHERE the model's phrasing differs, not just the score.
func (r RewriteReport) Detail() string {
var b strings.Builder
for _, o := range r.Cases {
mark := "MISS"
switch {
case o.Case.ExpectMiss:
mark = "n/a "
case o.Hit():
mark = fmt.Sprintf("hit@%d", o.Rank)
}
fmt.Fprintf(&b, " %-6s %-20s\n", mark, o.Case.ID)
fmt.Fprintf(&b, " asked: %s\n", o.Case.Question)
fmt.Fprintf(&b, " hand: %q\n", o.Case.Query)
fmt.Fprintf(&b, " model: %q\n", o.ModelQuery)
if o.RewriteErr != nil {
fmt.Fprintf(&b, " rewrite rejected: %v\n", o.RewriteErr)
continue
}
if o.Err != nil {
fmt.Fprintf(&b, " search error: %v\n", o.Err)
continue
}
fmt.Fprintf(&b, " got: %s\n", strings.Join(o.Titles, " | "))
}
return b.String()
}
+33
View File
@@ -0,0 +1,33 @@
package kiwix
import (
"context"
"os"
"testing"
"time"
"github.com/kami/maven/internal/llm"
)
// Opt-in: needs a live Kiwix server AND a live llama-server.
// MAVEN_KIWIX_URL=http://127.0.0.1:8034 MAVEN_LLM_URL=http://127.0.0.1:18099 \
//
// no_proxy=127.0.0.1,localhost go test -run RewriteEval -v ./internal/kiwix/
func TestRewriteEval(t *testing.T) {
kbase, lbase := os.Getenv("MAVEN_KIWIX_URL"), os.Getenv("MAVEN_LLM_URL")
if kbase == "" || lbase == "" {
t.Skip("set MAVEN_KIWIX_URL and MAVEN_LLM_URL to run the rewrite eval")
}
noProxyLoopback(t)
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Minute)
defer cancel()
rw := NewRewriter(llm.New(lbase, 3*time.Minute))
rep, err := RunRewriteEval(ctx, New(kbase), rw, 5)
if err != nil {
t.Fatalf("eval: %v", err)
}
// No pass bar on purpose: the number is the finding.
t.Log("\n" + rep.String() + rep.Detail())
}
+105
View File
@@ -0,0 +1,105 @@
package kiwix
import (
"context"
"testing"
"github.com/kami/maven/internal/llm"
)
// Bad model output must never reach Kiwix. No model needed for this.
func TestCleanQueryRejectsJunk(t *testing.T) {
bad := []struct{ name, raw string }{
{"empty", ""},
{"blank", " \n "},
{"russian came back", "почему небо синее"},
{"mixed russian", "sky синее scattering"},
{"full sentence", "The sky looks blue because of the scattering of sunlight by air molecules"},
{"prose with quotes", `Sure! Here is a good search query: "Rayleigh scattering", which explains it.`},
}
for _, c := range bad {
if got, err := CleanQuery(c.raw); err == nil {
t.Errorf("%s: want rejection, got %q", c.name, got)
}
}
}
func TestCleanQueryCleans(t *testing.T) {
ok := []struct{ raw, want string }{
{"Rayleigh scattering sky", "Rayleigh scattering sky"},
{" boiled egg cooking \n", "boiled egg cooking"},
{`"virtual private network"`, "virtual private network"},
{"solid-state drive", "solid-state drive"},
{"cat purr.", "cat purr"},
{"hiccup\nAlso: hiccough", "hiccup"},
// Question words are filler to a keyword ranker, so they go.
{"why is the sky blue", "sky blue"},
{"how much water to drink daily", "water drink daily"},
{"SSD vs HDD comparison", "SSD HDD comparison"},
// Nothing but filler: keep it rather than return nothing.
{"what is it", "what is it"},
}
for _, c := range ok {
got, err := CleanQuery(c.raw)
if err != nil {
t.Errorf("%q: %v", c.raw, err)
continue
}
if got != c.want {
t.Errorf("%q -> %q, want %q", c.raw, got, c.want)
}
}
}
type fakeCompleter struct {
out string
req llm.Req
}
func (f *fakeCompleter) Complete(_ context.Context, r llm.Req) (string, error) {
f.req = r
return f.out, nil
}
func TestRewriteConstrainsTheCall(t *testing.T) {
f := &fakeCompleter{out: `{"query":"Rayleigh scattering sky"}`}
got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?")
if err != nil {
t.Fatalf("rewrite: %v", err)
}
if got != "Rayleigh scattering sky" {
t.Errorf("query = %q", got)
}
if f.req.Grammar == "" {
t.Error("no grammar sent")
}
if f.req.MaxTokens == 0 || f.req.MaxTokens > 32 {
t.Errorf("max_tokens = %d, want a small cap", f.req.MaxTokens)
}
}
func TestRewriteRejectsBadModelOutput(t *testing.T) {
bad := []string{
`{"query":"почему небо синее"}`, // never translated
`{"query":""}`, // empty
`{"query":"the sky is blue because sunlight is scattered by air"}`, // an answer
// Note: a SHORT English prose fragment ("Let me analyze this request")
// is under the word cap and cannot be caught here. The grammar is what
// stops that one.
}
for _, out := range bad {
f := &fakeCompleter{out: out}
if got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?"); err == nil {
t.Errorf("%s: want rejection, got %q", out, got)
}
}
}
// A reply that is not the JSON shape but is still usable keywords should pass.
func TestRewriteFallsBackToPlainText(t *testing.T) {
f := &fakeCompleter{out: "Rayleigh scattering sky"}
got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?")
if err != nil || got != "Rayleigh scattering sky" {
t.Errorf("got %q, %v", got, err)
}
}