Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| ddb658ffbb |
@@ -0,0 +1,116 @@
|
|||||||
|
// Package kiwix reads a local Kiwix server (offline Wikipedia and friends).
|
||||||
|
//
|
||||||
|
// Why: the resident model is a 0.8B and invents facts. Letting her read a local
|
||||||
|
// article snippet beats letting her recall. Nothing here talks to the internet;
|
||||||
|
// the Kiwix server is on the same box.
|
||||||
|
//
|
||||||
|
// This is search only. Full articles are ~100KB of HTML, far too big for a 4096
|
||||||
|
// token context, so the unit of context is the search snippet (~500 chars).
|
||||||
|
package kiwix
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/xml"
|
||||||
|
"fmt"
|
||||||
|
"html"
|
||||||
|
"io"
|
||||||
|
"net/http"
|
||||||
|
"net/url"
|
||||||
|
"regexp"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Result is one search hit.
|
||||||
|
type Result struct {
|
||||||
|
Title string // article title, e.g. "Rayleigh scattering"
|
||||||
|
Path string // e.g. /content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering
|
||||||
|
Snippet string // plain text, tags stripped, entities decoded
|
||||||
|
WordCount int // 0 if the server did not say
|
||||||
|
}
|
||||||
|
|
||||||
|
// Client is a Kiwix HTTP client. Boring on purpose: no retries, no cache.
|
||||||
|
type Client struct {
|
||||||
|
base string
|
||||||
|
http *http.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
// New makes a client for a Kiwix base URL like http://127.0.0.1:8034.
|
||||||
|
func New(baseURL string) *Client {
|
||||||
|
return &Client{
|
||||||
|
base: strings.TrimRight(baseURL, "/"),
|
||||||
|
http: &http.Client{Timeout: 10 * time.Second},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search runs a keyword search in one ZIM (book) and returns up to limit hits.
|
||||||
|
//
|
||||||
|
// Ranking is keyword based, not semantic: "Rayleigh scattering" finds the right
|
||||||
|
// article, "why is the sky blue" finds a TV episode. Pass keywords, not questions.
|
||||||
|
func (c *Client) Search(ctx context.Context, pattern, book string, limit int) ([]Result, error) {
|
||||||
|
if limit <= 0 {
|
||||||
|
limit = 5
|
||||||
|
}
|
||||||
|
q := url.Values{}
|
||||||
|
q.Set("pattern", pattern)
|
||||||
|
q.Set("books.name", book)
|
||||||
|
q.Set("format", "xml")
|
||||||
|
q.Set("pageLength", strconv.Itoa(limit))
|
||||||
|
|
||||||
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.base+"/search?"+q.Encode(), nil)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
resp, err := c.http.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
return nil, fmt.Errorf("kiwix search: http %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
return ParseSearchRSS(resp.Body)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rss mirrors just the bits of the RSS 2.0 reply we use.
|
||||||
|
type rss struct {
|
||||||
|
Items []struct {
|
||||||
|
Title string `xml:"title"`
|
||||||
|
Link string `xml:"link"`
|
||||||
|
// innerxml keeps the <b> match markers so we can strip them ourselves.
|
||||||
|
Description struct {
|
||||||
|
Inner string `xml:",innerxml"`
|
||||||
|
} `xml:"description"`
|
||||||
|
WordCount string `xml:"wordCount"`
|
||||||
|
} `xml:"channel>item"`
|
||||||
|
}
|
||||||
|
|
||||||
|
var tagRE = regexp.MustCompile(`<[^>]*>`)
|
||||||
|
|
||||||
|
// ParseSearchRSS turns a Kiwix search reply into results. Exported so the parser
|
||||||
|
// is testable from a captured response, with no server running.
|
||||||
|
func ParseSearchRSS(r io.Reader) ([]Result, error) {
|
||||||
|
var doc rss
|
||||||
|
if err := xml.NewDecoder(r).Decode(&doc); err != nil {
|
||||||
|
return nil, fmt.Errorf("kiwix search: bad xml: %w", err)
|
||||||
|
}
|
||||||
|
out := make([]Result, 0, len(doc.Items))
|
||||||
|
for _, it := range doc.Items {
|
||||||
|
n, _ := strconv.Atoi(strings.ReplaceAll(it.WordCount, ",", ""))
|
||||||
|
out = append(out, Result{
|
||||||
|
Title: strings.TrimSpace(it.Title),
|
||||||
|
Path: strings.TrimSpace(it.Link),
|
||||||
|
Snippet: plainText(it.Description.Inner),
|
||||||
|
WordCount: n,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// plainText drops markup and decodes entities, leaving text a model can read.
|
||||||
|
func plainText(s string) string {
|
||||||
|
s = tagRE.ReplaceAllString(s, "")
|
||||||
|
s = html.UnescapeString(s)
|
||||||
|
return strings.TrimSpace(strings.Join(strings.Fields(s), " "))
|
||||||
|
}
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// A real reply from the live server, trimmed to two items.
|
||||||
|
const sampleRSS = `<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<rss version="2.0" xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/">
|
||||||
|
<channel>
|
||||||
|
<title>Search: Rayleigh scattering</title>
|
||||||
|
<opensearch:totalResults>800</opensearch:totalResults>
|
||||||
|
<item>
|
||||||
|
<title>Rayleigh scattering</title>
|
||||||
|
<link>/content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering</link>
|
||||||
|
<description><b>Rayleigh</b> scattering causes the blue color of the sky & yellow colors near the Sun.[1]</description>
|
||||||
|
<book><title>Wikipedia</title></book>
|
||||||
|
<wordCount>2,818</wordCount>
|
||||||
|
</item>
|
||||||
|
<item>
|
||||||
|
<title>Hyper–Rayleigh scattering</title>
|
||||||
|
<link>/content/wikipedia_en_all_maxi_2026-02/Hyper%E2%80%93Rayleigh_scattering</link>
|
||||||
|
<description>...<b>Rayleigh</b> scattering" is a nonlinear optical counterpart.</description>
|
||||||
|
<book><title>Wikipedia</title></book>
|
||||||
|
<wordCount>914</wordCount>
|
||||||
|
</item>
|
||||||
|
</channel>
|
||||||
|
</rss>`
|
||||||
|
|
||||||
|
func TestParseSearchRSS(t *testing.T) {
|
||||||
|
got, err := ParseSearchRSS(strings.NewReader(sampleRSS))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse: %v", err)
|
||||||
|
}
|
||||||
|
if len(got) != 2 {
|
||||||
|
t.Fatalf("want 2 results, got %d", len(got))
|
||||||
|
}
|
||||||
|
if got[0].Title != "Rayleigh scattering" {
|
||||||
|
t.Errorf("title = %q", got[0].Title)
|
||||||
|
}
|
||||||
|
if got[0].Path != "/content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering" {
|
||||||
|
t.Errorf("path = %q", got[0].Path)
|
||||||
|
}
|
||||||
|
if got[0].WordCount != 2818 {
|
||||||
|
t.Errorf("wordCount = %d", got[0].WordCount)
|
||||||
|
}
|
||||||
|
want := "Rayleigh scattering causes the blue color of the sky & yellow colors near the Sun.[1]"
|
||||||
|
if got[0].Snippet != want {
|
||||||
|
t.Errorf("snippet = %q, want %q", got[0].Snippet, want)
|
||||||
|
}
|
||||||
|
if strings.Contains(got[1].Snippet, "<b>") {
|
||||||
|
t.Errorf("second snippet still has tags: %q", got[1].Snippet)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseSearchRSSBadXML(t *testing.T) {
|
||||||
|
if _, err := ParseSearchRSS(strings.NewReader("not xml at all")); err == nil {
|
||||||
|
t.Fatal("want an error on junk input")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Opt-in: needs a live Kiwix server. CI has none.
|
||||||
|
// MAVEN_KIWIX_URL=http://127.0.0.1:8034 no_proxy=127.0.0.1,localhost go test -run Retrieval -v ./internal/kiwix/
|
||||||
|
func TestRetrievalEval(t *testing.T) {
|
||||||
|
base := os.Getenv("MAVEN_KIWIX_URL")
|
||||||
|
if base == "" {
|
||||||
|
t.Skip("set MAVEN_KIWIX_URL to run the retrieval eval")
|
||||||
|
}
|
||||||
|
noProxyLoopback(t)
|
||||||
|
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
rep, err := RunRetrievalEval(ctx, New(base), 5)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("eval: %v", err)
|
||||||
|
}
|
||||||
|
// No pass bar on purpose: the number is the finding.
|
||||||
|
t.Log("\n" + rep.String() + rep.Detail())
|
||||||
|
}
|
||||||
|
|
||||||
|
// noProxyLoopback stops the box's SOCKS bridge from eating loopback requests.
|
||||||
|
func noProxyLoopback(t *testing.T) {
|
||||||
|
t.Setenv("no_proxy", "127.0.0.1,localhost")
|
||||||
|
t.Setenv("NO_PROXY", "127.0.0.1,localhost")
|
||||||
|
}
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
{
|
||||||
|
"name": "kiwix-knowledge-v1",
|
||||||
|
"book": "wikipedia_en_all_maxi_2026-02",
|
||||||
|
"note": "The 9 knowledge cases from internal/phraser/eval/talk_v1.json. Queries are hand-written English keywords on purpose: Kiwix ranks by keyword, not meaning, so a natural question fails. Writing them by hand separates 'retrieval is broken' from 'the model writes bad queries'.",
|
||||||
|
"cases": [
|
||||||
|
{
|
||||||
|
"id": "know-sky-blue",
|
||||||
|
"question": "почему небо синее?",
|
||||||
|
"query": "Rayleigh scattering sky blue",
|
||||||
|
"want_titles": ["Rayleigh scattering", "Diffuse sky radiation"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-boil-egg",
|
||||||
|
"question": "сколько варить яйцо вкрутую?",
|
||||||
|
"query": "boiled egg cooking",
|
||||||
|
"want_titles": ["Boiled egg", "Egg as food"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-ssd-vs-hdd",
|
||||||
|
"question": "чем ssd отличается от hdd?",
|
||||||
|
"query": "solid-state drive",
|
||||||
|
"want_titles": ["Solid-state drive", "Hard disk drive"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-cat-purr",
|
||||||
|
"question": "почему кошки мурчат?",
|
||||||
|
"query": "cat purr",
|
||||||
|
"want_titles": ["Purr", "Cat communication"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-hiccups",
|
||||||
|
"question": "как быстро избавиться от икоты?",
|
||||||
|
"query": "hiccup",
|
||||||
|
"want_titles": ["Hiccup"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-polite-form",
|
||||||
|
"question": "не могли бы вы объяснить, что такое vpn?",
|
||||||
|
"query": "virtual private network",
|
||||||
|
"want_titles": ["Virtual private network"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-dont-know",
|
||||||
|
"question": "как зовут моего соседа снизу?",
|
||||||
|
"query": "name of my downstairs neighbour",
|
||||||
|
"want_titles": [],
|
||||||
|
"expect_miss": true,
|
||||||
|
"note": "Unanswerable by design. Retrieval SHOULD find nothing useful. Counted as a hit only when nothing relevant comes back."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-water-per-day",
|
||||||
|
"question": "сколько воды в день надо пить?",
|
||||||
|
"query": "human daily water requirement drinking",
|
||||||
|
"want_titles": ["Drinking water", "Water", "Dehydration", "Hydration"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-thunder-delay",
|
||||||
|
"question": "почему гром слышно позже молнии?",
|
||||||
|
"query": "thunder speed of sound lightning",
|
||||||
|
"want_titles": ["Thunder", "Lightning"]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
// This scores retrieval alone: no LLM. For each general-knowledge question we
|
||||||
|
// hand-write English keywords and ask whether the article that would answer it
|
||||||
|
// comes back in the top N hits. If this score is low, reading Wikipedia cannot
|
||||||
|
// help the model no matter how good the prompt is.
|
||||||
|
//
|
||||||
|
// The unanswerable case (know-dont-know) is not scored. Whether the junk it
|
||||||
|
// returns is "nothing useful" is a human judgement, so the report just prints
|
||||||
|
// the titles and leaves the score to the 8 answerable cases.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
_ "embed"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
//go:embed knowledge_v1.json
|
||||||
|
var knowledgeFixtureJSON []byte
|
||||||
|
|
||||||
|
// EvalCase — one question with hand-written keywords.
|
||||||
|
type EvalCase struct {
|
||||||
|
ID string `json:"id"`
|
||||||
|
Question string `json:"question"`
|
||||||
|
Query string `json:"query"`
|
||||||
|
WantTitles []string `json:"want_titles"`
|
||||||
|
ExpectMiss bool `json:"expect_miss"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type fixture struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
Book string `json:"book"`
|
||||||
|
Cases []EvalCase `json:"cases"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// Outcome — what one case retrieved.
|
||||||
|
type Outcome struct {
|
||||||
|
Case EvalCase
|
||||||
|
Titles []string // titles of the top N hits, in rank order
|
||||||
|
Rank int // 1-based rank of the first wanted title, 0 if none
|
||||||
|
Err error
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hit is true when a wanted title came back.
|
||||||
|
func (o Outcome) Hit() bool { return o.Rank > 0 }
|
||||||
|
|
||||||
|
// Report — the score plus per-case detail.
|
||||||
|
type Report struct {
|
||||||
|
Name string
|
||||||
|
Book string
|
||||||
|
TopN int
|
||||||
|
Scored int // answerable cases
|
||||||
|
Hits int
|
||||||
|
Errors int
|
||||||
|
Outcomes []Outcome
|
||||||
|
}
|
||||||
|
|
||||||
|
// Accuracy over the answerable cases.
|
||||||
|
func (r Report) Accuracy() float64 {
|
||||||
|
if r.Scored == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return float64(r.Hits) / float64(r.Scored)
|
||||||
|
}
|
||||||
|
|
||||||
|
// RunRetrievalEval searches for every fixture case.
|
||||||
|
func RunRetrievalEval(ctx context.Context, c *Client, topN int) (Report, error) {
|
||||||
|
var f fixture
|
||||||
|
if err := json.Unmarshal(knowledgeFixtureJSON, &f); err != nil {
|
||||||
|
return Report{}, err
|
||||||
|
}
|
||||||
|
rep := Report{Name: f.Name, Book: f.Book, TopN: topN}
|
||||||
|
for _, cs := range f.Cases {
|
||||||
|
res, err := c.Search(ctx, cs.Query, f.Book, topN)
|
||||||
|
o := Outcome{Case: cs, Err: err}
|
||||||
|
if err != nil {
|
||||||
|
rep.Errors++
|
||||||
|
}
|
||||||
|
for i, hit := range res {
|
||||||
|
o.Titles = append(o.Titles, hit.Title)
|
||||||
|
if o.Rank == 0 && matches(cs.WantTitles, hit.Title) {
|
||||||
|
o.Rank = i + 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !cs.ExpectMiss {
|
||||||
|
rep.Scored++
|
||||||
|
if o.Hit() {
|
||||||
|
rep.Hits++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
rep.Outcomes = append(rep.Outcomes, o)
|
||||||
|
}
|
||||||
|
return rep, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func matches(want []string, title string) bool {
|
||||||
|
for _, w := range want {
|
||||||
|
if strings.EqualFold(strings.TrimSpace(title), w) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// String — the headline number.
|
||||||
|
func (r Report) String() string {
|
||||||
|
var b strings.Builder
|
||||||
|
fmt.Fprintf(&b, "%s: %d/%d answerable questions retrieve a wanted article in top %d (%.1f%%), %d errors\n",
|
||||||
|
r.Name, r.Hits, r.Scored, r.TopN, 100*r.Accuracy(), r.Errors)
|
||||||
|
fmt.Fprintf(&b, " book: %s\n", r.Book)
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Detail — per case: what was asked, what was searched, what came back.
|
||||||
|
func (r Report) Detail() string {
|
||||||
|
var b strings.Builder
|
||||||
|
for _, o := range r.Outcomes {
|
||||||
|
mark := "MISS"
|
||||||
|
switch {
|
||||||
|
case o.Case.ExpectMiss:
|
||||||
|
mark = "n/a "
|
||||||
|
case o.Hit():
|
||||||
|
mark = fmt.Sprintf("hit@%d", o.Rank)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " %-6s %-20s q=%q\n", mark, o.Case.ID, o.Case.Query)
|
||||||
|
if o.Err != nil {
|
||||||
|
fmt.Fprintf(&b, " error: %v\n", o.Err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " got: %s\n", strings.Join(o.Titles, " | "))
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user