Compare commits
9 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 9e25f18a3e | |||
| 197897516e | |||
| 767748720a | |||
| 58051b5af1 | |||
| f9b2391a8b | |||
| f229795cea | |||
| 6e5364a0ed | |||
| 86817d6d06 | |||
| 0fc2e3a18a |
@@ -715,7 +715,7 @@ func (h *reactiveHandler) queryKiwix(ctx context.Context, t *queryTurn) (string,
|
||||
// not be sent to an upstream engine at all. The guard closes both holes with
|
||||
// the same test.
|
||||
func (h *reactiveHandler) queryPersonal(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
if !isPersonalQuery(t.dec.Utterance) {
|
||||
if !h.isPersonalTurn(ctx, t) {
|
||||
return "", false
|
||||
}
|
||||
log.Printf("voice: %q is about him and his own data did not answer it; not asking the world", t.dec.Utterance)
|
||||
@@ -740,7 +740,9 @@ var personalMarkers = []*regexp.Regexp{
|
||||
regexp.MustCompile(`(?i)\bdid\s+i\b`),
|
||||
}
|
||||
|
||||
// isPersonalQuery reports whether the utterance asks about something of his.
|
||||
// isPersonalQuery — the offline floor under the boundary. Possession only, and
|
||||
// deliberately still narrow: it answers when there is no embedder to ask, and a
|
||||
// broad guess made blind is worse than a narrow one.
|
||||
func isPersonalQuery(utterance string) bool {
|
||||
if utterance == "" {
|
||||
return false
|
||||
@@ -753,6 +755,23 @@ func isPersonalQuery(utterance string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// isPersonalTurn — the boundary test. The seeds decide when the embedder is
|
||||
// there, which is every deployed box; the possession markers are the floor
|
||||
// underneath, for a handler with no embedder or a turn whose vector never got
|
||||
// computed. Same shape as the cascade: the better test leads, the offline one
|
||||
// always answers.
|
||||
func (h *reactiveHandler) isPersonalTurn(ctx context.Context, t *queryTurn) bool {
|
||||
h.boundary.load(ctx, h.embedder)
|
||||
if personal, world, ok := h.boundary.score(t.vec); ok {
|
||||
if personal > world {
|
||||
log.Printf("voice: %q scores personal %.4f vs world %.4f", t.dec.Utterance, personal, world)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
return isPersonalQuery(t.dec.Utterance)
|
||||
}
|
||||
|
||||
// queryGeneral — general knowledge, the last source before giving up. It always
|
||||
// claims: either a model answers, or Maven names the gap, or she says she does
|
||||
// not know.
|
||||
|
||||
@@ -19,6 +19,10 @@ func TestIsPersonalQuery(t *testing.T) {
|
||||
"when is my meeting",
|
||||
"do i have anything today",
|
||||
"did i take my vitamins",
|
||||
// Speech, but only the forms possession already covers ("did i").
|
||||
// The verb forms the floor cannot see are the seeds' job, scored in
|
||||
// TestONNXPersonalBoundary.
|
||||
"what did i say about backups",
|
||||
} {
|
||||
if !isPersonalQuery(s) {
|
||||
t.Errorf("isPersonalQuery(%q) = false, want true", s)
|
||||
@@ -33,6 +37,10 @@ func TestIsPersonalQuery(t *testing.T) {
|
||||
"почему небо синее",
|
||||
"столица франции",
|
||||
"how do i boil an egg",
|
||||
// The floor is possession-only by design: a speech verb it cannot see
|
||||
// passes here and is caught by the seeds instead.
|
||||
"что я говорил про бэкапы?",
|
||||
"как я говорил, почему небо синее",
|
||||
"",
|
||||
} {
|
||||
if isPersonalQuery(s) {
|
||||
|
||||
@@ -251,6 +251,7 @@ func run(args []string) error {
|
||||
Listen: cfg.Phraser.Listen,
|
||||
NGpuLayers: cfg.Phraser.NGpuLayers,
|
||||
NCtx: cfg.Phraser.NCtx,
|
||||
CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB),
|
||||
Timeout: time.Duration(cfg.Phraser.Timeout),
|
||||
LLMNudges: cfg.Phraser.LLMNudges,
|
||||
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||
@@ -525,6 +526,7 @@ func run(args []string) error {
|
||||
Listen: cfg.Phraser.Listen,
|
||||
NGpuLayers: cfg.Phraser.NGpuLayers,
|
||||
NCtx: cfg.Phraser.NCtx,
|
||||
CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB),
|
||||
Timeout: time.Duration(cfg.Phraser.Timeout),
|
||||
LLMNudges: cfg.Phraser.LLMNudges,
|
||||
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||
@@ -788,6 +790,22 @@ func personaFacts(cfg *config.Config) persona.Facts {
|
||||
return f
|
||||
}
|
||||
|
||||
// cacheRAMMiB resolves phraser.cache_ram_mib into the phraser's field. Unset
|
||||
// means 512 MiB and not "whatever the server does", because the server's own
|
||||
// default is 8 GiB of prompt cache and that is what put 7.9 GB of RSS and half
|
||||
// a gigabyte of swap on homesrv for a 1.1 GB model. A negative value is the
|
||||
// deliberate opt-out: no flag is passed, the server's default applies, and the
|
||||
// operator owns the consequence.
|
||||
func cacheRAMMiB(configured int) int {
|
||||
if configured == 0 {
|
||||
return 512
|
||||
}
|
||||
if configured < 0 {
|
||||
return 0
|
||||
}
|
||||
return configured
|
||||
}
|
||||
|
||||
// contextBlockFn returns the per-turn renderer of the shared context block.
|
||||
// Per turn, not once at startup, because the block states the current time.
|
||||
func contextBlockFn(cfg *config.Config, now func() time.Time) func() string {
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"math"
|
||||
"sync"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// The personal boundary decides one thing: is this question about him. It used
|
||||
// to decide it by matching possession words, and that was the whole defect
|
||||
// behind Vikunja #495. "что я говорил про бэкапы?" is his data by definition —
|
||||
// nothing outside the box has ever heard him say anything — and it carried no
|
||||
// possession word, so it walked past the boundary into SearXNG and came back
|
||||
// answered out of a Habr article about somebody else's backups.
|
||||
//
|
||||
// The first fix was one more marker class, `я говорил|сказал|писал|…`, plus a
|
||||
// carve-out so "как я говорил, почему небо синее" stayed a world question. Both
|
||||
// halves are a lexicon, and a lexicon is the wrong instrument here: Russian
|
||||
// gives every verb a dozen surface forms, the preamble list has no end, and
|
||||
// every utterance the list misses is one that reaches the world. It also drifts
|
||||
// silently — a missing verb looks exactly like no bug.
|
||||
//
|
||||
// So the boundary asks the embedder instead. Two frozen seed sets — questions
|
||||
// about him, questions about the world — are embedded once, and the turn's own
|
||||
// query vector, already computed by queryEmbed upstream, is scored against
|
||||
// both. Nearest side wins. Word order, verb form and unseen phrasing stop
|
||||
// mattering, which is exactly what a lexicon could not do.
|
||||
//
|
||||
// Measured 03-08-2026 against multilingual-e5-small on 19 held-out utterances,
|
||||
// none of them a seed: 19 right (TestONNXPersonalBoundary). A 20th, "as i said,
|
||||
// what is the population of india", missed by +0.008 during the first pass and
|
||||
// is a world seed now, which is why it is not in the held-out set. True
|
||||
// positives clear the world side by +0.014 to +0.089 and the nearest true
|
||||
// negative sits at -0.005, so the gate is the sign of the difference and
|
||||
// nothing tighter: the margins are too thin to justify a threshold, and the
|
||||
// asymmetry favours claiming anyway. A false claim costs one honest "не знаю";
|
||||
// a false pass sends his life to an upstream engine.
|
||||
//
|
||||
// The embedder is the one model CLAUDE.md pins to homesrv permanently, and it
|
||||
// is what makes this affordable: no llama-server call, no network, one cosine
|
||||
// per seed against a vector the turn already has.
|
||||
|
||||
// personalSeeds — questions about him. Frozen: they are scoring data, so
|
||||
// editing one moves the boundary and must be re-measured, not eyeballed. Cover
|
||||
// both classes the boundary owns, possession and first-person speech, in both
|
||||
// languages.
|
||||
var personalSeeds = []string{
|
||||
"что я говорил про это",
|
||||
"я тебе рассказывал об этом?",
|
||||
"что я записал про врача",
|
||||
"я упоминал эту тему?",
|
||||
"что у меня сегодня",
|
||||
"когда моя встреча",
|
||||
"what did i say about this",
|
||||
"did i mention this to you",
|
||||
}
|
||||
|
||||
// worldSeeds — questions the world can answer, including the two shapes that
|
||||
// look personal and are not: a first-person preamble on a world question ("как
|
||||
// я говорил, ..."), and first person without possession ("что я могу
|
||||
// посмотреть вечером"). Refusing those is the opposite mistake and the older
|
||||
// comment on personalMarkers already named it.
|
||||
var worldSeeds = []string{
|
||||
"почему небо синее",
|
||||
"какая столица франции",
|
||||
"как сварить борщ",
|
||||
"кто написал эту книгу",
|
||||
"what is the capital of france",
|
||||
"how do i boil an egg",
|
||||
"как я говорил, почему небо синее",
|
||||
"as i said, why is the sky blue",
|
||||
"as i said, what is the population of india",
|
||||
"что я могу посмотреть вечером",
|
||||
"что мне почитать про историю",
|
||||
"что я должен знать про питон",
|
||||
"what can i watch tonight",
|
||||
}
|
||||
|
||||
// personalBoundary holds the embedded seeds. Zero value is usable and means
|
||||
// "not loaded yet"; a handler built without an embedder never loads and the
|
||||
// boundary falls back to personalMarkers.
|
||||
type personalBoundary struct {
|
||||
once sync.Once
|
||||
personal [][]float32
|
||||
world [][]float32
|
||||
loaded bool
|
||||
}
|
||||
|
||||
// load embeds both seed sets, once per process. Seeds are embedded on the QUERY
|
||||
// side, like the utterance they are compared with — a question against a
|
||||
// question. Mixing sides would measure the e5 prefix, not the meaning.
|
||||
func (b *personalBoundary) load(ctx context.Context, emb router.Embedder) {
|
||||
b.once.Do(func() {
|
||||
if emb == nil {
|
||||
return
|
||||
}
|
||||
embedAll := func(ss []string) [][]float32 {
|
||||
out := make([][]float32, 0, len(ss))
|
||||
for _, s := range ss {
|
||||
v, err := router.EmbedQuery(ctx, emb, s)
|
||||
if err != nil {
|
||||
log.Printf("voice: personal boundary seeds unavailable (%v); falling back to possession markers", err)
|
||||
return nil
|
||||
}
|
||||
out = append(out, v)
|
||||
}
|
||||
return out
|
||||
}
|
||||
p, w := embedAll(personalSeeds), embedAll(worldSeeds)
|
||||
if p == nil || w == nil {
|
||||
return
|
||||
}
|
||||
b.personal, b.world, b.loaded = p, w, true
|
||||
})
|
||||
}
|
||||
|
||||
// score returns the best similarity to each side. ok is false when the seeds
|
||||
// are not loaded, which is the caller's signal to use the markers instead.
|
||||
func (b *personalBoundary) score(vec []float32) (personal, world float64, ok bool) {
|
||||
if !b.loaded || len(vec) == 0 {
|
||||
return 0, 0, false
|
||||
}
|
||||
best := func(seeds [][]float32) float64 {
|
||||
m := -1.0
|
||||
for _, s := range seeds {
|
||||
if c := cosine(vec, s); c > m {
|
||||
m = c
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
return best(b.personal), best(b.world), true
|
||||
}
|
||||
|
||||
// cosine — same math as internal/router and internal/memory, small enough that
|
||||
// importing one of them for it would be the larger coupling.
|
||||
func cosine(a, b []float32) float64 {
|
||||
if len(a) != len(b) {
|
||||
return 0
|
||||
}
|
||||
var dot, na, nb float64
|
||||
for i := range a {
|
||||
dot += float64(a[i]) * float64(b[i])
|
||||
na += float64(a[i]) * float64(a[i])
|
||||
nb += float64(b[i]) * float64(b[i])
|
||||
}
|
||||
if na == 0 || nb == 0 {
|
||||
return 0
|
||||
}
|
||||
return dot / (math.Sqrt(na) * math.Sqrt(nb))
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// A handler with no embedder never loads the seeds, so the boundary falls back
|
||||
// to the possession markers. That is the offline floor and it must keep working
|
||||
// — an embedder that fails to load must not open the boundary.
|
||||
func TestBoundaryFallsBackToMarkersWithNoEmbedder(t *testing.T) {
|
||||
h := personalHandler()
|
||||
if !h.isPersonalTurn(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Utterance: "во сколько у меня встреча"},
|
||||
}) {
|
||||
t.Error("no embedder: a possession question must still be personal")
|
||||
}
|
||||
if h.isPersonalTurn(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Utterance: "почему небо синее"},
|
||||
}) {
|
||||
t.Error("no embedder: a world question must still pass")
|
||||
}
|
||||
}
|
||||
|
||||
// TestONNXPersonalBoundary — the number that matters, scored against the
|
||||
// embedder homesrv actually runs. Opt-in via MAVEN_ONNX_LIB, exactly like
|
||||
// TestONNXRecall in internal/memory/recalleval.
|
||||
//
|
||||
// Every case here is held out: none of these strings is a seed. The #495
|
||||
// regression is the first row — "что я говорил про бэкапы?" reached SearXNG and
|
||||
// was answered from a Habr article, and no possession word appears in it.
|
||||
func TestONNXPersonalBoundary(t *testing.T) {
|
||||
lib := os.Getenv("MAVEN_ONNX_LIB")
|
||||
if lib == "" {
|
||||
t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing")
|
||||
}
|
||||
dir := filepath.Join("../..", "models/embedder/multilingual-e5-small")
|
||||
emb, err := router.NewONNXEmbedder(filepath.Join(dir, "model_quantized.onnx"), filepath.Join(dir, "tokenizer.json"), lib)
|
||||
if err != nil {
|
||||
t.Skipf("onnx embedder unavailable: %v", err)
|
||||
}
|
||||
defer emb.Close()
|
||||
|
||||
cases := []struct {
|
||||
utterance string
|
||||
personal bool
|
||||
}{
|
||||
{"что я говорил про бэкапы?", true},
|
||||
{"что я сказал вчера про отпуск", true},
|
||||
{"я писал что-нибудь про сервер", true},
|
||||
{"я упоминал про конференцию?", true},
|
||||
{"что я отмечал по поводу переезда", true},
|
||||
{"я рассказывал тебе про новую работу?", true},
|
||||
{"во сколько у меня встреча", true},
|
||||
{"когда мой следующий отпуск", true},
|
||||
{"what did i say about backups", true},
|
||||
{"did i tell you about the doctor", true},
|
||||
{"как я говорил, почему небо синее", false},
|
||||
{"как уже я говорил, какая столица франции", false},
|
||||
{"почему трава зелёная", false},
|
||||
{"столица франции", false},
|
||||
{"как мне сварить борщ", false},
|
||||
{"что мне посмотреть вечером", false},
|
||||
{"я хочу узнать про рим", false},
|
||||
{"кто такой гагарин", false},
|
||||
{"how do i boil an egg", false},
|
||||
}
|
||||
|
||||
h := &reactiveHandler{embedder: emb}
|
||||
ctx := context.Background()
|
||||
wrong := 0
|
||||
for _, c := range cases {
|
||||
vec, err := router.EmbedQuery(ctx, emb, c.utterance)
|
||||
if err != nil {
|
||||
t.Fatalf("embed %q: %v", c.utterance, err)
|
||||
}
|
||||
turn := &queryTurn{dec: router.Decision{Utterance: c.utterance}, vec: vec}
|
||||
got := h.isPersonalTurn(ctx, turn)
|
||||
p, w, ok := h.boundary.score(vec)
|
||||
if !ok {
|
||||
t.Fatal("seeds did not load with a working embedder")
|
||||
}
|
||||
if got != c.personal {
|
||||
wrong++
|
||||
t.Errorf("%q: personal=%v want %v (personal %.4f world %.4f)", c.utterance, got, c.personal, p, w)
|
||||
}
|
||||
t.Logf("personal=%-5v personal %.4f world %.4f delta %+.4f %s", got, p, w, p-w, c.utterance)
|
||||
}
|
||||
t.Logf("personal boundary: %d/%d held-out utterances correct", len(cases)-wrong, len(cases))
|
||||
}
|
||||
@@ -76,6 +76,10 @@ type reactiveHandler struct {
|
||||
tts tts.Synthesizer
|
||||
router *router.Router
|
||||
embedder router.Embedder // reused for note write/query (same model as the classifier)
|
||||
// boundary — the embedded seed sets behind the personal boundary
|
||||
// (personalboundary.go). Zero value is usable and loads on first query;
|
||||
// with no embedder it never loads and the boundary uses personalMarkers.
|
||||
boundary personalBoundary
|
||||
// api — the CoreAPI the handler reads and writes through. Wired with the
|
||||
// bare store adapter and UPGRADED by main once the daemonAPI exists; see
|
||||
// upgradeAPI.
|
||||
|
||||
@@ -20,6 +20,7 @@
|
||||
"bin_path": "llama-server",
|
||||
"n_gpu_layers": 99,
|
||||
"n_ctx": 4096,
|
||||
"cache_ram_mib": 512,
|
||||
"timeout": "60s",
|
||||
"llm_nudges": false
|
||||
},
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
# Where the resident model's 7.9GB of RSS goes (2026-08-03, homesrv)
|
||||
|
||||
Measured for Vikunja #499. The deployed llama-server held 7.9GB RSS for a 1.1GB
|
||||
model file. Half a gigabyte of it was in swap, on a box that also runs
|
||||
whisper.cpp, piper and the embedder.
|
||||
|
||||
## Method
|
||||
|
||||
`maven-mavend-1` was stopped for the measurement, with the owner's approval.
|
||||
Its own binary then ran on the host with the exact deployed command line. That
|
||||
binary is `/opt/maven/bin/llama-server`, version `1 (4c65955)`, a Vulkan build.
|
||||
|
||||
```sh
|
||||
llama-server -m /mnt/hdd1/llms/qwen3/Qwen3-1.7B-UD-Q4_K_XL.gguf \
|
||||
--host 127.0.0.1 --port 18099 -c 4096 -ngl 99 --no-webui
|
||||
```
|
||||
|
||||
RSS was read from `/proc/<pid>/status` after load and after each of 8 distinct
|
||||
1521-token prompts. `smaps` of the deployed process was read first, from inside
|
||||
the container, since the host user cannot read another user's maps.
|
||||
|
||||
## The cause: the prompt cache, not the weights and not the offload
|
||||
|
||||
The startup log says it outright:
|
||||
|
||||
```text
|
||||
srv load_model: prompt cache is enabled, size limit: 8192 MiB
|
||||
srv llama_server: n_parallel is set to auto, using n_parallel = 4 and kv_unified = true
|
||||
```
|
||||
|
||||
The server saves the full KV state of every idle slot it evicts. It keeps up to
|
||||
8GiB of those states in host RAM (llama.cpp PR 16391). One saved prompt of 1521
|
||||
tokens costs 166.377 MiB. That is 112 kiB per token, exactly Qwen3-1.7B's KV
|
||||
footprint (28 layers x 2 x 1024 dims x 2 bytes).
|
||||
|
||||
RSS at rest, and per distinct prompt:
|
||||
|
||||
| Prompts served | RSS, default | RSS, `--cache-ram 512` |
|
||||
|---|---|---|
|
||||
| 0 (just loaded) | 443 MB | 411 MB |
|
||||
| 1 | 445 MB | 411 MB |
|
||||
| 4 | 958 MB | 929 MB |
|
||||
| 8 | 1641 MB | 932 MB |
|
||||
|
||||
Uncapped, RSS climbs about 170MB per distinct prompt and does not stop until
|
||||
the 8GiB limit. Capped at 512 MiB it plateaus at 932MB from the fourth prompt
|
||||
on, with the cache holding steady at `3 prompts, 499.132 MiB` and evicting.
|
||||
|
||||
The 7.9GB on the running daemon was that climb, weeks of it. Its `smaps` showed
|
||||
one 6.03GB anonymous mapping at 5.32GB resident plus a 1.45GB mapping at 1.27GB
|
||||
resident, and only 30MB of file-backed RSS.
|
||||
|
||||
## The task's leading guess was wrong
|
||||
|
||||
`-ngl 99` on the Vega iGPU costs almost no process RSS. A freshly loaded server
|
||||
has 95MB of anonymous RSS in total. RADV allocates device memory through the
|
||||
kernel, outside the process, and the log sees 8202 MiB free on `Vulkan0`. The
|
||||
weights are mmapped and file-backed, so they are evictable and do not pin RSS. The logit buffer is not visible in the numbers above at all.
|
||||
|
||||
## Decision
|
||||
|
||||
`--cache-ram 512` is now the default, wired as `phraser.cache_ram_mib` and set
|
||||
in `deploy/mavend.json`. 512 MiB caps total RSS near 1GB, an eighth of what the
|
||||
box carried. It still holds three of the 1521-token probes above. Maven's real
|
||||
routing and phrasing prompts are much shorter, so it holds more of those than
|
||||
the table suggests. `-c 4096` is untouched, as #499
|
||||
required. A negative `cache_ram_mib` passes no flag, for a llama-server too old
|
||||
to know it.
|
||||
|
||||
Not changed: `n_parallel = 4`. With `kv_unified = true` the four slots share one
|
||||
4096-token KV cache, so they do not multiply it.
|
||||
|
||||
The other half of #499 was that none of these lines were reachable. mavend
|
||||
scraped llama-server's stderr for the listen line and discarded it, and never
|
||||
piped stdout at all. Both streams now go to mavend's log with a `llama:` prefix.
|
||||
The last 12 startup lines go into the error when the server dies before it
|
||||
listens.
|
||||
|
||||
## Deployed
|
||||
|
||||
The `mavenai:latest` image was rebuilt and `maven-mavend-1` recreated the same
|
||||
day. The daemon's own log now carries the child's startup, it reads
|
||||
`prompt cache is enabled, size limit: 512 MiB`, and the resident server sat at
|
||||
439MB RSS after load and 613MB after one served turn.
|
||||
@@ -0,0 +1,46 @@
|
||||
# Personal boundary, seed scoring vs possession markers, 2026-08-03
|
||||
|
||||
Vikunja #495. `что я говорил про бэкапы?` walked past the personal boundary into
|
||||
SearXNG and came back answered from a Habr article. The boundary matched
|
||||
possession words only, so a first-person speech verb was not a personal
|
||||
question.
|
||||
|
||||
## What changed
|
||||
|
||||
The boundary now scores the turn's query vector against two frozen seed sets.
|
||||
It claims the turn when the personal side is nearer than the world side. Seeds
|
||||
and code are in `cmd/mavend/personalboundary.go`. The possession markers stay as
|
||||
the offline floor for a handler with no embedder.
|
||||
|
||||
A regex speech class was written first and dropped. Russian gives every verb a
|
||||
dozen surface forms, and the "как я говорил, ..." preamble list has no end. Each
|
||||
form the lexicon missed was one more question reaching the world.
|
||||
|
||||
## Measurement
|
||||
|
||||
Embedder: multilingual-e5-small int8, the one homesrv runs. Both sides are
|
||||
embedded on the query side. Cases are held out, none of them a seed. `make test`
|
||||
runs the offline part. The scored part is opt-in through `MAVEN_ONNX_LIB`, like
|
||||
`TestONNXRecall`.
|
||||
|
||||
19/19 held-out utterances correct (TestONNXPersonalBoundary)
|
||||
|
||||
true positive margins +0.014 to +0.089
|
||||
nearest true negative -0.005 ("кто такой гагарин")
|
||||
|
||||
One case missed during the first pass and is not held out any more: `as i said,
|
||||
what is the population of india`, +0.008 to the personal side. It is a world seed
|
||||
now.
|
||||
|
||||
The gate is the sign of the difference and nothing tighter. The margins are too
|
||||
thin for a threshold. The asymmetry favours claiming: a false claim costs one
|
||||
honest "не знаю", a false pass sends his life to an upstream engine.
|
||||
|
||||
`make eval-recall` unchanged, 18/27 answered at gate 0.55. Recall does not touch
|
||||
this path.
|
||||
|
||||
## Not verified
|
||||
|
||||
The live probe on the deployed box. The daemon was not rebuilt in this session.
|
||||
The reply to `что я говорил про бэкапы?` with no matching note is still untested
|
||||
against a real SearXNG.
|
||||
@@ -0,0 +1,65 @@
|
||||
# Recall topic veto, what it costs and what it buys, 2026-08-03
|
||||
|
||||
Vikunja #496. The task asked for a cross-language fix. Skip the topic veto in
|
||||
`memory.RecallAllowed` when the question and the hit are in different scripts.
|
||||
An English question would then stop losing a Russian note.
|
||||
|
||||
No such case exists. No fixture case puts the question and its wanted note in
|
||||
different scripts. The case the task named is not one either.
|
||||
|
||||
en-hard-024
|
||||
query "what fixed the screen problem"
|
||||
note "the flicker went away once i swapped the display cable"
|
||||
|
||||
Both are English. It is a paraphrase failure, not a language failure. A script
|
||||
test would not have changed a single case, and neither would a bilingual stem
|
||||
map.
|
||||
|
||||
## What the veto is worth today
|
||||
|
||||
Measured with the real embedder, multilingual-e5-small int8, gate 0.55, margin
|
||||
0.008. The first row is the veto as it ships. The second is `RecallAllowed`
|
||||
forced to true.
|
||||
|
||||
| | cases passing | answered | false recall | silenced by gate |
|
||||
|---|---|---|---|---|
|
||||
| veto on | 22/32 | 17/27 | 0/5 | 2 |
|
||||
| veto off | 22/32 | 18/27 | 1/5 | 1 |
|
||||
|
||||
The pass count does not move. The veto trades one true recall for one false one.
|
||||
It costs `en-hard-024` and it buys `ru-silent-029`:
|
||||
|
||||
ru-silent-029
|
||||
query "во сколько отходит поезд"
|
||||
note "погулял вдоль реки" 0.835, margin 0.019
|
||||
|
||||
The second case counted as silenced by the gate is `ru-home-026` at margin
|
||||
0.001, which the margin gate stops. The veto has nothing to do with it.
|
||||
|
||||
## Why no lexical rule separates the two
|
||||
|
||||
`en-hard-024` and `ru-silent-029` are in the same lexical class. Both questions
|
||||
share zero content words with their hit, and neither carries a first-person
|
||||
marker. The scores sit on top of each other, 0.826 against 0.835, and so do the
|
||||
margins, 0.023 against 0.019. Only one thing separates them. A screen problem
|
||||
and a swapped display cable are the same event. A train and a river walk are
|
||||
not. The embedder scores that difference at nine thousandths.
|
||||
|
||||
So the signal is semantic and the gate is lexical. Any rule cheap enough to sit
|
||||
in `RecallAllowed` and strong enough to recover `en-hard-024` also re-admits
|
||||
`ru-silent-029`, which puts false recall back to 1/5.
|
||||
|
||||
One near-miss rule was tried on paper and rejected: let the veto pass when the
|
||||
hit itself is first person. It works on these two, because the English note says
|
||||
"i swapped" and the Russian note says only "погулял". It is backwards as a
|
||||
principle. A first-person note is exactly the personal note the veto keeps away
|
||||
from a world question. The rule would weaken the veto where it was designed to
|
||||
bite. It survives here only because Russian drops the pronoun.
|
||||
|
||||
## Decision
|
||||
|
||||
Accept the loss. `en-hard-024` stays silenced and false recall stays 0/5.
|
||||
|
||||
The way out is a reranker, not a longer word list. Recall@3 is 85.2% against
|
||||
recall@1 at 70.4%, so the right note is usually in the returned set and ranked
|
||||
wrong. That is where the remaining points are, and it is not this task.
|
||||
@@ -1276,6 +1276,12 @@ type PhraserConfig struct {
|
||||
NCtx int `json:"n_ctx,omitempty"`
|
||||
Timeout Duration `json:"timeout,omitempty"`
|
||||
|
||||
// CacheRAMMiB bounds llama-server's prompt cache. Omitted ⇒ 512 MiB, which
|
||||
// is what keeps the resident model near 1 GB of RSS instead of the 7.9 GB
|
||||
// measured on 2026-08-03. Set it to -1 to pass no flag at all and let the
|
||||
// server apply its own 8 GiB default. See phraser.Config.CacheRAMMiB.
|
||||
CacheRAMMiB int `json:"cache_ram_mib,omitempty"`
|
||||
|
||||
// LLMNudges — let the model word nudges again. Off by default: nudges are
|
||||
// worded from hand-written Russian templates now (the model broke the
|
||||
// persona and invented units). Chat, query and reminder phrasing always go
|
||||
|
||||
@@ -60,6 +60,14 @@ var firstPerson = map[string]bool{
|
||||
// kill one false one. A question about his own life keeps the embedder alone
|
||||
// as its judge. A question about the world has to name something the memory
|
||||
// actually mentions.
|
||||
//
|
||||
// The veto's price was re-measured on 2026-08-03 (#496,
|
||||
// docs/evals/2026-08-03-recall-topic-veto.md). It costs one true recall and
|
||||
// buys one false one, and the fixture pass count is the same either way. The
|
||||
// lost case is an English paraphrase, not the cross-language loss it was
|
||||
// reported as, and the fixture has no cross-language case at all. Do not add a
|
||||
// script test or a bilingual stem map for it — both are no-ops here. The
|
||||
// separating signal is semantic and belongs in a reranker, not in this file.
|
||||
func RecallAllowed(query, text string) bool {
|
||||
if mentionsHim(query) {
|
||||
return true
|
||||
|
||||
@@ -31,6 +31,22 @@ func TestRecallAllowed(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The known cost of the veto and the thing that pays for it, both measured on
|
||||
// the held-out fixture with the real embedder (#496,
|
||||
// docs/evals/2026-08-03-recall-topic-veto.md). The two are one lexical class:
|
||||
// zero shared content words, no first-person marker, scores 0.826 against 0.835
|
||||
// and margins 0.023 against 0.019. Recovering the first re-admits the second,
|
||||
// which puts false recall back to 1/5. Anyone loosening the veto has to move
|
||||
// the first line without moving the second.
|
||||
func TestRecallVetoTradeIsPinned(t *testing.T) {
|
||||
if RecallAllowed("what fixed the screen problem", "the flicker went away once i swapped the display cable") {
|
||||
t.Error("en-hard-024 is expected to stay vetoed — if this passes now, re-measure false recall before celebrating")
|
||||
}
|
||||
if RecallAllowed("во сколько отходит поезд", "погулял вдоль реки") {
|
||||
t.Error("ru-silent-029 must stay vetoed — this is the false recall the veto exists to stop")
|
||||
}
|
||||
}
|
||||
|
||||
// A question made only of filler has no topic word to match on, and the score
|
||||
// gate is then the only judge it can have.
|
||||
func TestRecallAllowedFallsBackWhenNothingToCompare(t *testing.T) {
|
||||
|
||||
+102
-34
@@ -1,6 +1,7 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
@@ -8,6 +9,7 @@ import (
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strings"
|
||||
@@ -83,6 +85,19 @@ type Config struct {
|
||||
NCtx int
|
||||
Timeout time.Duration
|
||||
|
||||
// CacheRAMMiB bounds llama-server's prompt cache, which is what actually ate
|
||||
// this box. Measured on homesrv 2026-08-03: the server's own default limit is
|
||||
// 8192 MiB, it stores the full KV state of every idle slot it evicts (112 kiB
|
||||
// per token, so 166 MiB for one 1521-token prompt), and RSS climbed by that
|
||||
// much per distinct prompt until it hit 7.9 GB and half a gigabyte went to
|
||||
// swap. Weights are only 1.1 GB and mmapped, and -ngl 99 costs almost no RSS
|
||||
// because RADV keeps device memory outside the process.
|
||||
//
|
||||
// 0 ⇒ the flag is not passed and the server's own 8 GiB default applies. That
|
||||
// is the escape hatch for a llama-server too old to know --cache-ram, not a
|
||||
// recommendation. See docs/evals/2026-08-03-llama-prompt-cache.md.
|
||||
CacheRAMMiB int
|
||||
|
||||
// ContextBlock renders the shared context block (who he is, how to
|
||||
// address him, the time) fresh for each turn. See internal/persona.
|
||||
// nil ⇒ no block, the prompts stand alone.
|
||||
@@ -116,7 +131,9 @@ func DefaultConfig(modelPath string) Config {
|
||||
Listen: "127.0.0.1:0",
|
||||
NGpuLayers: -1,
|
||||
NCtx: 2048,
|
||||
Timeout: 30 * time.Second,
|
||||
// 512 MiB caps total RSS near 1 GB and still holds several recent prompts.
|
||||
CacheRAMMiB: 512,
|
||||
Timeout: 30 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,8 +242,10 @@ func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
|
||||
return p, nil
|
||||
}
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
p := &llamaProc{}
|
||||
// llamaArgs is the command line for one resident server. It is a function and
|
||||
// not an inline literal because kill-maven.sh's orphan sweep matches against
|
||||
// this exact line, and a test pins the two together.
|
||||
func llamaArgs(cfg Config) []string {
|
||||
args := []string{
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
@@ -235,7 +254,15 @@ func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, args...)
|
||||
if cfg.CacheRAMMiB > 0 {
|
||||
args = append(args, "--cache-ram", fmt.Sprintf("%d", cfg.CacheRAMMiB))
|
||||
}
|
||||
return args
|
||||
}
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
p := &llamaProc{}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...)
|
||||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||||
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
|
||||
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
|
||||
@@ -246,63 +273,104 @@ func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true, Pdeathsig: syscall.SIGKILL}
|
||||
p.cmd = cmd
|
||||
|
||||
stderr, err := cmd.StderrPipe()
|
||||
// One pipe for both streams. llama.cpp writes its buffer sizes, KV-cache
|
||||
// layout and offload lines to stderr and its request log to stdout, and
|
||||
// stdout used to go nowhere at all — so nothing about the model's memory was
|
||||
// diagnosable from a running box. Both ends land in mavend's log now.
|
||||
pr, pw, err := os.Pipe()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("llm: stderr pipe: %w", err)
|
||||
return nil, fmt.Errorf("llm: output pipe: %w", err)
|
||||
}
|
||||
cmd.Stdout = pw
|
||||
cmd.Stderr = pw
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
stderr.Close()
|
||||
pr.Close()
|
||||
pw.Close()
|
||||
return nil, fmt.Errorf("llm: start: %w", err)
|
||||
}
|
||||
// The child holds the only other reference to the write end. Dropping ours
|
||||
// is what makes the reader see EOF when the child dies.
|
||||
pw.Close()
|
||||
|
||||
portCh := make(chan string, 1)
|
||||
errCh := make(chan error, 1)
|
||||
tail := &lineTail{}
|
||||
p.wg.Add(1)
|
||||
go func() {
|
||||
defer p.wg.Done()
|
||||
buf := make([]byte, 4096)
|
||||
var leftover []byte
|
||||
for {
|
||||
n, err := stderr.Read(buf)
|
||||
if n > 0 {
|
||||
data := append(leftover, buf[:n]...)
|
||||
lines := bytes.Split(data, []byte("\n"))
|
||||
for _, line := range lines[:len(lines)-1] {
|
||||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||||
addr := string(m[1])
|
||||
portCh <- addr
|
||||
close(portCh)
|
||||
}
|
||||
defer pr.Close()
|
||||
sc := bufio.NewScanner(pr)
|
||||
// llama.cpp prints one prompt per line and a prompt can be long.
|
||||
sc.Buffer(make([]byte, 0, 64*1024), 1024*1024)
|
||||
listening := false
|
||||
for sc.Scan() {
|
||||
line := sc.Bytes()
|
||||
log.Printf("llama: %s", line)
|
||||
if !listening {
|
||||
tail.add(string(line))
|
||||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||||
listening = true
|
||||
portCh <- string(m[1])
|
||||
close(portCh)
|
||||
}
|
||||
leftover = lines[len(lines)-1]
|
||||
}
|
||||
if err != nil {
|
||||
errCh <- err
|
||||
return
|
||||
}
|
||||
}
|
||||
err := sc.Err()
|
||||
if err == nil {
|
||||
err = io.EOF
|
||||
}
|
||||
errCh <- err
|
||||
}()
|
||||
|
||||
fail := func(err error) (*llamaProc, error) {
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, err
|
||||
}
|
||||
select {
|
||||
case addr := <-portCh:
|
||||
p.base = addr
|
||||
return p, nil
|
||||
case err := <-errCh:
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, fmt.Errorf("llm: server output: %w", err)
|
||||
// The tail is the whole diagnosis when the server dies during load: bare
|
||||
// "EOF" never said which layer or which allocation it choked on.
|
||||
return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String()))
|
||||
case <-ctx.Done():
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, ctx.Err()
|
||||
return fail(ctx.Err())
|
||||
case <-time.After(60 * time.Second):
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, fmt.Errorf("llm: server did not start within 60s")
|
||||
return fail(fmt.Errorf("llm: server did not start within 60s; last output: %s", tail.String()))
|
||||
}
|
||||
}
|
||||
|
||||
// lineTail keeps the last few startup lines so a server that dies before it
|
||||
// listens can say why in the error, not just "EOF". Written by the reader
|
||||
// goroutine and read by whoever gives up on startup, so it takes a lock.
|
||||
type lineTail struct {
|
||||
mu sync.Mutex
|
||||
lines []string
|
||||
}
|
||||
|
||||
const lineTailMax = 12
|
||||
|
||||
func (t *lineTail) add(line string) {
|
||||
t.mu.Lock()
|
||||
defer t.mu.Unlock()
|
||||
t.lines = append(t.lines, line)
|
||||
if len(t.lines) > lineTailMax {
|
||||
t.lines = t.lines[len(t.lines)-lineTailMax:]
|
||||
}
|
||||
}
|
||||
|
||||
func (t *lineTail) String() string {
|
||||
t.mu.Lock()
|
||||
defer t.mu.Unlock()
|
||||
if len(t.lines) == 0 {
|
||||
return "(no output)"
|
||||
}
|
||||
return strings.Join(t.lines, " | ")
|
||||
}
|
||||
|
||||
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
||||
// the model is swapped, so callers that cache it must register an observer
|
||||
// (OnSwap) rather than keeping the string forever.
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
@@ -59,6 +61,19 @@ func TestExtractPort(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The prompt cache is what ate 6.8GB of the deployed server's RSS, so the cap
|
||||
// has to reach the command line, and the opt-out has to leave it off.
|
||||
func TestLlamaArgsCapsPromptCache(t *testing.T) {
|
||||
cfg := DefaultConfig("/m.gguf")
|
||||
if got := strings.Join(llamaArgs(cfg), " "); !strings.Contains(got, "--cache-ram 512") {
|
||||
t.Errorf("default args = %q, want --cache-ram 512", got)
|
||||
}
|
||||
cfg.CacheRAMMiB = 0
|
||||
if got := strings.Join(llamaArgs(cfg), " "); strings.Contains(got, "--cache-ram") {
|
||||
t.Errorf("args with the cap off = %q, want no --cache-ram flag", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartLlamaProcScrapesPortAndReaps(t *testing.T) {
|
||||
bin := fakeLlama(t, listensThenSleeps)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
@@ -86,6 +101,48 @@ func TestStartLlamaProcScrapesPortAndReaps(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// captureLog redirects the standard logger for the duration of a test and
|
||||
// returns what was written to it.
|
||||
func captureLog(t *testing.T) *bytes.Buffer {
|
||||
t.Helper()
|
||||
var buf bytes.Buffer
|
||||
old := log.Writer()
|
||||
flags := log.Flags()
|
||||
log.SetOutput(&buf)
|
||||
log.SetFlags(0)
|
||||
t.Cleanup(func() { log.SetOutput(old); log.SetFlags(flags) })
|
||||
return &buf
|
||||
}
|
||||
|
||||
// The child's buffer-size, KV-cache and offload lines are the only way to
|
||||
// account for its memory on a running box, and they used to be dropped: stderr
|
||||
// was scraped for the listen line and thrown away, stdout was never piped.
|
||||
func TestStartLlamaProcForwardsChildOutput(t *testing.T) {
|
||||
buf := captureLog(t)
|
||||
bin := fakeLlama(t, `echo "load_tensors: Vulkan0 model buffer size = 1053.34 MiB" >&2
|
||||
echo "llama_context: KV self size = 448.00 MiB"
|
||||
`+listensThenSleeps)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
p, err := startLlamaProc(ctx, testCfg(bin))
|
||||
if err != nil {
|
||||
t.Fatalf("startLlamaProc: %v", err)
|
||||
}
|
||||
p.cancel = cancel
|
||||
defer p.Close()
|
||||
|
||||
got := buf.String()
|
||||
for _, want := range []string{
|
||||
"llama: load_tensors: Vulkan0 model buffer size = 1053.34 MiB", // stderr
|
||||
"llama: llama_context: KV self size = 448.00 MiB", // stdout, previously discarded
|
||||
} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("log missing %q\nlog was:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartLlamaProcFailureArms(t *testing.T) {
|
||||
t.Run("binary missing", func(t *testing.T) {
|
||||
cfg := testCfg(filepath.Join(t.TempDir(), "does-not-exist"))
|
||||
@@ -96,13 +153,18 @@ func TestStartLlamaProcFailureArms(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("server exits without listening", func(t *testing.T) {
|
||||
// stderr closes, so the reader goroutine reports EOF on errCh.
|
||||
// stderr closes, so the reader goroutine reports EOF on errCh. The error
|
||||
// must carry the child's last words: bare "EOF" named no cause.
|
||||
captureLog(t)
|
||||
bin := fakeLlama(t, `echo "ggml_vulkan: no device" >&2
|
||||
exit 1`)
|
||||
_, err := startLlamaProc(context.Background(), testCfg(bin))
|
||||
if err == nil || !strings.Contains(err.Error(), "llm: server output") {
|
||||
t.Fatalf("err = %v, want the server-output arm", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "ggml_vulkan: no device") {
|
||||
t.Errorf("err = %v, want the child's last output in it", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("context cancelled during startup", func(t *testing.T) {
|
||||
@@ -241,15 +303,7 @@ func TestKillMavenScriptMatchesRealCommandLine(t *testing.T) {
|
||||
// startLlamaProc that breaks the sweep fails here instead of on the box.
|
||||
cfg := DefaultConfig("/opt/maven/models/llm/Qwen3-1.7B-UD-Q4_K_XL.gguf")
|
||||
cfg.NCtx, cfg.NGpuLayers = 4096, 99
|
||||
cmdline := strings.Join([]string{
|
||||
cfg.BinPath,
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
"--port", extractPort(cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}, " ")
|
||||
cmdline := cfg.BinPath + " " + strings.Join(llamaArgs(cfg), " ")
|
||||
if !pat.MatchString(cmdline) {
|
||||
t.Fatalf("kill-maven.sh pattern %q does not match %q — orphans would leak", m[1], cmdline)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user