Compare commits
73 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0f361d2034 | |||
| 19242ef73b | |||
| 26d6d71588 | |||
| df58710141 | |||
| 442d3ec08e | |||
| e481ad4930 | |||
| 3d8224fb04 | |||
| 990a4a99e9 | |||
| 5b622389c5 | |||
| a820a95ebb | |||
| ea9c746852 | |||
| d60a51c9e7 | |||
| 9aabb01e2a | |||
| 2815adee03 | |||
| 9f51596e2f | |||
| 87d176153a | |||
| 7d4b4ad736 | |||
| 569991bb15 | |||
| c915115096 | |||
| 908d92a7e8 | |||
| 43f2c37538 | |||
| 6d3f5b5b01 | |||
| eda1112f3b | |||
| 71041029e2 | |||
| 35018226ef | |||
| 8833a9c76b | |||
| 6c07409452 | |||
| 1c2541f7d6 | |||
| 9e25f18a3e | |||
| 197897516e | |||
| 767748720a | |||
| 58051b5af1 | |||
| f9b2391a8b | |||
| f229795cea | |||
| 6e5364a0ed | |||
| 86817d6d06 | |||
| 0fc2e3a18a | |||
| 453919db20 | |||
| ad60e10e95 | |||
| 1528697287 | |||
| dbdab2d570 | |||
| b9371dcac6 | |||
| 62c2e92ec0 | |||
| aec94eb2e8 | |||
| 4dfe106fe3 | |||
| 2e0e2fd0bb | |||
| f3fa6b353a | |||
| 6645f64c3e | |||
| f10e0068dd | |||
| 9b124d9194 | |||
| 12530c8a95 | |||
| 51256c4c9a | |||
| 76481c2736 | |||
| bcc2305cd0 | |||
| 0ceeac8df4 | |||
| 4fae13af75 | |||
| 774217199e | |||
| 2db59d52a7 | |||
| 92d5fd580c | |||
| edeef19ff0 | |||
| 018f7a6f47 | |||
| eca41798bd | |||
| cc423567e7 | |||
| 8088ef9e00 | |||
| 666b924d29 | |||
| e52c616592 | |||
| 2b97bac51e | |||
| ab42db2b87 | |||
| 94d553570d | |||
| 2e97b905b4 | |||
| fbcca449be | |||
| 2076e4a788 | |||
| 30eb6add1b |
@@ -9,6 +9,7 @@
|
||||
/mavwaked
|
||||
/mavmaild
|
||||
/mavupdate
|
||||
/mavgpud
|
||||
|
||||
# Certs (private keys, don't commit)
|
||||
certs/
|
||||
|
||||
@@ -36,6 +36,13 @@ stays on homesrv permanently, because it backs that floor. Read `docs/offload.md
|
||||
touching a daemon seam or adding a model caller. Vikunja #483 is the umbrella, #484 to #487
|
||||
are the work.
|
||||
|
||||
Both halves are wired as of 2026-08-03. Routing and replies prefer the workstation silently
|
||||
through `modelSeam`; nudge and reminder phrasing prefer it silently inside the phraser. A
|
||||
world question goes through `LLMPhraser.PhraseWorld` and names the gap when the card is not
|
||||
free — `worldGap` in `cmd/mavend/worldmodel.go`, which he hears instead of an invented
|
||||
answer. A box with no `workstation` block behaves exactly as it did before the seam: naming
|
||||
a gap requires a gap. The offload table in `docs/offload.md` says which caller is which.
|
||||
|
||||
## Build & test
|
||||
|
||||
CGO daemons (`mavend`, `mavsttd`, `mavttsd`, `mavenclient`) need the vendored toolchain
|
||||
@@ -143,7 +150,15 @@ seed additions, both of which now score inside the classifier baseline. Qwen3-1.
|
||||
not a doubling, and the trade is worth re-arguing rather than assuming. **The ≈2.7s figure
|
||||
that stood here until 2026-08-02 was contention, not the model.** See `docs/evals/2026-07-31-routing.md` line 61, which measures the LLM router at
|
||||
p50 825ms / p95 1.2s / max 3.0s and the full cascade at p50 0.80-1.04s. Do not plan latency
|
||||
work off the bakeoff table. `Confidence: 1.0` used to be hardcoded in `llmrouter.go`, so the LLM
|
||||
work off the bakeoff table.
|
||||
|
||||
**The numbers above are the homesrv floor, not the ceiling.** With the workstation up, routing
|
||||
completes through `llm.Pair` against gemma-4-12b and scores **84.4% full / 93.5% intent-only at
|
||||
p50 329ms** — better than the resident model and about 2.5× faster (`docs/evals/2026-08-02-workstation-gemma4-12b.md`,
|
||||
Vikunja #485). The workstation is never assumed up, so both sets of numbers are live. Judge a
|
||||
routing change against the classifier and the resident model, since those are what always answer.
|
||||
|
||||
`Confidence: 1.0` used to be hardcoded in `llmrouter.go`, so the LLM
|
||||
path could never ask for clarification (6/6 refusal cases missed on the fixture) — Vikunja
|
||||
#359. Fixed 31-07-2026 with structural signal (single-token utterance, keyless fact, act with
|
||||
no allowlisted fn) feeding the same stage-3 gate the classifier path already had — see
|
||||
|
||||
@@ -16,11 +16,11 @@ PIPER_BIN := $(shell pwd)/deps/piper/piper
|
||||
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
||||
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
||||
|
||||
.PHONY: simulate stt-fixtures test-stt-golden all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||
.PHONY: simulate stt-fixtures test-stt-golden all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models build-gpud
|
||||
|
||||
all: build
|
||||
|
||||
build: build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav build-mail build-update
|
||||
build: build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav build-mail build-update build-gpud
|
||||
|
||||
build-stt:
|
||||
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||
@@ -59,6 +59,12 @@ build-mail:
|
||||
build-update:
|
||||
$(GO) build $(GOFLAGS) -o mavupdate ./cmd/mavupdate/
|
||||
|
||||
# mavgpud runs on the workstation, not here. It is built with the rest so a
|
||||
# broken supervisor is caught by `make build` on homesrv rather than by the
|
||||
# workstation refusing to serve. Copy the binary over, do not `make deploy` it.
|
||||
build-gpud:
|
||||
$(GO) build $(GOFLAGS) -o mavgpud ./cmd/mavgpud/
|
||||
|
||||
run-web: build-web
|
||||
./mavweb -addr :9200 -voice 127.0.0.1:9100
|
||||
|
||||
|
||||
@@ -57,6 +57,9 @@ var actionHandlers = map[router.Intent]func(*reactiveHandler, context.Context, r
|
||||
func (h *reactiveHandler) actionChat(ctx context.Context, dec router.Decision) string {
|
||||
// Conversational: build history from dialogue session (prior user turns)
|
||||
// and let the LLM respond from general knowledge + context.
|
||||
if h.phraser == nil {
|
||||
return "поговорили."
|
||||
}
|
||||
history := h.chatHistory()
|
||||
reply, err := h.phraser.PhraseChat(ctx, dec.Utterance, history)
|
||||
if err != nil {
|
||||
|
||||
+65
-14
@@ -7,6 +7,7 @@ import (
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
|
||||
// actionFact handles router.IntentFact: persist a tapped self-fact, index
|
||||
@@ -15,14 +16,56 @@ func (h *reactiveHandler) actionFact(ctx context.Context, dec router.Decision) s
|
||||
if !dec.Slots.HasKey {
|
||||
return "не разобрала, что записать — попробуй иначе."
|
||||
}
|
||||
// A question is never a fact about him (#470). "какая последняя версия
|
||||
// языка Go?" used to land here, and the value stored was whatever the
|
||||
// model invented for it, at confidence 1.00, indexed for recall under the
|
||||
// question's own text. Two such rows then claimed seven unrelated world
|
||||
// questions through recall and silently disabled world answering.
|
||||
//
|
||||
// The routing error itself is not fixed here — the answer is to answer.
|
||||
// Sending the turn down the query chain is what he asked for anyway, and
|
||||
// it costs a mis-routed capture nothing: an explicit "запиши ..." is not
|
||||
// question-shaped, so it never takes this branch.
|
||||
if router.IsQuestionShaped(dec.Utterance) {
|
||||
log.Printf("voice: fact write refused, utterance is a question: %q (key %q) — answering as a query",
|
||||
dec.Utterance, dec.Slots.Key)
|
||||
q := dec
|
||||
q.Intent = router.IntentQuery
|
||||
// The key the model extracted is its guess at what to store, not a
|
||||
// fact he has. Left in place, queryFactByKey would read it back and
|
||||
// claim the turn before any real source ran.
|
||||
q.Slots.Key, q.Slots.HasKey = "", false
|
||||
q.Slots.Value = ""
|
||||
return h.actionQuery(ctx, q)
|
||||
}
|
||||
// A complaint is not a fact either (#481). "сеть какая-то медленная" and
|
||||
// "интернет не работает" were stored as `self` rows at confidence 1.00, and
|
||||
// recall reads a self row back later as if it were still true — the same
|
||||
// class of row that outranked live search in #470. The sentence describes a
|
||||
// moment, so she answers it and stores nothing. An explicit "запомни ..."
|
||||
// and anything about him are both left alone by the test.
|
||||
if router.IsTransientComplaint(dec.Utterance) {
|
||||
log.Printf("voice: fact write refused, utterance is a passing complaint: %q (key %q) — answering as chat",
|
||||
dec.Utterance, dec.Slots.Key)
|
||||
c := dec
|
||||
c.Intent = router.IntentChat
|
||||
c.Slots.Key, c.Slots.HasKey = "", false
|
||||
c.Slots.Value = ""
|
||||
return h.actionChat(ctx, c)
|
||||
}
|
||||
now := h.now()
|
||||
req := ipc.WriteFactReq{
|
||||
Ts: now,
|
||||
Kind: "self",
|
||||
Key: dec.Slots.Key,
|
||||
Value: dec.Slots.Value,
|
||||
Source: "tap:voice",
|
||||
Confidence: 1.0,
|
||||
Ts: now,
|
||||
Kind: "self",
|
||||
Key: dec.Slots.Key,
|
||||
Value: dec.Slots.Value,
|
||||
Source: "tap:voice",
|
||||
// Not 1.00 unconditionally any more (#470). A value he said is
|
||||
// evidence; a value the model supplied for words he never said is a
|
||||
// guess, and writing a guess at full confidence is the same mistake
|
||||
// the act path already refuses under "LLM output is not
|
||||
// authorization".
|
||||
Confidence: factConfidence(dec.Utterance, dec.Slots.Value),
|
||||
// Subject: the key doubles as the entity-resolution candidate —
|
||||
// a voice-tapped fact's key is usually the thing/person it's
|
||||
// about ("espresso_machine", "kate"), so queueing it for Nexus
|
||||
@@ -36,17 +79,25 @@ func (h *reactiveHandler) actionFact(ctx context.Context, dec router.Decision) s
|
||||
log.Printf("voice: write fact: %v", err)
|
||||
return "не получилось сохранить факт."
|
||||
}
|
||||
// Index the fact utterance in long-term memory (best-effort, must not
|
||||
// fail the fact write). Facts aren't in the notes table, so this is the
|
||||
// only recall path for them — "когда я пил воду?" reads back from here.
|
||||
// Index the fact in long-term memory (best-effort, must not fail the fact
|
||||
// write). Facts aren't in the notes table, so this is the only recall path
|
||||
// for them — "когда я пил воду?" reads back from here.
|
||||
//
|
||||
// The indexed text is the fact, not the utterance (#493). queryMemory
|
||||
// returns a fact's stored text verbatim, so what goes in here is what he
|
||||
// hears; storing the utterance meant recall answered with his own sentence
|
||||
// rather than the value. The utterance stays alongside as provenance —
|
||||
// readable on /trace, never the answer and never embedded.
|
||||
if h.memStore != nil {
|
||||
if vec, err := router.EmbedPassage(ctx, h.embedder, dec.Utterance); err != nil {
|
||||
text := store.FactRecallText(dec.Slots.Key, dec.Slots.Value)
|
||||
if vec, err := router.EmbedPassage(ctx, h.embedder, text); err != nil {
|
||||
log.Printf("voice: embed fact for memory: %v", err)
|
||||
} else if err := h.memStore.Insert(ctx, "fact:"+dec.Slots.Key+":"+strconv.FormatInt(now.Unix(), 10), vec, map[string]string{
|
||||
"source": "voice",
|
||||
"type": "fact",
|
||||
"text": dec.Utterance,
|
||||
"ts": strconv.FormatInt(now.Unix(), 10),
|
||||
"source": "voice",
|
||||
"type": "fact",
|
||||
"text": text,
|
||||
"utterance": dec.Utterance,
|
||||
"ts": strconv.FormatInt(now.Unix(), 10),
|
||||
}); err != nil {
|
||||
log.Printf("voice: memory insert fact: %v", err)
|
||||
}
|
||||
|
||||
+103
-29
@@ -13,6 +13,7 @@ import (
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
"github.com/kami/maven/internal/morning"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/rss"
|
||||
"github.com/kami/maven/internal/store"
|
||||
@@ -80,11 +81,22 @@ var querySources = []querySource{
|
||||
// matcher requires a task noun or an explicit "что … сделать", so a
|
||||
// date-bearing question still reaches the calendar.
|
||||
{name: "tasks", answer: (*reactiveHandler).queryTasks},
|
||||
// Next to "tasks" and for the same reason: "что требует внимания?" is a
|
||||
// question about the operational state Praxis holds, and it used to fall
|
||||
// through every source to the web search (Vikunja #475). Its matcher needs
|
||||
// an attention marker, and it falls through when Praxis is not configured.
|
||||
{name: "attention", answer: (*reactiveHandler).queryAttention},
|
||||
// Before the recall sources too: "сколько я потратил?" is a question about
|
||||
// the money facts the poller wrote, and the notes pass would otherwise
|
||||
// answer it from whatever he once said about spending. Its matcher needs a
|
||||
// money noun plus an actual ask, so "я потратил весь день" is untouched.
|
||||
{name: "money", answer: (*reactiveHandler).queryMoney},
|
||||
// Also above the recall sources: "что я тебе говорил?" is a question about
|
||||
// the facts he tapped in, and the notes pass would answer it with whatever
|
||||
// note is nearest (Vikunja #456). Its matcher needs both halves of a
|
||||
// history phrase and bails out when he names a topic, so "что я говорил
|
||||
// про сервер" is still recall.
|
||||
{name: "history", answer: (*reactiveHandler).queryHistory},
|
||||
// Before the recall sources and before general knowledge: "что нового?" is
|
||||
// a question about the feeds she reads, and general knowledge would answer
|
||||
// it by inventing news. Its matcher needs a feed noun plus an ask, so
|
||||
@@ -139,6 +151,13 @@ func (h *reactiveHandler) actionQuery(ctx context.Context, dec router.Decision)
|
||||
continue
|
||||
}
|
||||
if reply, ok := src.answer(h, ctx, t); ok {
|
||||
// Which source claimed is the one thing about a query turn that was
|
||||
// invisible from outside: /trace is the nudge-rule trace and carries
|
||||
// no query-source field, so a wrong answer could not be told from a
|
||||
// wrongly-ordered chain (Vikunja #474). Only the name is logged —
|
||||
// the utterance and the answer are already on the voice lines above
|
||||
// and below this one.
|
||||
log.Printf("voice: query claimed by source %q", src.name)
|
||||
return reply
|
||||
}
|
||||
}
|
||||
@@ -277,9 +296,17 @@ func (h *reactiveHandler) queryFeeds(ctx context.Context, t *queryTurn) (string,
|
||||
return "", false
|
||||
}
|
||||
if !h.feedsOn {
|
||||
// Claim the turn rather than fall through: "не читаю ленты" is true, and
|
||||
// letting general knowledge answer "что нового?" would be an invented
|
||||
// news bulletin.
|
||||
// Claim only when nothing below can read the world. The reason this
|
||||
// source used to claim unconditionally was that general knowledge would
|
||||
// answer "что нового?" with an invented news bulletin — true, and it
|
||||
// stopped being the only alternative on 2026-08-02, when live search
|
||||
// took the lead. With SearXNG or the ZIMs configured, "что происходит
|
||||
// в новостях про искусственный интеллект?" has a real answer below,
|
||||
// and a configuration status is the wrong thing to say instead
|
||||
// (Vikunja #474).
|
||||
if h.search != nil || h.kiwix != nil {
|
||||
return "", false
|
||||
}
|
||||
return "я пока не читаю ленты — они не настроены.", true
|
||||
}
|
||||
// By source, not the last 200 notes of any kind: a busy day of voice notes
|
||||
@@ -316,6 +343,15 @@ func (h *reactiveHandler) queryFeeds(ctx context.Context, t *queryTurn) (string,
|
||||
// h.now(), not time.Now(): the handler's clock is the injected one, so this
|
||||
// source can be tested at a fixed time like the rest.
|
||||
func (h *reactiveHandler) queryCalendar(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
// A day word is all this source matches on, so any question that merely
|
||||
// names a day reached it first. "какая сегодня погода в Москве?" answered
|
||||
// "на 02.08.2026 ничего нет." (Vikunja #474). Weather is asked about a day
|
||||
// far more often than the calendar is, and the weather source sits right
|
||||
// below, so the calendar steps aside on weather wording — the same bail-out
|
||||
// queryHome already does for the same reason.
|
||||
if isWeatherQuery(t.dec.Utterance) {
|
||||
return "", false
|
||||
}
|
||||
date, ok := router.ParseCalendarDate(t.dec.Utterance, h.now())
|
||||
if !ok {
|
||||
return "", false
|
||||
@@ -389,6 +425,11 @@ func (h *reactiveHandler) queryWeather(ctx context.Context, t *queryTurn) (strin
|
||||
if errors.Is(err, weather.ErrNotConfigured) {
|
||||
return "погода не настроена.", true
|
||||
}
|
||||
if errors.Is(err, weather.ErrLocationUnknown) {
|
||||
// He named a place and the geocoder does not have it. Saying so beats
|
||||
// reading out the default city's temperature (Vikunja #421).
|
||||
return "не знаю такого города — " + loc + ".", true
|
||||
}
|
||||
if err != nil {
|
||||
log.Printf("voice: weather: %v", err)
|
||||
return "не получилось узнать погоду.", true
|
||||
@@ -433,6 +474,14 @@ func (h *reactiveHandler) queryMemory(ctx context.Context, t *queryTurn) (string
|
||||
return "", false
|
||||
}
|
||||
text := hit.Meta["text"]
|
||||
// The score cleared the gate and the topic still has to match (#470). A
|
||||
// note about his slow network scored high enough to answer "почему небо
|
||||
// синее?", because the right-note and must-be-silent score ranges overlap
|
||||
// and no threshold sits between them.
|
||||
if !memory.RecallAllowed(t.dec.Utterance, text) {
|
||||
log.Printf("voice: recall %q rejected for %q: a world question and no shared topic word", text, t.dec.Utterance)
|
||||
return "", false
|
||||
}
|
||||
// A note is phrased in Maven's voice; a fact is read back as it was
|
||||
// stored.
|
||||
if hit.Meta["type"] == "note" {
|
||||
@@ -468,6 +517,12 @@ func (h *reactiveHandler) queryNotes(ctx context.Context, t *queryTurn) (string,
|
||||
if !memory.ConfidentScores(noteScores, h.queryMinScore, h.queryMinMargin) {
|
||||
return "", false
|
||||
}
|
||||
// Same topic veto as queryMemory above: the best note must be about what
|
||||
// he asked, not merely the nearest vector in the index.
|
||||
if !memory.RecallAllowed(t.dec.Utterance, notes[0].Text) {
|
||||
log.Printf("voice: note %q rejected for %q: a world question and no shared topic word", notes[0].Text, t.dec.Utterance)
|
||||
return "", false
|
||||
}
|
||||
texts := make([]string, len(notes))
|
||||
for i, n := range notes {
|
||||
texts[i] = n.Text
|
||||
@@ -523,10 +578,7 @@ func (h *reactiveHandler) queryWeb(ctx context.Context, t *queryTurn) (string, b
|
||||
// the question he actually asked. She answers the question, she does not
|
||||
// recite the page.
|
||||
snippet := page.Title + "\n" + crawl.TrimRunes(page.Text, webPageContextRunes)
|
||||
reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
|
||||
if perr != nil {
|
||||
log.Printf("voice: web: phrase: %v", perr)
|
||||
}
|
||||
reply := h.phraseSource(ctx, "web", t.dec.Utterance, []string{snippet})
|
||||
if reply == "" {
|
||||
// No phraser (or it failed): read back the top of the page rather than
|
||||
// pretend the fetch did not happen.
|
||||
@@ -592,14 +644,7 @@ func (h *reactiveHandler) querySearch(ctx context.Context, t *queryTurn) (string
|
||||
// question he asked, not something to recite. The trim is one budget over the
|
||||
// joined block, so a long first snippet cannot crowd out the rest.
|
||||
evidence := crawl.TrimRunes(strings.Join(resp.Snippets(), "\n"), h.search.runes)
|
||||
var reply string
|
||||
if h.phraser != nil {
|
||||
var perr error
|
||||
reply, perr = h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{evidence})
|
||||
if perr != nil {
|
||||
log.Printf("voice: search: phrase: %v", perr)
|
||||
}
|
||||
}
|
||||
reply := h.phraseSource(ctx, "search", t.dec.Utterance, []string{evidence})
|
||||
if reply == "" {
|
||||
// No phraser, or it failed. Read back the best evidence rather than
|
||||
// pretend the search did not happen.
|
||||
@@ -680,14 +725,7 @@ func (h *reactiveHandler) queryKiwix(ctx context.Context, t *queryTurn) (string,
|
||||
// Handed over the same way a note or a page is: context for the question he
|
||||
// asked, not something to recite.
|
||||
snippet := top.Title + "\n" + crawl.TrimRunes(page.Text, h.kiwix.runes)
|
||||
var reply string
|
||||
if h.phraser != nil {
|
||||
var perr error
|
||||
reply, perr = h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
|
||||
if perr != nil {
|
||||
log.Printf("voice: kiwix: phrase: %v", perr)
|
||||
}
|
||||
}
|
||||
reply := h.phraseSource(ctx, "kiwix", t.dec.Utterance, []string{snippet})
|
||||
if reply == "" {
|
||||
// No phraser, or it failed. Read back the best hit rather than pretend
|
||||
// the search did not happen.
|
||||
@@ -717,7 +755,7 @@ func (h *reactiveHandler) queryKiwix(ctx context.Context, t *queryTurn) (string,
|
||||
// not be sent to an upstream engine at all. The guard closes both holes with
|
||||
// the same test.
|
||||
func (h *reactiveHandler) queryPersonal(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
if !isPersonalQuery(t.dec.Utterance) {
|
||||
if !h.isPersonalTurn(ctx, t) {
|
||||
return "", false
|
||||
}
|
||||
log.Printf("voice: %q is about him and his own data did not answer it; not asking the world", t.dec.Utterance)
|
||||
@@ -742,7 +780,9 @@ var personalMarkers = []*regexp.Regexp{
|
||||
regexp.MustCompile(`(?i)\bdid\s+i\b`),
|
||||
}
|
||||
|
||||
// isPersonalQuery reports whether the utterance asks about something of his.
|
||||
// isPersonalQuery — the offline floor under the boundary. Possession only, and
|
||||
// deliberately still narrow: it answers when there is no embedder to ask, and a
|
||||
// broad guess made blind is worse than a narrow one.
|
||||
func isPersonalQuery(utterance string) bool {
|
||||
if utterance == "" {
|
||||
return false
|
||||
@@ -755,11 +795,45 @@ func isPersonalQuery(utterance string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// queryGeneral — general knowledge from the phraser, the last source before
|
||||
// giving up. It always claims: either the model answers or Maven says she
|
||||
// doesn't know.
|
||||
// isPersonalTurn — the boundary test. The seeds decide when the embedder is
|
||||
// there, which is every deployed box; the possession markers are the floor
|
||||
// underneath, for a handler with no embedder or a turn whose vector never got
|
||||
// computed. Same shape as the cascade: the better test leads, the offline one
|
||||
// always answers.
|
||||
func (h *reactiveHandler) isPersonalTurn(ctx context.Context, t *queryTurn) bool {
|
||||
h.boundary.load(ctx, h.embedder)
|
||||
if personal, world, ok := h.boundary.score(t.vec); ok {
|
||||
if personal > world {
|
||||
log.Printf("voice: %q scores personal %.4f vs world %.4f", t.dec.Utterance, personal, world)
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
return isPersonalQuery(t.dec.Utterance)
|
||||
}
|
||||
|
||||
// queryGeneral — general knowledge, the last source before giving up. It always
|
||||
// claims: either a model answers, or Maven names the gap, or she says she does
|
||||
// not know.
|
||||
//
|
||||
// This is the sharpest case for the naming half. Nothing has been fetched, so
|
||||
// there is no passage to fall back on and no floor under the answer except the
|
||||
// model's weights — and a 1.7B's weights are where the invented answers come
|
||||
// from. With a workstation configured and asleep he is told that, rather than
|
||||
// told something false in a confident voice. With no workstation configured at
|
||||
// all the resident model answers exactly as it does today: naming a gap requires
|
||||
// a gap, and on that box the 1.7B is the whole product.
|
||||
func (h *reactiveHandler) queryGeneral(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
reply, err := h.phraser.PhraseQuery(ctx, t.dec.Utterance, nil)
|
||||
if h.phraser == nil {
|
||||
// No model of any size. That is not the workstation being asleep, so it
|
||||
// is not that gap: it is simply not knowing.
|
||||
return "не знаю.", true
|
||||
}
|
||||
reply, err := h.phraseWorld(ctx, t.dec.Utterance, nil)
|
||||
if errors.Is(err, phraser.ErrNoWorldModel) {
|
||||
log.Printf("voice: %q needs the world model and it is not available", t.dec.Utterance)
|
||||
return worldGap, true
|
||||
}
|
||||
if err != nil || reply == "" {
|
||||
return "не знаю.", true
|
||||
}
|
||||
|
||||
@@ -19,6 +19,10 @@ func TestIsPersonalQuery(t *testing.T) {
|
||||
"when is my meeting",
|
||||
"do i have anything today",
|
||||
"did i take my vitamins",
|
||||
// Speech, but only the forms possession already covers ("did i").
|
||||
// The verb forms the floor cannot see are the seeds' job, scored in
|
||||
// TestONNXPersonalBoundary.
|
||||
"what did i say about backups",
|
||||
} {
|
||||
if !isPersonalQuery(s) {
|
||||
t.Errorf("isPersonalQuery(%q) = false, want true", s)
|
||||
@@ -33,6 +37,10 @@ func TestIsPersonalQuery(t *testing.T) {
|
||||
"почему небо синее",
|
||||
"столица франции",
|
||||
"how do i boil an egg",
|
||||
// The floor is possession-only by design: a speech verb it cannot see
|
||||
// passes here and is caught by the seeds instead.
|
||||
"что я говорил про бэкапы?",
|
||||
"как я говорил, почему небо синее",
|
||||
"",
|
||||
} {
|
||||
if isPersonalQuery(s) {
|
||||
|
||||
@@ -24,7 +24,10 @@ func (h *reactiveHandler) actionReminder(ctx context.Context, dec router.Decisio
|
||||
return "не получилось разобрать время напоминания."
|
||||
}
|
||||
}
|
||||
payload := `{"text":` + jsonString(dec.Utterance) + `}`
|
||||
// The body is what she says at the hour, so the marker and the time come
|
||||
// out of it: the fire time is already a column, and "напомни" is an
|
||||
// instruction that has been carried out (Vikunja #469).
|
||||
payload := `{"text":` + jsonString(reminderBody(dec.Utterance, dec.Slots.Text)) + `}`
|
||||
if _, err := h.api.CreateReminder(ctx, dec.Slots.Time, payload, ""); err != nil {
|
||||
log.Printf("voice: create reminder: %v", err)
|
||||
return "не получилось поставить напоминание."
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// attentionMarkers — the ways he asks what Praxis is holding. Substrings on a
|
||||
// stem, because "внимание", "внимания" and "вниманию" are one word to him.
|
||||
//
|
||||
// "что нового" is deliberately absent: the feeds source claims it, and it
|
||||
// still should — a question about news is a question about the feeds she
|
||||
// reads. This list is about the operational state of his things.
|
||||
var attentionMarkers = []string{
|
||||
"внимани", "что требует", "что не так", "что важн", "что срочн",
|
||||
"needs attention", "what needs looking",
|
||||
}
|
||||
|
||||
// isAttentionQuery reports whether the utterance asks what needs looking at.
|
||||
func isAttentionQuery(u string) bool {
|
||||
s := strings.ToLower(strings.TrimSpace(u))
|
||||
if s == "" {
|
||||
return false
|
||||
}
|
||||
for _, m := range attentionMarkers {
|
||||
if strings.Contains(s, m) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// queryAttention answers "что требует внимания?" from Praxis.
|
||||
//
|
||||
// The capability was already built and already degraded correctly, and no
|
||||
// utterance could reach it (Vikunja #475). Its aliases live on the act
|
||||
// dispatch, and the question routes to IntentQuery, so it fell through every
|
||||
// source to the web search and came back with an encyclopedia article about
|
||||
// the concept of attention — worse than silence, because it reads as an
|
||||
// answer.
|
||||
//
|
||||
// Placed above the recall sources and well above the personal boundary: this
|
||||
// is operational state about his things, and a notes pass would otherwise
|
||||
// answer it from whatever he once wrote about a server. An unconfigured or
|
||||
// absent Praxis falls through rather than claiming the turn, the same
|
||||
// convention queryHome and queryNetwork follow. A Praxis that is configured
|
||||
// and down does claim it, and says it cannot reach the service — that is the
|
||||
// degradation the ecosystem contract asks for, and it comes from the same
|
||||
// handler the act path uses.
|
||||
func (h *reactiveHandler) queryAttention(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
if !isAttentionQuery(t.dec.Utterance) {
|
||||
return "", false
|
||||
}
|
||||
if h.ecosystem == nil || h.ecosystem.praxis == nil {
|
||||
return "", false
|
||||
}
|
||||
reply := h.handlePraxisAct(ctx, router.Decision{
|
||||
Utterance: t.dec.Utterance,
|
||||
Intent: router.IntentAct,
|
||||
Slots: router.Slots{Fn: "list_attention", HasFn: true},
|
||||
})
|
||||
if reply == "" {
|
||||
return "", false
|
||||
}
|
||||
return reply, true
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
func TestIsAttentionQuery(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
text string
|
||||
want bool
|
||||
}{
|
||||
{"что требует внимания", true},
|
||||
{"на что обратить внимание?", true},
|
||||
{"что не так?", true},
|
||||
{"что важного?", true},
|
||||
// The feeds source owns this one, and should keep owning it.
|
||||
{"что нового?", false},
|
||||
{"какая погода?", false},
|
||||
{"", false},
|
||||
} {
|
||||
if got := isAttentionQuery(tc.text); got != tc.want {
|
||||
t.Errorf("isAttentionQuery(%q) = %v, want %v", tc.text, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestAttentionQuestionReachesPraxis — the defect (Vikunja #475). The question
|
||||
// routes to IntentQuery, and every source used to pass, so a web search about
|
||||
// the concept of attention answered it.
|
||||
func TestAttentionQuestionReachesPraxis(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
praxis := newFakePraxis(t, fixturePraxisAttentionItems(map[string]any{
|
||||
"id": "item_1", "title": "disk almost full", "importance": 3.0,
|
||||
}))
|
||||
h := ecoHandler(t, nil, praxis, nil)
|
||||
|
||||
reply, ok := h.queryAttention(ctx, &queryTurn{dec: router.Decision{
|
||||
Intent: router.IntentQuery, Utterance: "что требует внимания",
|
||||
}})
|
||||
if !ok {
|
||||
t.Fatal("the attention question must be claimed before the world sources")
|
||||
}
|
||||
if !strings.Contains(reply, "disk almost full") {
|
||||
t.Fatalf("reply = %q, want the praxis item", reply)
|
||||
}
|
||||
}
|
||||
|
||||
// A configured Praxis that is down claims the turn and says so. Falling
|
||||
// through here would answer an outage with an encyclopedia article.
|
||||
func TestAttentionQuestionSaysWhenPraxisIsDown(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
praxis := newFakePraxis(t, fixturePraxisAttentionItems())
|
||||
h := ecoHandler(t, nil, praxis, nil)
|
||||
praxis.SetFault(503)
|
||||
|
||||
reply, ok := h.queryAttention(ctx, &queryTurn{dec: router.Decision{
|
||||
Intent: router.IntentQuery, Utterance: "что требует внимания",
|
||||
}})
|
||||
if !ok || !strings.Contains(reply, "не могу") {
|
||||
t.Fatalf("an outage must name the gap, got ok=%v reply=%q", ok, reply)
|
||||
}
|
||||
}
|
||||
|
||||
// No Praxis configured means no claim: the rest of the chain still runs.
|
||||
func TestAttentionQuestionFallsThroughWithoutPraxis(t *testing.T) {
|
||||
h, _ := newFactGateHandler(t, time.Now())
|
||||
if _, ok := h.queryAttention(context.Background(), &queryTurn{dec: router.Decision{
|
||||
Intent: router.IntentQuery, Utterance: "что требует внимания",
|
||||
}}); ok {
|
||||
t.Fatal("an unconfigured praxis must not claim the turn")
|
||||
}
|
||||
}
|
||||
@@ -100,7 +100,7 @@ func (l llmCompleter) Complete(ctx context.Context, system, user string) (string
|
||||
// grammar, or a llama-server too old to honour one, gets the plain text it used
|
||||
// to get rather than an empty meeting summary.
|
||||
func unwrapSummary(raw string) string {
|
||||
s := stripThink(strings.TrimSpace(raw))
|
||||
s := phraser.StripThink(strings.TrimSpace(raw))
|
||||
start := strings.Index(s, "{")
|
||||
end := strings.LastIndex(s, "}")
|
||||
if start < 0 || end <= start {
|
||||
|
||||
+22
-29
@@ -33,19 +33,10 @@ var wantedSlots = map[router.Intent][]dialogue.Slot{
|
||||
router.IntentAct: {dialogue.SlotFn},
|
||||
}
|
||||
|
||||
// clarifyQuestions — one short question per missing slot.
|
||||
//
|
||||
// These are fixed templates, not model output. The resident model is a 0.8B; it
|
||||
// would wander, and a question whose wording changes every time is harder to
|
||||
// answer than a blunt one that always reads the same. They are infinitive
|
||||
// questions, so there is no gender agreement to get wrong; the feminine
|
||||
// self-reference lives in the reply she gives when she drops the request.
|
||||
var clarifyQuestions = map[dialogue.Slot]string{
|
||||
dialogue.SlotTime: "Когда?",
|
||||
dialogue.SlotText: "О чём напомнить?",
|
||||
dialogue.SlotKey: "Что записать?",
|
||||
dialogue.SlotFn: "Что сделать?",
|
||||
}
|
||||
// The questions themselves live in clarifytemplates.go, one list per slot,
|
||||
// picked by attempt (Vikunja #457). The first ask is the short one this map
|
||||
// used to hold; a re-ask says it differently, because a question he already
|
||||
// failed to answer is the worst one to repeat unchanged.
|
||||
|
||||
// clarifyGaveUp — she is out of questions and still does not have the slot. She
|
||||
// says so out loud: dropping the request in silence would leave him thinking it
|
||||
@@ -106,11 +97,11 @@ func trimClarifyExpired(s string) string {
|
||||
// out, and "" when nothing was parked. Call it right after
|
||||
// resolveClarifyAnswer: a live question is answered there, an expired one is
|
||||
// only reported here — the words themselves still go on to be routed fresh.
|
||||
func (h *reactiveHandler) clarifyExpiredNotice() string {
|
||||
func (h *reactiveHandler) clarifyExpiredNotice(ctx context.Context) string {
|
||||
if h.clarifyStore == nil {
|
||||
return ""
|
||||
}
|
||||
if !h.clarifyStore.TakeExpired(voiceDialogueID, h.now()) {
|
||||
if !h.clarifyStore.TakeExpired(dialogueIDOf(ctx), h.now()) {
|
||||
return ""
|
||||
}
|
||||
log.Printf("voice: clarify — parked question expired, telling him and routing the words fresh")
|
||||
@@ -147,7 +138,7 @@ func clarifyQuestion(dec router.Decision) (dialogue.Slot, string, bool) {
|
||||
if len(missing) == 0 {
|
||||
return "", "", false
|
||||
}
|
||||
q, ok := clarifyQuestions[missing[0]]
|
||||
q, ok := clarifyQuestionFor(missing[0], 1)
|
||||
if !ok {
|
||||
return "", "", false
|
||||
}
|
||||
@@ -157,7 +148,7 @@ func clarifyQuestion(dec router.Decision) (dialogue.Slot, string, bool) {
|
||||
// askClarify parks the request and returns the question to ask instead of the
|
||||
// canned "не поняла". Returns ("", false) when there is nothing to ask about, so
|
||||
// the caller falls back to the canned reply.
|
||||
func (h *reactiveHandler) askClarify(dec router.Decision) (string, bool) {
|
||||
func (h *reactiveHandler) askClarify(ctx context.Context, dec router.Decision) (string, bool) {
|
||||
if h.clarifyStore == nil {
|
||||
return "", false
|
||||
}
|
||||
@@ -165,7 +156,7 @@ func (h *reactiveHandler) askClarify(dec router.Decision) (string, bool) {
|
||||
if !ok {
|
||||
return "", false
|
||||
}
|
||||
h.clarifyStore.Put(voiceDialogueID, &dialogue.PendingQuestion{
|
||||
h.clarifyStore.Put(dialogueIDOf(ctx), &dialogue.PendingQuestion{
|
||||
Intent: dialogue.Intent(dec.Intent),
|
||||
Slots: toDialogueSlots(dec.Slots),
|
||||
Missing: []dialogue.Slot{slot},
|
||||
@@ -192,7 +183,7 @@ func (h *reactiveHandler) resolveClarifyAnswer(ctx context.Context, text string)
|
||||
if h.clarifyStore == nil {
|
||||
return "", false
|
||||
}
|
||||
q := h.clarifyStore.Get(voiceDialogueID, h.now())
|
||||
q := h.clarifyStore.Get(dialogueIDOf(ctx), h.now())
|
||||
if q == nil {
|
||||
return "", false
|
||||
}
|
||||
@@ -206,9 +197,9 @@ func (h *reactiveHandler) resolveClarifyAnswer(ctx context.Context, text string)
|
||||
// would fire at 11:00 saying "напомни" and nothing else.
|
||||
q.Utterance = foldAnswerIntoUtterance(q.Utterance, merged.Text)
|
||||
if len(dialogue.StillMissing(q.Missing, merged)) > 0 {
|
||||
return h.reaskOrGiveUp(q, merged, text), true
|
||||
return h.reaskOrGiveUp(ctx, q, merged, text), true
|
||||
}
|
||||
h.clarifyStore.Delete(voiceDialogueID)
|
||||
h.clarifyStore.Delete(dialogueIDOf(ctx))
|
||||
|
||||
// One gap filled is not the same as a complete request. askClarify parks
|
||||
// only the first gap, because one question per turn is the rule, but a
|
||||
@@ -217,7 +208,7 @@ func (h *reactiveHandler) resolveClarifyAnswer(ctx context.Context, text string)
|
||||
// a reminder with no time, which answered "не получилось разобрать время
|
||||
// напоминания." — an error for a request she never finished asking about.
|
||||
// Re-enter the loop instead, one question at a time as before.
|
||||
if reply, asked := h.askRemainingGap(q, intent, merged); asked {
|
||||
if reply, asked := h.askRemainingGap(ctx, q, intent, merged); asked {
|
||||
return reply, true
|
||||
}
|
||||
|
||||
@@ -261,16 +252,18 @@ func foldAnswerIntoUtterance(utterance, subject string) string {
|
||||
// The attempt budget is shared with the re-ask path on purpose. A second gap
|
||||
// costs a question exactly like a second try at the first one does, so the cap
|
||||
// still bounds how many times she can speak before acting or letting go.
|
||||
func (h *reactiveHandler) askRemainingGap(q *dialogue.PendingQuestion, intent router.Intent, merged dialogue.Slots) (string, bool) {
|
||||
func (h *reactiveHandler) askRemainingGap(ctx context.Context, q *dialogue.PendingQuestion, intent router.Intent, merged dialogue.Slots) (string, bool) {
|
||||
remaining := dialogue.StillMissing(wantedSlots[intent], merged)
|
||||
if len(remaining) == 0 {
|
||||
return "", false
|
||||
}
|
||||
question, ok := clarifyQuestions[remaining[0]]
|
||||
// Attempts+1 is the question she is about to ask, and the budget is shared
|
||||
// with the re-ask path, so the second gap is worded like a second try.
|
||||
question, ok := clarifyQuestionFor(remaining[0], q.Attempts+1)
|
||||
if !ok || !q.CanAsk() {
|
||||
return "", false
|
||||
}
|
||||
h.clarifyStore.Put(voiceDialogueID, &dialogue.PendingQuestion{
|
||||
h.clarifyStore.Put(dialogueIDOf(ctx), &dialogue.PendingQuestion{
|
||||
Intent: q.Intent,
|
||||
Slots: merged,
|
||||
Missing: []dialogue.Slot{remaining[0]},
|
||||
@@ -287,13 +280,13 @@ func (h *reactiveHandler) askRemainingGap(q *dialogue.PendingQuestion, intent ro
|
||||
// reaskOrGiveUp handles an answer that left the gap open: ask the same question
|
||||
// again while she has attempts left, otherwise say she did not understand and
|
||||
// let the request go. Never returns "" — a mute give-up reads as "done".
|
||||
func (h *reactiveHandler) reaskOrGiveUp(q *dialogue.PendingQuestion, merged dialogue.Slots, text string) string {
|
||||
func (h *reactiveHandler) reaskOrGiveUp(ctx context.Context, q *dialogue.PendingQuestion, merged dialogue.Slots, text string) string {
|
||||
question := ""
|
||||
if len(q.Missing) > 0 {
|
||||
question = clarifyQuestions[q.Missing[0]]
|
||||
question, _ = clarifyQuestionFor(q.Missing[0], q.Attempts+1)
|
||||
}
|
||||
if question == "" || !q.CanAsk() {
|
||||
h.clarifyStore.Delete(voiceDialogueID)
|
||||
h.clarifyStore.Delete(dialogueIDOf(ctx))
|
||||
log.Printf("voice: clarify — gave up on %v after %d question(s), answer was %q", q.Missing, q.Attempts, text)
|
||||
return clarifyGaveUp
|
||||
}
|
||||
@@ -302,7 +295,7 @@ func (h *reactiveHandler) reaskOrGiveUp(q *dialogue.PendingQuestion, merged dial
|
||||
q.Slots = merged
|
||||
q.Attempts++
|
||||
q.Asked = h.now()
|
||||
h.clarifyStore.Put(voiceDialogueID, q)
|
||||
h.clarifyStore.Put(dialogueIDOf(ctx), q)
|
||||
log.Printf("voice: clarify — answer %q did not fill %v, asking again (attempt %d)", text, q.Missing, q.Attempts)
|
||||
return question
|
||||
}
|
||||
|
||||
+125
-26
@@ -81,7 +81,7 @@ func TestClarifyReminderCompletesOnAnswer(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
|
||||
question, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме"))
|
||||
question, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме"))
|
||||
if !asked || question != "Когда?" {
|
||||
t.Fatalf("expected the time question, got %q asked=%v", question, asked)
|
||||
}
|
||||
@@ -112,7 +112,7 @@ func TestClarifyFactCompletesOnAnswer(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentFact, router.Slots{Text: "запиши"}, "запиши")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentFact, router.Slots{Text: "запиши"}, "запиши")); !asked {
|
||||
t.Fatal("a fact with no key should be asked about")
|
||||
}
|
||||
if reply, handled := h.resolveClarifyAnswer(ctx, "пил воду"); !handled || reply == clarifyGaveUp {
|
||||
@@ -128,7 +128,7 @@ func TestClarifyAnswerAfterTTLIsANewRequest(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, now := newClarifyHandler(t)
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a question")
|
||||
}
|
||||
*now = now.Add(clarifyTTL + time.Second)
|
||||
@@ -147,7 +147,7 @@ func TestClarifyAsksThreeTimesThenSaysSo(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a first question")
|
||||
}
|
||||
// Two more unclear answers ⇒ two more questions (3 asks in total).
|
||||
@@ -156,8 +156,14 @@ func TestClarifyAsksThreeTimesThenSaysSo(t *testing.T) {
|
||||
if !handled {
|
||||
t.Fatalf("answer %d must be consumed as an answer", i)
|
||||
}
|
||||
if reply != "Когда?" {
|
||||
t.Fatalf("attempt %d should ask again, got %q", i, reply)
|
||||
// The wording changes with the attempt (Vikunja #457): repeating a
|
||||
// question he already failed to answer is the worst way to ask it.
|
||||
want, _ := clarifyQuestionFor(dialogue.SlotTime, i)
|
||||
if reply != want {
|
||||
t.Fatalf("attempt %d should ask again as %q, got %q", i, want, reply)
|
||||
}
|
||||
if first, _ := clarifyQuestionFor(dialogue.SlotTime, 1); reply == first {
|
||||
t.Fatalf("attempt %d repeated the first wording: %q", i, reply)
|
||||
}
|
||||
if h.clarifyStore.Get(voiceDialogueID, h.now()) == nil {
|
||||
t.Fatalf("attempt %d must leave the question armed", i)
|
||||
@@ -185,7 +191,7 @@ func TestClarifyMaxAttemptsIsConfigurable(t *testing.T) {
|
||||
h, _, _ := newClarifyHandler(t)
|
||||
h.clarifyMaxAttempts = 1
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a question")
|
||||
}
|
||||
if reply, handled := h.resolveClarifyAnswer(ctx, "ну не знаю"); !handled || reply != clarifyGaveUp {
|
||||
@@ -199,7 +205,7 @@ func TestClarifyRestatedAnswerWins(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме")); !asked {
|
||||
t.Fatal("expected a question")
|
||||
}
|
||||
// First answer parses, but re-park it by hand as if she had asked again:
|
||||
@@ -232,7 +238,7 @@ func TestClarifiedActOffAllowlistIsStillRefused(t *testing.T) {
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
marker := filepath.Join(t.TempDir(), "not-allowed-ran")
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это")); !asked {
|
||||
t.Fatal("an act with no fn should be asked about")
|
||||
}
|
||||
reply, handled := h.resolveClarifyAnswer(ctx, "rm "+marker)
|
||||
@@ -260,7 +266,7 @@ func TestClarifiedDestructiveActStillNeedsConfirm(t *testing.T) {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это")); !asked {
|
||||
t.Fatal("expected a question")
|
||||
}
|
||||
reply, handled := h.resolveClarifyAnswer(ctx, "delete_backups")
|
||||
@@ -284,7 +290,7 @@ func TestNoQuestionWhenNothingIsMissing(t *testing.T) {
|
||||
clarifyDec(router.IntentQuery, router.Slots{Text: "ммм"}, "ммм"),
|
||||
clarifyDec(router.IntentNote, router.Slots{Text: "..."}, "..."),
|
||||
} {
|
||||
if question, asked := h.askClarify(dec); asked {
|
||||
if question, asked := h.askClarify(context.Background(), dec); asked {
|
||||
t.Fatalf("intent %s should keep the canned reply, got %q", dec.Intent, question)
|
||||
}
|
||||
}
|
||||
@@ -296,29 +302,29 @@ func TestNoQuestionWhenNothingIsMissing(t *testing.T) {
|
||||
// TestClarifyExpiryIsAnnouncedAndWordsStillRoute — his answer lands after the
|
||||
// TTL: she must say the old request is gone AND still answer the new words.
|
||||
func TestClarifyExpiryIsAnnouncedAndWordsStillRoute(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
ctx := withDialogueID(context.Background(), dialogueIDFor(sourceText, ""))
|
||||
h, _, now := newClarifyHandler(t)
|
||||
emb := router.NewHashEmbedder(1024)
|
||||
h.embedder = emb
|
||||
h.router = buildRouter(emb, h.matcher, 0.55, nil)
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a question")
|
||||
}
|
||||
*now = now.Add(clarifyTTL + time.Second)
|
||||
|
||||
reply := h.handleText(ctx, "как дела")
|
||||
reply := h.handleText(ctx, "", "как дела")
|
||||
if !isClarifyExpired(reply) {
|
||||
t.Fatalf("expired question must be announced first, got %q", reply)
|
||||
}
|
||||
if trimClarifyExpired(reply) == "" {
|
||||
t.Fatalf("the new words must still be answered, got only the notice: %q", reply)
|
||||
}
|
||||
if h.clarifyStore.Get(voiceDialogueID, h.now()) != nil {
|
||||
if h.clarifyStore.Get(textDialogueID, h.now()) != nil {
|
||||
t.Fatal("the expired question must be gone")
|
||||
}
|
||||
// The notice is said once, not on every later utterance.
|
||||
if reply := h.handleText(ctx, "как дела"); isClarifyExpired(reply) {
|
||||
if reply := h.handleText(ctx, "", "как дела"); isClarifyExpired(reply) {
|
||||
t.Fatalf("notice repeated on a later turn: %q", reply)
|
||||
}
|
||||
}
|
||||
@@ -340,7 +346,7 @@ func TestClarifyAsksAboutTheSecondGapToo(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
|
||||
question, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{}, "напомни"))
|
||||
question, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{}, "напомни"))
|
||||
if !asked || question != "О чём напомнить?" {
|
||||
t.Fatalf("expected the subject question, got %q asked=%v", question, asked)
|
||||
}
|
||||
@@ -349,8 +355,11 @@ func TestClarifyAsksAboutTheSecondGapToo(t *testing.T) {
|
||||
if !handled {
|
||||
t.Fatal("the answer must be consumed as an answer")
|
||||
}
|
||||
if reply != "Когда?" {
|
||||
t.Fatalf("a filled subject with no time must ask about the time, got %q", reply)
|
||||
// Second gap, second attempt, so it is the second wording of the time
|
||||
// question — the attempt budget is shared between the two paths.
|
||||
want, _ := clarifyQuestionFor(dialogue.SlotTime, 2)
|
||||
if reply != want {
|
||||
t.Fatalf("a filled subject with no time must ask about the time as %q, got %q", want, reply)
|
||||
}
|
||||
q := h.clarifyStore.Get(voiceDialogueID, h.now())
|
||||
if q == nil {
|
||||
@@ -380,7 +389,7 @@ func TestClarifySecondGapRespectsTheAttemptCap(t *testing.T) {
|
||||
h, _, _ := newClarifyHandler(t)
|
||||
h.clarifyMaxAttempts = 1
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{}, "напомни")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{}, "напомни")); !asked {
|
||||
t.Fatal("expected the subject question")
|
||||
}
|
||||
reply, handled := h.resolveClarifyAnswer(ctx, "позвонить маме")
|
||||
@@ -411,8 +420,9 @@ func TestClarifyProseHoldsThePersona(t *testing.T) {
|
||||
eval.CheckCringe: true,
|
||||
}
|
||||
lines := append([]string{clarifyGaveUp}, clarifyExpiredVariants...)
|
||||
for _, q := range clarifyQuestions {
|
||||
lines = append(lines, q)
|
||||
lines = append(lines, clarifyMissedVariants...)
|
||||
for _, variants := range clarifyQuestionVariants {
|
||||
lines = append(lines, variants...)
|
||||
}
|
||||
for _, line := range lines {
|
||||
for _, r := range eval.RunChecks(eval.Case{}, line, "neutral") {
|
||||
@@ -440,10 +450,10 @@ func TestClarifyProseHoldsThePersona(t *testing.T) {
|
||||
// The confirm turn used to return before the notice was even computed, so he
|
||||
// answered the confirm and never heard that the older request was let go.
|
||||
func TestExpiryNoticeSurvivesAConfirmTurn(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
ctx := withDialogueID(context.Background(), dialogueIDFor(sourceText, ""))
|
||||
h, _, now := newClarifyHandler(t)
|
||||
|
||||
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a question")
|
||||
}
|
||||
// A confirm parked with a longer life than the question, so only the
|
||||
@@ -451,7 +461,7 @@ func TestExpiryNoticeSurvivesAConfirmTurn(t *testing.T) {
|
||||
h.pending = &pendingAct{fn: "delete_backups", phrase: "удалить бэкапы", expiry: now.Add(time.Hour)}
|
||||
*now = now.Add(clarifyTTL + time.Second)
|
||||
|
||||
reply := h.handleText(ctx, "нет")
|
||||
reply := h.handleText(ctx, "", "нет")
|
||||
if !isClarifyExpired(reply) {
|
||||
t.Fatalf("the expired question must be announced on a confirm turn too, got %q", reply)
|
||||
}
|
||||
@@ -461,7 +471,96 @@ func TestExpiryNoticeSurvivesAConfirmTurn(t *testing.T) {
|
||||
if h.pending != nil {
|
||||
t.Fatal("the confirm must still have been consumed")
|
||||
}
|
||||
if h.clarifyStore.Get(voiceDialogueID, h.now()) != nil {
|
||||
if h.clarifyStore.Get(textDialogueID, h.now()) != nil {
|
||||
t.Fatal("the expired question must be gone")
|
||||
}
|
||||
}
|
||||
|
||||
// The other half of the subject question: his answer must fill the empty slot,
|
||||
// not replace the request. Slots.Text used to be the whole raw utterance for
|
||||
// every intent, so the branch that fills a text slot could only ever overwrite
|
||||
// (Vikunja #383). Here the parked request holds the hour and the answer holds
|
||||
// what to say at it, and the reminder that lands has both.
|
||||
func TestClarifySubjectAnswerFillsRatherThanClobbers(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
at := h.now().Add(2 * time.Hour)
|
||||
|
||||
question, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder,
|
||||
router.Slots{Time: at, HasTime: true}, "напомни в 11"))
|
||||
if !asked || question != "О чём напомнить?" {
|
||||
t.Fatalf("expected the subject question, got %q asked=%v", question, asked)
|
||||
}
|
||||
|
||||
reply, handled := h.resolveClarifyAnswer(ctx, "позвонить маме")
|
||||
if !handled {
|
||||
t.Fatal("the answer to an open question must be consumed as an answer")
|
||||
}
|
||||
if reply == clarifyGaveUp {
|
||||
t.Fatalf("a good answer must not drop the request: %q", reply)
|
||||
}
|
||||
|
||||
reminders, err := st.DueReminders(ctx, h.now().Add(48*time.Hour))
|
||||
if err != nil || len(reminders) != 1 {
|
||||
t.Fatalf("clarified reminder was not created: reminders=%v err=%v", reminders, err)
|
||||
}
|
||||
if !strings.Contains(reminders[0].Payload, "маме") {
|
||||
t.Fatalf("the answer never reached the reminder: %q", reminders[0].Payload)
|
||||
}
|
||||
// The hour is the fire time, not a word in the body: the body is what she
|
||||
// says at the hour, and the time expression is stripped out of it
|
||||
// (Vikunja #469). Clobbering the parked request would show up here as a
|
||||
// reminder that fires at some other time than the one he asked for.
|
||||
if got := reminders[0].FireTs.UTC(); !got.Equal(at.UTC()) {
|
||||
t.Fatalf("the answer clobbered the original request: fires at %v, want %v", got, at.UTC())
|
||||
}
|
||||
}
|
||||
|
||||
// TestClarifyIsPerConversation — the parked question belongs to the reach that
|
||||
// was asked. Before this the clarify store had one global key, so a question
|
||||
// asked in the web chat and never answered captured the next utterance from
|
||||
// telegram, or from the mic, and answered it against a request the speaker had
|
||||
// never made (Vikunja #466).
|
||||
func TestClarifyIsPerConversation(t *testing.T) {
|
||||
h, _, _ := newClarifyHandler(t)
|
||||
web := withDialogueID(context.Background(), dialogueIDFor(sourceText, "web"))
|
||||
telegram := withDialogueID(context.Background(), dialogueIDFor(sourceText, "telegram:42"))
|
||||
|
||||
if _, asked := h.askClarify(web, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a question on the web conversation")
|
||||
}
|
||||
if _, handled := h.resolveClarifyAnswer(telegram, "в 11:00"); handled {
|
||||
t.Fatal("a question asked on the web must not eat a telegram utterance")
|
||||
}
|
||||
if _, handled := h.resolveClarifyAnswer(voiceCtx(), "в 11:00"); handled {
|
||||
t.Fatal("a question asked on the web must not eat what he says at the mic")
|
||||
}
|
||||
if reply, handled := h.resolveClarifyAnswer(web, "в 11:00"); !handled || reply == clarifyGaveUp {
|
||||
t.Fatalf("the asker's own answer must land, handled=%v reply=%q", handled, reply)
|
||||
}
|
||||
}
|
||||
|
||||
// voiceCtx — the mic's conversation, which carries no id of its own.
|
||||
func voiceCtx() context.Context {
|
||||
return withDialogueID(context.Background(), dialogueIDFor(sourceVoice, ""))
|
||||
}
|
||||
|
||||
// TestARestartExpiresTheParkedQuestion pins the Vikunja #385 decision: the
|
||||
// question dies with the process, and she does not claim to have let it go —
|
||||
// the words that follow are routed as a fresh request. Restarting is modelled
|
||||
// the way the daemon does it, by building a second handler over the same store.
|
||||
func TestARestartExpiresTheParkedQuestion(t *testing.T) {
|
||||
h, _, _ := newClarifyHandler(t)
|
||||
ctx := voiceCtx()
|
||||
if _, asked := h.askClarify(ctx, clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||
t.Fatal("expected a question before the restart")
|
||||
}
|
||||
|
||||
restarted, _, _ := newClarifyHandler(t)
|
||||
if _, handled := restarted.resolveClarifyAnswer(ctx, "в 11:00"); handled {
|
||||
t.Fatal("a question parked before the restart must not eat the next utterance")
|
||||
}
|
||||
if notice := restarted.clarifyExpiredNotice(ctx); notice != "" {
|
||||
t.Fatalf("notice = %q, want silence: nothing survived to expire", notice)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"github.com/kami/maven/internal/dialogue"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// The clarify copy deck (Vikunja #457).
|
||||
//
|
||||
// Every clarify turn used to say one sentence per gap, and a re-ask repeated
|
||||
// that sentence word for word. A question he already failed to answer is the
|
||||
// worst one to ask again unchanged: the second wording is the one that tells
|
||||
// him which part she missed.
|
||||
//
|
||||
// Fixed templates, not model output, for the reason clarifyQuestions has always
|
||||
// given: the resident model would wander, and a question whose wording changes
|
||||
// at random is harder to answer than a blunt one. What changes here is that the
|
||||
// wording varies with the attempt rather than with a die roll — the first ask is
|
||||
// short, the second names the gap, the third spells it out.
|
||||
//
|
||||
// No schema_version, unlike internal/phraser/nudge_templates.go. These are Go
|
||||
// constants compiled into the daemon, so there is no file that can drift out of
|
||||
// step with the code that reads it.
|
||||
//
|
||||
// Persona holds: infinitive and imperative questions, so there is no gender
|
||||
// agreement to get wrong, "ты" throughout, and no pet names.
|
||||
var clarifyQuestionVariants = map[dialogue.Slot][]string{
|
||||
dialogue.SlotTime: {
|
||||
"Когда?",
|
||||
"Во сколько напомнить?",
|
||||
"Скажи время — например, «в семь вечера» или «через час».",
|
||||
},
|
||||
dialogue.SlotText: {
|
||||
"О чём напомнить?",
|
||||
"Что сказать тебе в это время?",
|
||||
"Скажи одной фразой, о чём напомнить.",
|
||||
},
|
||||
dialogue.SlotKey: {
|
||||
"Что записать?",
|
||||
"Что именно отметить?",
|
||||
"Назови, что записать — например, «выпил воды».",
|
||||
},
|
||||
dialogue.SlotFn: {
|
||||
"Что сделать?",
|
||||
"Какое действие выполнить?",
|
||||
"Назови действие — я умею только то, что ты мне разрешил.",
|
||||
},
|
||||
}
|
||||
|
||||
// clarifyQuestionFor picks the wording for this attempt. attempt is 1-based, as
|
||||
// PendingQuestion.Attempts counts it; anything past the list uses the last and
|
||||
// most explicit phrasing rather than wrapping round to the short one, because
|
||||
// wrapping would ask the same short question he has already not answered.
|
||||
//
|
||||
// Deterministic on purpose, unlike clarifyExpiredLine: an expiry notice is the
|
||||
// same statement however it is worded, and a re-ask is not.
|
||||
func clarifyQuestionFor(slot dialogue.Slot, attempt int) (string, bool) {
|
||||
variants, ok := clarifyQuestionVariants[slot]
|
||||
if !ok || len(variants) == 0 {
|
||||
return "", false
|
||||
}
|
||||
i := attempt - 1
|
||||
if i < 0 {
|
||||
i = 0
|
||||
}
|
||||
if i >= len(variants) {
|
||||
i = len(variants) - 1
|
||||
}
|
||||
return variants[i], true
|
||||
}
|
||||
|
||||
// clarifyMissedVariants — she is asking for the whole utterance again, because
|
||||
// the gate fired on an intent with nothing identifiable to ask about (note,
|
||||
// query, chat, system are not in wantedSlots).
|
||||
//
|
||||
// Rotated like the expiry lines and for the same reason: this is the line he
|
||||
// hears whenever she misses him completely, so it is a line that repeats, and
|
||||
// the same sentence every time is what makes a house assistant sound like a
|
||||
// kiosk. All of them say the same two things — she did not catch it, and he
|
||||
// should say it again — because the wording may vary and the meaning may not.
|
||||
var clarifyMissedVariants = []string{
|
||||
"Не совсем поняла — скажи, пожалуйста, ещё раз.",
|
||||
"Я тебя не разобрала. Повтори, пожалуйста.",
|
||||
"Не уловила. Скажи это по-другому?",
|
||||
"Прости, не поняла — попробуй сказать иначе.",
|
||||
}
|
||||
|
||||
// clarifyMissedFor picks a wording by the utterance itself, so the same words
|
||||
// asked twice get the same answer and two different misses sound different.
|
||||
//
|
||||
// A hash, not rand: a test that drives an utterance twice must not depend on a
|
||||
// die roll, and the point of rotating is only that consecutive misses differ.
|
||||
func clarifyMissedFor(utterance string) string {
|
||||
var sum int
|
||||
for _, r := range utterance {
|
||||
sum += int(r)
|
||||
}
|
||||
return clarifyMissedVariants[sum%len(clarifyMissedVariants)]
|
||||
}
|
||||
|
||||
// clarifyMissedLine is the canned reply for a clarify decision she cannot turn
|
||||
// into a question. Returns "" for a decision that is not a clarify, so the
|
||||
// caller keeps its own reply.
|
||||
func clarifyMissedLine(dec router.Decision) string {
|
||||
if !dec.Clarify {
|
||||
return ""
|
||||
}
|
||||
return clarifyMissedFor(dec.Utterance)
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/dialogue"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// Every slot she can ask about has a wording for every attempt she is allowed,
|
||||
// and no two attempts on one slot read the same. A deck with a repeated line is
|
||||
// the defect this deck exists to fix (Vikunja #457).
|
||||
func TestClarifyQuestionsVaryByAttempt(t *testing.T) {
|
||||
for slot, variants := range clarifyQuestionVariants {
|
||||
seen := map[string]bool{}
|
||||
for _, v := range variants {
|
||||
if v == "" {
|
||||
t.Errorf("%s: empty wording in the deck", slot)
|
||||
}
|
||||
if seen[v] {
|
||||
t.Errorf("%s: repeated wording %q", slot, v)
|
||||
}
|
||||
seen[v] = true
|
||||
}
|
||||
for attempt := 1; attempt <= len(variants); attempt++ {
|
||||
got, ok := clarifyQuestionFor(slot, attempt)
|
||||
if !ok || got != variants[attempt-1] {
|
||||
t.Errorf("%s attempt %d = %q ok=%v, want %q", slot, attempt, got, ok, variants[attempt-1])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Past the end she keeps the most explicit wording. Wrapping round would ask
|
||||
// the short question he has already not answered twice.
|
||||
func TestClarifyQuestionPastTheEndKeepsTheLastWording(t *testing.T) {
|
||||
last := clarifyQuestionVariants[dialogue.SlotTime][len(clarifyQuestionVariants[dialogue.SlotTime])-1]
|
||||
for _, attempt := range []int{0, 4, 9} {
|
||||
if got, _ := clarifyQuestionFor(dialogue.SlotTime, attempt); attempt > 1 && got != last {
|
||||
t.Errorf("attempt %d = %q, want the last wording %q", attempt, got, last)
|
||||
}
|
||||
}
|
||||
if _, ok := clarifyQuestionFor("nonesuch", 1); ok {
|
||||
t.Error("an unknown slot must have no question")
|
||||
}
|
||||
}
|
||||
|
||||
// The missed line is stable for one utterance and absent for a decision that is
|
||||
// not a clarify.
|
||||
func TestClarifyMissedLine(t *testing.T) {
|
||||
d := router.Decision{Clarify: true, Utterance: "мгм"}
|
||||
first := clarifyMissedLine(d)
|
||||
if first == "" || first != clarifyMissedLine(d) {
|
||||
t.Fatalf("the missed line must be stable for one utterance, got %q", first)
|
||||
}
|
||||
if got := clarifyMissedLine(router.Decision{Intent: router.IntentNote}); got != "" {
|
||||
t.Errorf("a decision that is not a clarify got %q", got)
|
||||
}
|
||||
// The empty utterance still gets a line: she has to say something.
|
||||
if got := clarifyMissedLine(router.Decision{Clarify: true}); got == "" {
|
||||
t.Error("an empty utterance must still be answered out loud")
|
||||
}
|
||||
}
|
||||
@@ -514,11 +514,14 @@ func (h *reactiveHandler) handleHexisAct(ctx context.Context, dec router.Decisio
|
||||
|
||||
// Resolve the utterance text as an entity reference through Nexus. An
|
||||
// ambiguous match must stop and clarify — never guess a mutation target.
|
||||
// The name comes from entityReferenceText, not straight from the Text slot:
|
||||
// the model transliterates Latin names as it routes (Vikunja #476).
|
||||
subject := entityReferenceText(dec)
|
||||
started := h.now()
|
||||
entityID, displayName, ambiguous, err := h.ecosystem.resolveEntityReference(ctx, dec.Slots.Text, nil)
|
||||
entityID, displayName, ambiguous, err := h.ecosystem.resolveEntityReference(ctx, subject, nil)
|
||||
if err != nil {
|
||||
h.recordEcosystemTrace(ctx, "nexus", "resolve", traceStatusForError(err), started,
|
||||
mergeFields(traceErrorFields(err), map[string]any{"subject": redactSubject(dec.Slots.Text)}))
|
||||
mergeFields(traceErrorFields(err), map[string]any{"subject": redactSubject(subject)}))
|
||||
if unauthorizedEcosystemError(err) {
|
||||
return "экосистема отклоняет доступ, проверь токен."
|
||||
}
|
||||
@@ -535,7 +538,7 @@ func (h *reactiveHandler) handleHexisAct(ctx context.Context, dec router.Decisio
|
||||
}
|
||||
if entityID == "" {
|
||||
h.recordEcosystemTrace(ctx, "nexus", "resolve", traceNotFound, started,
|
||||
map[string]any{"subject": redactSubject(dec.Slots.Text)})
|
||||
map[string]any{"subject": redactSubject(subject)})
|
||||
return ""
|
||||
}
|
||||
h.recordEcosystemTrace(ctx, "nexus", "resolve", traceOK, started,
|
||||
@@ -569,10 +572,21 @@ func (h *reactiveHandler) handleHexisAct(ctx context.Context, dec router.Decisio
|
||||
}
|
||||
verbLower := strings.ToLower(verb)
|
||||
|
||||
// With no allowlisted fn the verb is a whole phrase ("restart status muzick
|
||||
// indexer"), which no capability name ever contains. Read it the other way
|
||||
// round then: the phrase is the haystack and the capability name is what we
|
||||
// look for in it (Vikunja #476). Only when the fn slot is empty — a matched
|
||||
// fn is a single verb and containment already means what it says.
|
||||
loose := !dec.Slots.HasFn
|
||||
var matches []*hexisclient.Capability
|
||||
for i, c := range caps {
|
||||
if strings.Contains(strings.ToLower(c.Name), verbLower) ||
|
||||
(c.Description != "" && strings.Contains(strings.ToLower(c.Description), verbLower)) {
|
||||
name := strings.ToLower(c.Name)
|
||||
hit := strings.Contains(name, verbLower) ||
|
||||
(c.Description != "" && strings.Contains(strings.ToLower(c.Description), verbLower))
|
||||
if loose && name != "" && strings.Contains(verbLower, name) {
|
||||
hit = true
|
||||
}
|
||||
if hit {
|
||||
matches = append(matches, &caps[i])
|
||||
}
|
||||
}
|
||||
@@ -632,3 +646,29 @@ func (h *reactiveHandler) execHexis(ctx context.Context, capID, capName, entityI
|
||||
})
|
||||
return "команда выполнена для " + displayName + "."
|
||||
}
|
||||
|
||||
// hexisBeforeClarify gives an entity-shaped act one chance at Hexis before she
|
||||
// asks what to do.
|
||||
//
|
||||
// The stage-3 gate thins an act that never matched an allowlisted fn, so
|
||||
// "перезапусти muzick indexer" was answered with "Что сделать?" and the Hexis
|
||||
// path was never entered — the capability existed and no utterance could reach
|
||||
// it (Vikunja #476). Hexis is exactly where an act with no local fn belongs:
|
||||
// the verb is matched against the capabilities Hexis registers for the entity,
|
||||
// not against the allowlist.
|
||||
//
|
||||
// Narrow on purpose. Only an act, only when the fn slot is still empty, and
|
||||
// only when Hexis is wired — a box with no ecosystem asks the question it
|
||||
// always asked. A "" back means Nexus knew no such entity or Hexis had no
|
||||
// matching capability, and then she asks after all. Authority is unchanged:
|
||||
// resolution stops on ambiguity and a mutating capability still goes through
|
||||
// the spoken confirm in handleHexisAct.
|
||||
func (h *reactiveHandler) hexisBeforeClarify(ctx context.Context, dec router.Decision) string {
|
||||
if h.ecosystem == nil || h.ecosystem.hexis == nil {
|
||||
return ""
|
||||
}
|
||||
if dec.Intent != router.IntentAct || dec.Slots.HasFn || dec.Slots.Text == "" {
|
||||
return ""
|
||||
}
|
||||
return h.handleHexisAct(ctx, dec)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
"unicode"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// latinRun matches a run of Latin-script words — the shape a service, host or
|
||||
// project name takes in a Russian sentence. Digits, dot, dash and underscore
|
||||
// ride along because "muzick-indexer" and "nginx.conf" are one name, not two.
|
||||
var latinRun = regexp.MustCompile(`[A-Za-z][A-Za-z0-9._-]*(?:\s+[A-Za-z][A-Za-z0-9._-]*)*`)
|
||||
|
||||
// hasLatin reports whether s carries a Latin letter.
|
||||
func hasLatin(s string) bool {
|
||||
for _, r := range s {
|
||||
if unicode.In(r, unicode.Latin) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// entityReferenceText is the name Nexus is asked to resolve.
|
||||
//
|
||||
// Normally that is the router's Text slot, which is the verb phrase the model
|
||||
// wrote. But the resident model rewrites a Russian utterance as it routes, and
|
||||
// on the way it transliterates: "перезапусти muzick indexer" came back as
|
||||
// "перезагрузить музик индексер" (Vikunja #476). Nexus is then asked for a
|
||||
// service nobody has ever named, so the act cannot resolve its target even
|
||||
// with every gate open.
|
||||
//
|
||||
// The recovery is deliberately narrow. Only when the utterance holds a Latin
|
||||
// run and the model's Text holds none has a name certainly been rewritten —
|
||||
// then the longest Latin run in his own words is the reference. Anything else
|
||||
// keeps the Text slot, so an English utterance and a Russian entity name are
|
||||
// both untouched. Un-transliterating the Cyrillic back is not attempted: the
|
||||
// surface form he said is right there, and guessing at a reverse mapping would
|
||||
// invent a second name to be wrong about.
|
||||
func entityReferenceText(dec router.Decision) string {
|
||||
text := dec.Slots.Text
|
||||
if hasLatin(text) || !hasLatin(dec.Utterance) {
|
||||
return text
|
||||
}
|
||||
longest := ""
|
||||
for _, m := range latinRun.FindAllString(dec.Utterance, -1) {
|
||||
if len(m) > len(longest) {
|
||||
longest = m
|
||||
}
|
||||
}
|
||||
longest = strings.TrimSpace(longest)
|
||||
// A single stray letter is not a name.
|
||||
if len(longest) < 2 {
|
||||
return text
|
||||
}
|
||||
return longest
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// TestEntityReferenceText pins when his own words win over the model's.
|
||||
func TestEntityReferenceText(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
utterance string
|
||||
text string
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "the model transliterated the name",
|
||||
utterance: "перезапусти muzick indexer",
|
||||
text: "перезагрузить музик индексер",
|
||||
want: "muzick indexer",
|
||||
},
|
||||
{
|
||||
name: "it kept the name, so nothing to repair",
|
||||
utterance: "перезапусти muzick indexer",
|
||||
text: "перезагрузить muzick indexer",
|
||||
want: "перезагрузить muzick indexer",
|
||||
},
|
||||
{
|
||||
name: "an all-Russian entity name is not a rewrite",
|
||||
utterance: "перезапусти домашний сервер",
|
||||
text: "перезагрузить домашний сервер",
|
||||
want: "перезагрузить домашний сервер",
|
||||
},
|
||||
{
|
||||
name: "an English turn never enters the recovery",
|
||||
utterance: "restart muzick indexer",
|
||||
text: "restart muzick indexer",
|
||||
want: "restart muzick indexer",
|
||||
},
|
||||
{
|
||||
name: "the longest Latin run is the name",
|
||||
utterance: "а перезапусти-ка nginx на muzick-indexer, пожалуйста",
|
||||
text: "перезагрузить нгинкс",
|
||||
want: "muzick-indexer",
|
||||
},
|
||||
{
|
||||
name: "one stray letter is not a name",
|
||||
utterance: "перезапусти сервер a",
|
||||
text: "перезагрузить сервер",
|
||||
want: "перезагрузить сервер",
|
||||
},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
dec := router.Decision{Utterance: tc.utterance, Slots: router.Slots{Text: tc.text}}
|
||||
if got := entityReferenceText(dec); got != tc.want {
|
||||
t.Fatalf("entityReferenceText = %q, want %q", got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestNexusIsAskedForTheNameHeSaid — the defect end to end (Vikunja #476): the
|
||||
// router hands over a transliterated Text, and Nexus must still be asked about
|
||||
// the service that exists.
|
||||
func TestNexusIsAskedForTheNameHeSaid(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
nexus := newFakeNexus(t, fixtureNexusResolved("ent_muzick", "Muzick indexer", "service"))
|
||||
hexis := newFakeHexis(t, restartCaps(), fixtureHexisExecuted("exec_1", "succeeded"))
|
||||
h := ecoHandler(t, nexus, nil, hexis)
|
||||
|
||||
dec := router.Decision{
|
||||
Utterance: "перезапусти muzick indexer",
|
||||
Intent: router.IntentAct,
|
||||
Slots: router.Slots{Text: "перезагрузить музик индексер", Fn: "restart", HasFn: true},
|
||||
}
|
||||
h.handleHexisAct(ctx, dec)
|
||||
|
||||
reqs := nexus.Requests()
|
||||
if len(reqs) == 0 {
|
||||
t.Fatal("nexus was never asked")
|
||||
}
|
||||
body := string(reqs[0].Body)
|
||||
if !strings.Contains(body, "muzick indexer") {
|
||||
t.Fatalf("nexus resolve body = %s, want the name he said", body)
|
||||
}
|
||||
}
|
||||
|
||||
// TestAnEntityActReachesHexisInsteadOfAsking — the second half of #476. The
|
||||
// stage-3 gate thins an act with no allowlisted fn, and that question used to
|
||||
// be the whole turn, so the Hexis path was unreachable from voice or chat.
|
||||
func TestAnEntityActReachesHexisInsteadOfAsking(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
nexus := newFakeNexus(t, fixtureNexusResolved("ent_muzick", "Muzick indexer", "service"))
|
||||
hexis := newFakeHexis(t, restartCaps(), fixtureHexisExecuted("exec_1", "succeeded"))
|
||||
h := ecoHandler(t, nexus, nil, hexis)
|
||||
|
||||
dec := router.Decision{
|
||||
Utterance: "перезапусти muzick indexer",
|
||||
Intent: router.IntentAct,
|
||||
Stage: 3,
|
||||
Clarify: true,
|
||||
Slots: router.Slots{Text: "restart status muzick indexer"},
|
||||
}
|
||||
reply := h.hexisBeforeClarify(ctx, dec)
|
||||
if reply == "" {
|
||||
t.Fatal("a resolvable entity act must reach hexis rather than fall through to the question")
|
||||
}
|
||||
if hexis.Count("", "/api/v1") == 0 {
|
||||
t.Fatal("hexis was never contacted")
|
||||
}
|
||||
}
|
||||
|
||||
// TestClarifyStillAsksWithoutHexis — the narrowing. No ecosystem, no change:
|
||||
// she asks exactly what she asked before.
|
||||
func TestClarifyStillAsksWithoutHexis(t *testing.T) {
|
||||
h, _, _ := newClarifyHandler(t)
|
||||
dec := router.Decision{
|
||||
Utterance: "перезапусти muzick indexer",
|
||||
Intent: router.IntentAct,
|
||||
Stage: 3,
|
||||
Clarify: true,
|
||||
Slots: router.Slots{Text: "перезагрузить музик индексер"},
|
||||
}
|
||||
if reply := h.hexisBeforeClarify(context.Background(), dec); reply != "" {
|
||||
t.Fatalf("no hexis must mean no reply, got %q", reply)
|
||||
}
|
||||
if _, asked := h.askClarify(voiceCtx(), dec); !asked {
|
||||
t.Fatal("she must still ask what to do")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"log"
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// ungroundedConfidence — what a self fact is worth when its value appears
|
||||
// nowhere in what he said. Below `query_min_score` is not the point (recall
|
||||
// gates on vector distance, not on this number); the point is that
|
||||
// `/history` and every future reader can tell a value he said from a value
|
||||
// the model supplied.
|
||||
const ungroundedConfidence = 0.6
|
||||
|
||||
// factConfidence scores a self fact by whether its value is grounded in the
|
||||
// utterance it came from. Grounded stays 1.00, which is what a tapped fact
|
||||
// has always been worth. Ungrounded drops, and says so in the log.
|
||||
//
|
||||
// An empty value is grounded by definition: the key alone carries the fact
|
||||
// ("поужинал"), and there is nothing for the model to have invented.
|
||||
func factConfidence(utterance, value string) float64 {
|
||||
if strings.TrimSpace(value) == "" {
|
||||
return 1.0
|
||||
}
|
||||
if valueGrounded(utterance, value) {
|
||||
return 1.0
|
||||
}
|
||||
log.Printf("voice: fact value %q is not in %q — writing at confidence %.2f",
|
||||
value, utterance, ungroundedConfidence)
|
||||
return ungroundedConfidence
|
||||
}
|
||||
|
||||
// valueGrounded reports whether every word of value traces back to a word he
|
||||
// actually said. The comparison is on a 4-rune prefix, so the model's
|
||||
// normalization survives ("пил воду" → "вода") while an invented value
|
||||
// ("1.20" for a question about Go) does not.
|
||||
func valueGrounded(utterance, value string) bool {
|
||||
said := factTokens(utterance)
|
||||
words := factTokens(value)
|
||||
if len(words) == 0 {
|
||||
return true
|
||||
}
|
||||
for _, w := range words {
|
||||
if !anyTokenMatches(said, w) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func anyTokenMatches(said []string, w string) bool {
|
||||
for _, s := range said {
|
||||
if s == w || sameStem(s, w) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// sameStem is inflection tolerance and nothing more: it compares all but the
|
||||
// last rune of the shorter word, and never fewer than three. Russian marks
|
||||
// case on the ending, so "пил воду" and the stored "вода" are the same word he
|
||||
// said, while "1.20" and "версия" are not. A word of three runes or fewer must
|
||||
// match outright, where a shorter prefix would match half the language.
|
||||
func sameStem(a, b string) bool {
|
||||
ar, br := []rune(a), []rune(b)
|
||||
shorter := min(len(ar), len(br))
|
||||
n := shorter - 1
|
||||
if n < 3 || len(ar) < n || len(br) < n {
|
||||
return false
|
||||
}
|
||||
return string(ar[:n]) == string(br[:n])
|
||||
}
|
||||
|
||||
// factTokens lowercases and splits on everything that is not a letter or a
|
||||
// digit, the same shape planTokens uses in the router.
|
||||
func factTokens(s string) []string {
|
||||
return strings.FieldsFunc(strings.ToLower(s), func(r rune) bool {
|
||||
return !unicode.IsLetter(r) && !unicode.IsDigit(r)
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,170 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/tool"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
|
||||
func newFactGateHandler(t *testing.T, now time.Time) (*reactiveHandler, ipc.CoreAPI) {
|
||||
t.Helper()
|
||||
st := newTestStore(t)
|
||||
api := ipc.NewStoreAPI(st)
|
||||
emb := router.NewHashEmbedder(1024)
|
||||
h := &reactiveHandler{
|
||||
api: api,
|
||||
embedder: emb,
|
||||
router: buildRouter(emb, tool.NewMatcher(api), 0.55, nil),
|
||||
replier: voice.NewStubReplier(),
|
||||
now: func() time.Time { return now },
|
||||
memStore: memory.NewInMemoryStore(),
|
||||
dataStore: st,
|
||||
}
|
||||
return h, api
|
||||
}
|
||||
|
||||
// The write half of #470: a question routed to IntentFact must not become a
|
||||
// fact about him, and must not leave a vector behind for recall to serve.
|
||||
func TestActionFact_QuestionIsNotWritten(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, api := newFactGateHandler(t, time.Now())
|
||||
|
||||
reply := h.actionFact(ctx, router.Decision{
|
||||
Intent: router.IntentFact,
|
||||
Utterance: "какая последняя версия языка Go?",
|
||||
Slots: router.Slots{Key: "go_version", HasKey: true, Value: `"1.20"`},
|
||||
})
|
||||
|
||||
if _, err := api.LatestFact(ctx, "go_version"); err == nil {
|
||||
t.Fatal("a question was stored as a fact about him")
|
||||
}
|
||||
hits, err := h.memStore.Search(ctx, mustEmbedPassage(t, h, "какая последняя версия языка Go?"), 3)
|
||||
if err != nil {
|
||||
t.Fatalf("memory search: %v", err)
|
||||
}
|
||||
if len(hits) != 0 {
|
||||
t.Fatalf("the question was indexed for recall: %+v", hits)
|
||||
}
|
||||
// It went down the query chain instead. Nothing is configured to answer a
|
||||
// world question in this harness, so "не знаю." is the honest outcome —
|
||||
// what matters is that the turn was answered, not stored.
|
||||
if reply == "" {
|
||||
t.Fatal("the turn was neither stored nor answered")
|
||||
}
|
||||
}
|
||||
|
||||
// The capture that must survive the gate: an explicit instruction to record,
|
||||
// even though it contains an interrogative.
|
||||
func TestActionFact_ExplicitCaptureStillWrites(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, api := newFactGateHandler(t, time.Now())
|
||||
|
||||
h.actionFact(ctx, router.Decision{
|
||||
Intent: router.IntentFact,
|
||||
Utterance: "запиши что я пил воду",
|
||||
Slots: router.Slots{Key: "water", HasKey: true, Value: `"вода"`},
|
||||
})
|
||||
|
||||
f, err := api.LatestFact(ctx, "water")
|
||||
if err != nil {
|
||||
t.Fatalf("an explicit capture was refused: %v", err)
|
||||
}
|
||||
if f.Confidence != 1.0 {
|
||||
t.Errorf("confidence = %v, want 1.0 for a value he said", f.Confidence)
|
||||
}
|
||||
// #493: what recall reads back is the fact, not the sentence he said.
|
||||
// queryMemory returns a fact's text verbatim, so the utterance sitting here
|
||||
// meant "запиши что я пил воду" was the answer to "когда я пил воду?".
|
||||
hits, err := h.memStore.Search(ctx, mustEmbedPassage(t, h, "вода"), 3)
|
||||
if err != nil {
|
||||
t.Fatalf("memory search: %v", err)
|
||||
}
|
||||
if len(hits) != 1 {
|
||||
t.Fatalf("the fact was not indexed once: %+v", hits)
|
||||
}
|
||||
if got := hits[0].Meta["text"]; got != "water — вода" {
|
||||
t.Errorf("indexed text = %q, want the fact", got)
|
||||
}
|
||||
if got := hits[0].Meta["utterance"]; got != "запиши что я пил воду" {
|
||||
t.Errorf("utterance provenance = %q, want it kept alongside", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFactConfidence(t *testing.T) {
|
||||
cases := []struct {
|
||||
utterance, value string
|
||||
want float64
|
||||
}{
|
||||
{"запиши что я пил воду", `"вода"`, 1.0},
|
||||
{"я выпил кофе", `"кофе"`, 1.0},
|
||||
{"поужинал", "", 1.0},
|
||||
{"отметь что я полил кактус", `"полил кактус"`, 1.0},
|
||||
{"какая последняя версия языка Go", `"1.20"`, ungroundedConfidence},
|
||||
{"кто премьер Японии", `"Тонио Озаки"`, ungroundedConfidence},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := factConfidence(c.utterance, c.value); got != c.want {
|
||||
t.Errorf("factConfidence(%q, %q) = %v, want %v", c.utterance, c.value, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func mustEmbedPassage(t *testing.T, h *reactiveHandler, text string) []float32 {
|
||||
t.Helper()
|
||||
vec, err := router.EmbedQuery(context.Background(), h.embedder, text)
|
||||
if err != nil {
|
||||
t.Fatalf("embed %q: %v", text, err)
|
||||
}
|
||||
return vec
|
||||
}
|
||||
|
||||
// The write half of #481: a complaint about a thing is a state of the
|
||||
// afternoon, not a fact about him. Stored as a `self` row at confidence 1.00
|
||||
// it comes back on recall as if the network were still down.
|
||||
func TestActionFact_ComplaintIsNotWritten(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, api := newFactGateHandler(t, time.Now())
|
||||
|
||||
reply := h.actionFact(ctx, router.Decision{
|
||||
Intent: router.IntentFact,
|
||||
Utterance: "сеть какая-то медленная",
|
||||
Slots: router.Slots{Key: "network_speed", HasKey: true, Value: "медленная"},
|
||||
})
|
||||
|
||||
if _, err := api.LatestFact(ctx, "network_speed"); err == nil {
|
||||
t.Fatal("a passing complaint was stored as a fact about him")
|
||||
}
|
||||
hits, err := h.memStore.Search(ctx, mustEmbedPassage(t, h, "сеть какая-то медленная"), 3)
|
||||
if err != nil {
|
||||
t.Fatalf("memory search: %v", err)
|
||||
}
|
||||
if len(hits) != 0 {
|
||||
t.Fatalf("the complaint was indexed for recall: %+v", hits)
|
||||
}
|
||||
if reply == "" {
|
||||
t.Fatal("the turn was neither stored nor answered")
|
||||
}
|
||||
}
|
||||
|
||||
// And the complaint he asked her to keep: the capture verb wins, as it does
|
||||
// over the question gate.
|
||||
func TestActionFact_AskedToRememberAComplaintStillWrites(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, api := newFactGateHandler(t, time.Now())
|
||||
|
||||
h.actionFact(ctx, router.Decision{
|
||||
Intent: router.IntentFact,
|
||||
Utterance: "запомни что интернет не работает",
|
||||
Slots: router.Slots{Key: "internet", HasKey: true, Value: "не работает"},
|
||||
})
|
||||
|
||||
if _, err := api.LatestFact(ctx, "internet"); err != nil {
|
||||
t.Fatalf("an explicit capture was refused: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -194,3 +194,22 @@ func TestFeedWorkerFetcherIsAllowlisted(t *testing.T) {
|
||||
t.Fatal("the poller fetched a private address")
|
||||
}
|
||||
}
|
||||
|
||||
// TestQueryFeedsPassesWhenTheWorldCanAnswer — the defect (Vikunja #474). The
|
||||
// deployed box has no feeds block and does have SearXNG, and "что происходит
|
||||
// сейчас в новостях про искусственный интеллект?" got a configuration status
|
||||
// instead of the live answer sitting one source below.
|
||||
func TestQueryFeedsPassesWhenTheWorldCanAnswer(t *testing.T) {
|
||||
h := buildFeedHandler(t, false)
|
||||
h.search = &searchWiring{max: 3, runes: 1500}
|
||||
|
||||
if reply, ok := askFeeds(t, h, "что нового в лентах?"); ok {
|
||||
t.Fatalf("feeds off with a search configured must fall through, got %q", reply)
|
||||
}
|
||||
// With nothing below that reads the world, the honest status is still said:
|
||||
// general knowledge would otherwise answer with an invented bulletin.
|
||||
h.search = nil
|
||||
if reply, ok := askFeeds(t, h, "что нового в лентах?"); !ok || !strings.Contains(reply, "не настроены") {
|
||||
t.Fatalf("no search and no ZIMs: reply = %q, ok = %v", reply, ok)
|
||||
}
|
||||
}
|
||||
|
||||
+50
-3
@@ -1,17 +1,64 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/dialogue"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// voiceDialogueID — the single dialogue-session key. This is a single-user box
|
||||
// (ponytail), so one slot suffices; a second speaker would need per-speaker ids,
|
||||
// which waits on voice-print attribution (see PROGRESS multi-user deferral).
|
||||
// voiceDialogueID — the dialogue-session key for the microphone, and the
|
||||
// clarify key for it too. This is a single-user box (ponytail), so one slot
|
||||
// suffices; a second speaker would need per-speaker ids, which waits on
|
||||
// voice-print attribution (see PROGRESS multi-user deferral).
|
||||
const voiceDialogueID = "voice"
|
||||
|
||||
// textDialogueID — the clarify key for a text turn that named no conversation.
|
||||
// Separate from the mic: an old client that sends no id still must not answer
|
||||
// a question she asked out loud.
|
||||
const textDialogueID = "text"
|
||||
|
||||
// dialogueKey — the context key carrying the id of the conversation this turn
|
||||
// belongs to. It rides the context rather than a parameter for the same reason
|
||||
// the correlation id does: every step of the turn needs it, most of them only
|
||||
// to hand to the next one, and threading it by hand would put it in six
|
||||
// clarify signatures that have nothing else to say about it.
|
||||
type dialogueKey struct{}
|
||||
|
||||
// dialogueIDFor builds the id a turn is held under: the conversation the reach
|
||||
// named, qualified by the tap it arrived on, or the tap's own fallback when it
|
||||
// named none.
|
||||
//
|
||||
// A parked clarifying question used to be held under voiceDialogueID no matter
|
||||
// where the turn came from, so one unanswerable question captured the next
|
||||
// three utterances from anywhere. Three independent curl sessions fed a
|
||||
// capture attempt that had already failed, and a reminder among them was lost
|
||||
// (Vikunja #466).
|
||||
func dialogueIDFor(src turnSource, conversation string) string {
|
||||
if conversation != "" {
|
||||
return string(src) + ":" + conversation
|
||||
}
|
||||
if src == sourceVoice {
|
||||
return voiceDialogueID
|
||||
}
|
||||
return textDialogueID
|
||||
}
|
||||
|
||||
// withDialogueID tags a turn with that id.
|
||||
func withDialogueID(ctx context.Context, id string) context.Context {
|
||||
return context.WithValue(ctx, dialogueKey{}, id)
|
||||
}
|
||||
|
||||
// dialogueIDOf reads it back. Falls back to the microphone's slot, which is
|
||||
// what an unthreaded caller — a test, an internal replay — gets.
|
||||
func dialogueIDOf(ctx context.Context) string {
|
||||
if id, ok := ctx.Value(dialogueKey{}).(string); ok && id != "" {
|
||||
return id
|
||||
}
|
||||
return voiceDialogueID
|
||||
}
|
||||
|
||||
// toDialogueSlots and applyDialogueSlots are the only bridge between
|
||||
// router.Slots and dialogue.Slots. dialogue must not import router (import
|
||||
// cycle), so the two structs are hand-kept copies and every field has to be
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Command history — "что я тебе говорил?", "что ты записала сегодня?"
|
||||
// (Vikunja #456).
|
||||
//
|
||||
// Read-only over the facts that already exist. No new mechanism and no new
|
||||
// storage: everything he tapped in is already a row with a source and a
|
||||
// timestamp, and this only reads them back.
|
||||
|
||||
// historyMarkers — the ways he asks what he told her. Each entry is a pair of
|
||||
// substrings that must BOTH appear, because either half alone is a different
|
||||
// question: "что я говорил про сервер" is a recall question the notes pass
|
||||
// answers better, and "что ты записала" with no "что" is not a question at all.
|
||||
var historyMarkers = [][2]string{
|
||||
{"что я", "говорил"},
|
||||
{"что я", "сказал"},
|
||||
{"что я", "рассказ"},
|
||||
{"что ты", "записал"},
|
||||
{"что ты", "запомнил"},
|
||||
{"что я", "отмечал"},
|
||||
{"что я", "отметил"},
|
||||
{"what did i", "tell"},
|
||||
{"what did you", "record"},
|
||||
}
|
||||
|
||||
// historyRecall — the word that turns a history question into a recall
|
||||
// question. "что я говорил про сервер" names a topic, and the notes pass
|
||||
// answers a topic far better than a list of the last five facts does.
|
||||
var historyRecall = []string{" про ", " об ", " о ", " about "}
|
||||
|
||||
// isHistoryQuery reports whether he is asking what he told her.
|
||||
func isHistoryQuery(u string) bool {
|
||||
s := " " + strings.ToLower(strings.TrimSpace(u)) + " "
|
||||
if s == " " {
|
||||
return false
|
||||
}
|
||||
for _, r := range historyRecall {
|
||||
if strings.Contains(s, r) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
for _, pair := range historyMarkers {
|
||||
if strings.Contains(s, pair[0]) && strings.Contains(s, pair[1]) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// historyScan — how many recent facts are read before filtering. Deliberately
|
||||
// larger than historyReadOut: a poller writing every few minutes would
|
||||
// otherwise push everything he said out of the window, the same way his own
|
||||
// notes used to crowd out the feed headlines.
|
||||
const historyScan = 100
|
||||
|
||||
// historyReadOut — how many she says out loud. Five is what fits in one spoken
|
||||
// breath; the rest are on /history, which is the surface for reading a list.
|
||||
const historyReadOut = 5
|
||||
|
||||
// historyWindow — how far back "recently" reaches. A day, because the question
|
||||
// is about this conversation and not about the archive.
|
||||
const historyWindow = 24 * time.Hour
|
||||
|
||||
// queryHistory answers what he told her, from the facts he tapped in.
|
||||
//
|
||||
// Only "tap:" sources. A fact written by a poller, an inference or the ambient
|
||||
// relay is a thing she learned, not a thing he said, and reading those back
|
||||
// under "что я тебе говорил?" would put words in his mouth.
|
||||
//
|
||||
// Placed with the other sources that read his own rows and above the recall
|
||||
// pass: the notes pass would otherwise answer this from whatever note happens
|
||||
// to be nearest, which reads as an answer and is not one.
|
||||
func (h *reactiveHandler) queryHistory(ctx context.Context, t *queryTurn) (string, bool) {
|
||||
if !isHistoryQuery(t.dec.Utterance) {
|
||||
return "", false
|
||||
}
|
||||
facts, err := h.api.RecentFacts(ctx, historyScan)
|
||||
if err != nil {
|
||||
log.Printf("voice: history: recent facts: %v", err)
|
||||
return "не получилось посмотреть, что ты говорил.", true
|
||||
}
|
||||
cutoff := h.now().Add(-historyWindow)
|
||||
var said []string
|
||||
for _, f := range facts {
|
||||
if !strings.HasPrefix(f.Source, "tap:") || f.Ts.Before(cutoff) {
|
||||
continue
|
||||
}
|
||||
said = append(said, historyLine(f.Key, f.Value, f.Ts))
|
||||
if len(said) == historyReadOut {
|
||||
break
|
||||
}
|
||||
}
|
||||
if len(said) == 0 {
|
||||
// Claim the turn rather than fall through. "ничего не говорил" is the
|
||||
// true answer, and recall would answer it with an old note instead.
|
||||
return "за последние сутки ты мне ничего такого не говорил.", true
|
||||
}
|
||||
return "ты говорил: " + strings.Join(said, "; "), true
|
||||
}
|
||||
|
||||
// historyLine — one fact as she says it. The hour and minute, because the day
|
||||
// is already bounded by historyWindow and a date would be noise.
|
||||
func historyLine(key, value string, ts time.Time) string {
|
||||
what := key
|
||||
if value != "" {
|
||||
what = key + " — " + value
|
||||
}
|
||||
return fmt.Sprintf("%s (%s)", what, ts.Local().Format("15:04"))
|
||||
}
|
||||
@@ -0,0 +1,104 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// historyAPI serves a fixed set of recent facts.
|
||||
type historyAPI struct {
|
||||
ipc.UnimplementedCoreAPI
|
||||
facts []ipc.Fact
|
||||
calls int
|
||||
}
|
||||
|
||||
func (a *historyAPI) RecentFacts(context.Context, int) ([]ipc.Fact, error) {
|
||||
a.calls++
|
||||
return a.facts, nil
|
||||
}
|
||||
|
||||
func historyHandler(now time.Time, facts ...ipc.Fact) (*reactiveHandler, *historyAPI) {
|
||||
api := &historyAPI{facts: facts}
|
||||
return &reactiveHandler{api: api, now: func() time.Time { return now }}, api
|
||||
}
|
||||
|
||||
func askHistory(h *reactiveHandler, u string) (string, bool) {
|
||||
return h.queryHistory(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Intent: router.IntentQuery, Utterance: u},
|
||||
})
|
||||
}
|
||||
|
||||
func TestIsHistoryQuery(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
text string
|
||||
want bool
|
||||
}{
|
||||
{"что я тебе говорил?", true},
|
||||
{"что ты записала сегодня?", true},
|
||||
{"что я отмечал?", true},
|
||||
// A named topic is a recall question, and the notes pass answers it
|
||||
// better than a list of the last five facts does.
|
||||
{"что я говорил про сервер?", false},
|
||||
{"что у меня сегодня?", false},
|
||||
{"", false},
|
||||
} {
|
||||
if got := isHistoryQuery(tc.text); got != tc.want {
|
||||
t.Errorf("isHistoryQuery(%q) = %v, want %v", tc.text, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHistoryReadsOnlyWhatHeSaid(t *testing.T) {
|
||||
now := time.Date(2026, 8, 4, 20, 0, 0, 0, time.UTC)
|
||||
h, api := historyHandler(now,
|
||||
ipc.Fact{Key: "water", Value: "выпил", Source: "tap:voice", Ts: now.Add(-time.Hour)},
|
||||
// Learned, not said: a poller writing this back under "что я тебе
|
||||
// говорил?" would put words in his mouth.
|
||||
ipc.Fact{Key: "spent_today", Value: "1200", Source: "poll:zenmoney", Ts: now.Add(-time.Hour)},
|
||||
// Older than the window.
|
||||
ipc.Fact{Key: "shower", Value: "принял", Source: "tap:voice", Ts: now.Add(-30 * time.Hour)},
|
||||
)
|
||||
reply, ok := askHistory(h, "что я тебе говорил?")
|
||||
if !ok {
|
||||
t.Fatal("the history question must be claimed before the recall sources")
|
||||
}
|
||||
if !strings.Contains(reply, "water") {
|
||||
t.Errorf("reply = %q, want the fact he tapped in", reply)
|
||||
}
|
||||
if strings.Contains(reply, "spent_today") || strings.Contains(reply, "shower") {
|
||||
t.Errorf("reply = %q, want only what he said inside the window", reply)
|
||||
}
|
||||
if api.calls != 1 {
|
||||
t.Errorf("RecentFacts called %d times, want 1", api.calls)
|
||||
}
|
||||
}
|
||||
|
||||
// Nothing said is an answer of its own. Falling through would hand the question
|
||||
// to recall, which answers it with an old note.
|
||||
func TestHistorySaysWhenThereIsNothing(t *testing.T) {
|
||||
now := time.Date(2026, 8, 4, 20, 0, 0, 0, time.UTC)
|
||||
h, _ := historyHandler(now)
|
||||
reply, ok := askHistory(h, "что я тебе говорил?")
|
||||
if !ok || !strings.Contains(reply, "ничего") {
|
||||
t.Fatalf("reply = %q, ok = %v", reply, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// Five is what fits in one spoken breath; the rest are on /history.
|
||||
func TestHistoryStopsAtFive(t *testing.T) {
|
||||
now := time.Date(2026, 8, 4, 20, 0, 0, 0, time.UTC)
|
||||
var facts []ipc.Fact
|
||||
for i := 0; i < 12; i++ {
|
||||
facts = append(facts, ipc.Fact{Key: "k", Value: "v", Source: "tap:voice", Ts: now.Add(-time.Minute)})
|
||||
}
|
||||
h, _ := historyHandler(now, facts...)
|
||||
reply, _ := askHistory(h, "что ты записала?")
|
||||
if got := strings.Count(reply, ";"); got != historyReadOut-1 {
|
||||
t.Fatalf("reply = %q has %d separators, want %d", reply, got, historyReadOut-1)
|
||||
}
|
||||
}
|
||||
@@ -251,6 +251,7 @@ func run(args []string) error {
|
||||
Listen: cfg.Phraser.Listen,
|
||||
NGpuLayers: cfg.Phraser.NGpuLayers,
|
||||
NCtx: cfg.Phraser.NCtx,
|
||||
CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB),
|
||||
Timeout: time.Duration(cfg.Phraser.Timeout),
|
||||
LLMNudges: cfg.Phraser.LLMNudges,
|
||||
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||
@@ -525,6 +526,7 @@ func run(args []string) error {
|
||||
Listen: cfg.Phraser.Listen,
|
||||
NGpuLayers: cfg.Phraser.NGpuLayers,
|
||||
NCtx: cfg.Phraser.NCtx,
|
||||
CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB),
|
||||
Timeout: time.Duration(cfg.Phraser.Timeout),
|
||||
LLMNudges: cfg.Phraser.LLMNudges,
|
||||
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||
@@ -788,6 +790,22 @@ func personaFacts(cfg *config.Config) persona.Facts {
|
||||
return f
|
||||
}
|
||||
|
||||
// cacheRAMMiB resolves phraser.cache_ram_mib into the phraser's field. Unset
|
||||
// means 512 MiB and not "whatever the server does", because the server's own
|
||||
// default is 8 GiB of prompt cache and that is what put 7.9 GB of RSS and half
|
||||
// a gigabyte of swap on homesrv for a 1.1 GB model. A negative value is the
|
||||
// deliberate opt-out: no flag is passed, the server's default applies, and the
|
||||
// operator owns the consequence.
|
||||
func cacheRAMMiB(configured int) int {
|
||||
if configured == 0 {
|
||||
return 512
|
||||
}
|
||||
if configured < 0 {
|
||||
return 0
|
||||
}
|
||||
return configured
|
||||
}
|
||||
|
||||
// contextBlockFn returns the per-turn renderer of the shared context block.
|
||||
// Per turn, not once at startup, because the block states the current time.
|
||||
func contextBlockFn(cfg *config.Config, now func() time.Time) func() string {
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/llm"
|
||||
)
|
||||
|
||||
// No `workstation` block is the shipping deploy. The seam must then be the
|
||||
// resident client itself, with nothing probing anything.
|
||||
func TestModelSeamUnconfiguredIsResidentOnly(t *testing.T) {
|
||||
resident := llm.New("http://127.0.0.1:1", time.Second)
|
||||
hot, pair := modelSeam(&config.Config{}, resident)
|
||||
if pair != nil {
|
||||
t.Error("built a pair with no workstation configured")
|
||||
}
|
||||
if hot == nil {
|
||||
t.Fatal("no seam at all, so the cascade would route with the classifier")
|
||||
}
|
||||
}
|
||||
|
||||
// A workstation with no resident model behind it has no floor, and a Pair with
|
||||
// no floor is a configuration mistake rather than a degraded mode.
|
||||
func TestModelSeamWithoutResidentIsNil(t *testing.T) {
|
||||
cfg := &config.Config{Workstation: &config.WorkstationConfig{URL: "http://127.0.0.1:1"}}
|
||||
cfg.Workstation.Health = strings.TrimRight(cfg.Workstation.URL, "/") + "/health"
|
||||
hot, pair := modelSeam(cfg, nil)
|
||||
if hot != nil || pair != nil {
|
||||
t.Errorf("built a seam with no floor: hot=%v pair=%v", hot, pair)
|
||||
}
|
||||
}
|
||||
|
||||
// The configured case: the seam is the pair, and the pair notices a workstation
|
||||
// that answers /health.
|
||||
func TestModelSeamPrefersAnAnsweringWorkstation(t *testing.T) {
|
||||
up := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}))
|
||||
defer up.Close()
|
||||
|
||||
cfg := &config.Config{Workstation: &config.WorkstationConfig{
|
||||
URL: up.URL,
|
||||
Probe: config.Duration(10 * time.Millisecond),
|
||||
}}
|
||||
cfg.Workstation.Health = strings.TrimRight(cfg.Workstation.URL, "/") + "/health"
|
||||
|
||||
hot, pair := modelSeam(cfg, llm.New("http://127.0.0.1:1", time.Second))
|
||||
if pair == nil || hot == nil {
|
||||
t.Fatal("no pair built for a configured workstation")
|
||||
}
|
||||
defer pair.Stop()
|
||||
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for !pair.Available() && time.Now().Before(deadline) {
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
if !pair.Available() {
|
||||
t.Fatal("the pair never saw a workstation that answers /health")
|
||||
}
|
||||
}
|
||||
|
||||
// A card held by a CPT run answers 503, and that must read as unavailable
|
||||
// rather than as an error a turn has to handle.
|
||||
func TestModelSeamHeldCardIsUnavailable(t *testing.T) {
|
||||
busy := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "model not loaded", http.StatusServiceUnavailable)
|
||||
}))
|
||||
defer busy.Close()
|
||||
|
||||
cfg := &config.Config{Workstation: &config.WorkstationConfig{
|
||||
URL: busy.URL,
|
||||
Probe: config.Duration(10 * time.Millisecond),
|
||||
}}
|
||||
cfg.Workstation.Health = strings.TrimRight(cfg.Workstation.URL, "/") + "/health"
|
||||
|
||||
_, pair := modelSeam(cfg, llm.New("http://127.0.0.1:1", time.Second))
|
||||
if pair == nil {
|
||||
t.Fatal("no pair built for a configured workstation")
|
||||
}
|
||||
defer pair.Stop()
|
||||
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
if pair.Available() {
|
||||
t.Error("a 503 from the supervisor read as available")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/morning"
|
||||
)
|
||||
|
||||
// TestMorningNudgeBodySeparatesOptional — the one message a routine is allowed
|
||||
// per day says what was not done, then what he could still do (Vikunja #473).
|
||||
func TestMorningNudgeBodySeparatesOptional(t *testing.T) {
|
||||
cand := morning.Candidate{
|
||||
Routine: morning.Routine{Name: "утро"},
|
||||
Missing: []morning.Item{
|
||||
{Key: "meds", Label: "таблетки"},
|
||||
{Key: "stretch", Label: "растяжка", Optional: true},
|
||||
},
|
||||
}
|
||||
body := morningNudgeBody(cand)
|
||||
if !strings.Contains(body, "не сделано — таблетки") {
|
||||
t.Fatalf("the required item must be named as not done: %q", body)
|
||||
}
|
||||
if !strings.Contains(body, "если будет время — растяжка") {
|
||||
t.Fatalf("the optional item must read softer: %q", body)
|
||||
}
|
||||
if strings.Contains(body, "не сделано — таблетки, растяжка") {
|
||||
t.Fatalf("optional must not be folded into the required list: %q", body)
|
||||
}
|
||||
|
||||
// Nothing optional missing: the sentence is what it always was.
|
||||
only := morning.Candidate{
|
||||
Routine: morning.Routine{Name: "утро"},
|
||||
Missing: []morning.Item{{Key: "meds", Label: "таблетки"}},
|
||||
}
|
||||
if got, want := morningNudgeBody(only), "утро: не сделано — таблетки"; got != want {
|
||||
t.Fatalf("morningNudgeBody = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"math"
|
||||
"sync"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// The personal boundary decides one thing: is this question about him. It used
|
||||
// to decide it by matching possession words, and that was the whole defect
|
||||
// behind Vikunja #495. "что я говорил про бэкапы?" is his data by definition —
|
||||
// nothing outside the box has ever heard him say anything — and it carried no
|
||||
// possession word, so it walked past the boundary into SearXNG and came back
|
||||
// answered out of a Habr article about somebody else's backups.
|
||||
//
|
||||
// The first fix was one more marker class, `я говорил|сказал|писал|…`, plus a
|
||||
// carve-out so "как я говорил, почему небо синее" stayed a world question. Both
|
||||
// halves are a lexicon, and a lexicon is the wrong instrument here: Russian
|
||||
// gives every verb a dozen surface forms, the preamble list has no end, and
|
||||
// every utterance the list misses is one that reaches the world. It also drifts
|
||||
// silently — a missing verb looks exactly like no bug.
|
||||
//
|
||||
// So the boundary asks the embedder instead. Two frozen seed sets — questions
|
||||
// about him, questions about the world — are embedded once, and the turn's own
|
||||
// query vector, already computed by queryEmbed upstream, is scored against
|
||||
// both. Nearest side wins. Word order, verb form and unseen phrasing stop
|
||||
// mattering, which is exactly what a lexicon could not do.
|
||||
//
|
||||
// Measured 03-08-2026 against multilingual-e5-small on 19 held-out utterances,
|
||||
// none of them a seed: 19 right (TestONNXPersonalBoundary). A 20th, "as i said,
|
||||
// what is the population of india", missed by +0.008 during the first pass and
|
||||
// is a world seed now, which is why it is not in the held-out set. True
|
||||
// positives clear the world side by +0.014 to +0.089 and the nearest true
|
||||
// negative sits at -0.005, so the gate is the sign of the difference and
|
||||
// nothing tighter: the margins are too thin to justify a threshold, and the
|
||||
// asymmetry favours claiming anyway. A false claim costs one honest "не знаю";
|
||||
// a false pass sends his life to an upstream engine.
|
||||
//
|
||||
// The embedder is the one model CLAUDE.md pins to homesrv permanently, and it
|
||||
// is what makes this affordable: no llama-server call, no network, one cosine
|
||||
// per seed against a vector the turn already has.
|
||||
|
||||
// personalSeeds — questions about him. Frozen: they are scoring data, so
|
||||
// editing one moves the boundary and must be re-measured, not eyeballed. Cover
|
||||
// both classes the boundary owns, possession and first-person speech, in both
|
||||
// languages.
|
||||
var personalSeeds = []string{
|
||||
"что я говорил про это",
|
||||
"я тебе рассказывал об этом?",
|
||||
"что я записал про врача",
|
||||
"я упоминал эту тему?",
|
||||
"что у меня сегодня",
|
||||
"когда моя встреча",
|
||||
"what did i say about this",
|
||||
"did i mention this to you",
|
||||
}
|
||||
|
||||
// worldSeeds — questions the world can answer, including the two shapes that
|
||||
// look personal and are not: a first-person preamble on a world question ("как
|
||||
// я говорил, ..."), and first person without possession ("что я могу
|
||||
// посмотреть вечером"). Refusing those is the opposite mistake and the older
|
||||
// comment on personalMarkers already named it.
|
||||
var worldSeeds = []string{
|
||||
"почему небо синее",
|
||||
"какая столица франции",
|
||||
"как сварить борщ",
|
||||
"кто написал эту книгу",
|
||||
"what is the capital of france",
|
||||
"how do i boil an egg",
|
||||
"как я говорил, почему небо синее",
|
||||
"as i said, why is the sky blue",
|
||||
"as i said, what is the population of india",
|
||||
"что я могу посмотреть вечером",
|
||||
"что мне почитать про историю",
|
||||
"что я должен знать про питон",
|
||||
"what can i watch tonight",
|
||||
}
|
||||
|
||||
// personalBoundary holds the embedded seeds. Zero value is usable and means
|
||||
// "not loaded yet"; a handler built without an embedder never loads and the
|
||||
// boundary falls back to personalMarkers.
|
||||
type personalBoundary struct {
|
||||
once sync.Once
|
||||
personal [][]float32
|
||||
world [][]float32
|
||||
loaded bool
|
||||
}
|
||||
|
||||
// load embeds both seed sets, once per process. Seeds are embedded on the QUERY
|
||||
// side, like the utterance they are compared with — a question against a
|
||||
// question. Mixing sides would measure the e5 prefix, not the meaning.
|
||||
func (b *personalBoundary) load(ctx context.Context, emb router.Embedder) {
|
||||
b.once.Do(func() {
|
||||
if emb == nil {
|
||||
return
|
||||
}
|
||||
embedAll := func(ss []string) [][]float32 {
|
||||
out := make([][]float32, 0, len(ss))
|
||||
for _, s := range ss {
|
||||
v, err := router.EmbedQuery(ctx, emb, s)
|
||||
if err != nil {
|
||||
log.Printf("voice: personal boundary seeds unavailable (%v); falling back to possession markers", err)
|
||||
return nil
|
||||
}
|
||||
out = append(out, v)
|
||||
}
|
||||
return out
|
||||
}
|
||||
p, w := embedAll(personalSeeds), embedAll(worldSeeds)
|
||||
if p == nil || w == nil {
|
||||
return
|
||||
}
|
||||
b.personal, b.world, b.loaded = p, w, true
|
||||
})
|
||||
}
|
||||
|
||||
// score returns the best similarity to each side. ok is false when the seeds
|
||||
// are not loaded, which is the caller's signal to use the markers instead.
|
||||
func (b *personalBoundary) score(vec []float32) (personal, world float64, ok bool) {
|
||||
if !b.loaded || len(vec) == 0 {
|
||||
return 0, 0, false
|
||||
}
|
||||
best := func(seeds [][]float32) float64 {
|
||||
m := -1.0
|
||||
for _, s := range seeds {
|
||||
if c := cosine(vec, s); c > m {
|
||||
m = c
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
return best(b.personal), best(b.world), true
|
||||
}
|
||||
|
||||
// cosine — same math as internal/router and internal/memory, small enough that
|
||||
// importing one of them for it would be the larger coupling.
|
||||
func cosine(a, b []float32) float64 {
|
||||
if len(a) != len(b) {
|
||||
return 0
|
||||
}
|
||||
var dot, na, nb float64
|
||||
for i := range a {
|
||||
dot += float64(a[i]) * float64(b[i])
|
||||
na += float64(a[i]) * float64(a[i])
|
||||
nb += float64(b[i]) * float64(b[i])
|
||||
}
|
||||
if na == 0 || nb == 0 {
|
||||
return 0
|
||||
}
|
||||
return dot / (math.Sqrt(na) * math.Sqrt(nb))
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// A handler with no embedder never loads the seeds, so the boundary falls back
|
||||
// to the possession markers. That is the offline floor and it must keep working
|
||||
// — an embedder that fails to load must not open the boundary.
|
||||
func TestBoundaryFallsBackToMarkersWithNoEmbedder(t *testing.T) {
|
||||
h := personalHandler()
|
||||
if !h.isPersonalTurn(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Utterance: "во сколько у меня встреча"},
|
||||
}) {
|
||||
t.Error("no embedder: a possession question must still be personal")
|
||||
}
|
||||
if h.isPersonalTurn(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Utterance: "почему небо синее"},
|
||||
}) {
|
||||
t.Error("no embedder: a world question must still pass")
|
||||
}
|
||||
}
|
||||
|
||||
// TestONNXPersonalBoundary — the number that matters, scored against the
|
||||
// embedder homesrv actually runs. Opt-in via MAVEN_ONNX_LIB, exactly like
|
||||
// TestONNXRecall in internal/memory/recalleval.
|
||||
//
|
||||
// Every case here is held out: none of these strings is a seed. The #495
|
||||
// regression is the first row — "что я говорил про бэкапы?" reached SearXNG and
|
||||
// was answered from a Habr article, and no possession word appears in it.
|
||||
func TestONNXPersonalBoundary(t *testing.T) {
|
||||
lib := os.Getenv("MAVEN_ONNX_LIB")
|
||||
if lib == "" {
|
||||
t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing")
|
||||
}
|
||||
dir := filepath.Join("../..", "models/embedder/multilingual-e5-small")
|
||||
emb, err := router.NewONNXEmbedder(filepath.Join(dir, "model_quantized.onnx"), filepath.Join(dir, "tokenizer.json"), lib)
|
||||
if err != nil {
|
||||
t.Skipf("onnx embedder unavailable: %v", err)
|
||||
}
|
||||
defer emb.Close()
|
||||
|
||||
cases := []struct {
|
||||
utterance string
|
||||
personal bool
|
||||
}{
|
||||
{"что я говорил про бэкапы?", true},
|
||||
{"что я сказал вчера про отпуск", true},
|
||||
{"я писал что-нибудь про сервер", true},
|
||||
{"я упоминал про конференцию?", true},
|
||||
{"что я отмечал по поводу переезда", true},
|
||||
{"я рассказывал тебе про новую работу?", true},
|
||||
{"во сколько у меня встреча", true},
|
||||
{"когда мой следующий отпуск", true},
|
||||
{"what did i say about backups", true},
|
||||
{"did i tell you about the doctor", true},
|
||||
{"как я говорил, почему небо синее", false},
|
||||
{"как уже я говорил, какая столица франции", false},
|
||||
{"почему трава зелёная", false},
|
||||
{"столица франции", false},
|
||||
{"как мне сварить борщ", false},
|
||||
{"что мне посмотреть вечером", false},
|
||||
{"я хочу узнать про рим", false},
|
||||
{"кто такой гагарин", false},
|
||||
{"how do i boil an egg", false},
|
||||
}
|
||||
|
||||
h := &reactiveHandler{embedder: emb}
|
||||
ctx := context.Background()
|
||||
wrong := 0
|
||||
for _, c := range cases {
|
||||
vec, err := router.EmbedQuery(ctx, emb, c.utterance)
|
||||
if err != nil {
|
||||
t.Fatalf("embed %q: %v", c.utterance, err)
|
||||
}
|
||||
turn := &queryTurn{dec: router.Decision{Utterance: c.utterance}, vec: vec}
|
||||
got := h.isPersonalTurn(ctx, turn)
|
||||
p, w, ok := h.boundary.score(vec)
|
||||
if !ok {
|
||||
t.Fatal("seeds did not load with a working embedder")
|
||||
}
|
||||
if got != c.personal {
|
||||
wrong++
|
||||
t.Errorf("%q: personal=%v want %v (personal %.4f world %.4f)", c.utterance, got, c.personal, p, w)
|
||||
}
|
||||
t.Logf("personal=%-5v personal %.4f world %.4f delta %+.4f %s", got, p, w, p-w, c.utterance)
|
||||
}
|
||||
t.Logf("personal boundary: %d/%d held-out utterances correct", len(cases)-wrong, len(cases))
|
||||
}
|
||||
@@ -2,12 +2,14 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/store"
|
||||
"github.com/kami/maven/internal/tool"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
@@ -85,3 +87,38 @@ func TestReactiveNotesReminders(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestSpokenTaskCaptureFilesATask — the whole path, from the utterance to the
|
||||
// task table. It went dead when the router started claiming the marker as an
|
||||
// act: capture rides the note intent, so nothing below actionNote was ever
|
||||
// reached and every capture answered "Что сделать?" (Vikunja #467).
|
||||
func TestSpokenTaskCaptureFilesATask(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
st := newTestStore(t)
|
||||
api := ipc.NewStoreAPI(st)
|
||||
now := time.Now()
|
||||
emb := router.NewHashEmbedder(1024)
|
||||
matcher := tool.NewMatcher(api)
|
||||
h := &reactiveHandler{
|
||||
api: api,
|
||||
embedder: emb,
|
||||
router: buildRouter(emb, matcher, 0.55, nil),
|
||||
replier: voice.NewStubReplier(),
|
||||
now: func() time.Time { return now },
|
||||
memStore: memory.NewInMemoryStore(),
|
||||
dataStore: st,
|
||||
}
|
||||
|
||||
reply := h.handleText(ctx, "web", "добавь в задачи купить молоко")
|
||||
if !strings.Contains(reply, "купить молоко") {
|
||||
t.Fatalf("capture did not claim the turn: %q", reply)
|
||||
}
|
||||
open, err := st.ListTasks(ctx, store.TaskOpen)
|
||||
if err != nil || len(open) != 1 {
|
||||
t.Fatalf("task was not filed: tasks=%v err=%v", open, err)
|
||||
}
|
||||
// The words he said, not the model's rewrite of them.
|
||||
if open[0].Text != "купить молоко" {
|
||||
t.Fatalf("task text was rewritten: %q", open[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// reminderMarker — the words that open a reminder. Stripped because they are
|
||||
// the instruction, not the thing to say at the hour.
|
||||
var reminderMarker = regexp.MustCompile(`(?i)^\s*(?:напомни(?:те)?|напомнить|remind)\s*(?:мне|me)?[\s,:—-]*`)
|
||||
|
||||
// reminderTimeWords — the time expressions a reminder carries, removed from
|
||||
// the body because the fire time is already a column. Ordered longest-first
|
||||
// where two could match the same words, so "через полтора часа" does not leave
|
||||
// "полтора" behind.
|
||||
//
|
||||
// Go's \b is ASCII-only and never fires next to a Cyrillic letter, so the word
|
||||
// boundaries here are written out as whitespace or an end of string — the same
|
||||
// trap the agenda grammars hit.
|
||||
var reminderTimeWords = []*regexp.Regexp{
|
||||
regexp.MustCompile(`(?i)(^|\s)через\s+\S+(\s+(часа?|часов|минут[уы]?|секунд[уы]?|дня|дней|недел[юи]))?(\s|$)`),
|
||||
regexp.MustCompile(`(?i)(^|\s)(в|во)\s+\d{1,2}(:\d{2})?(\s*(часа?|часов))?(\s*(утра|вечера|дня|ночи))?(\s|$)`),
|
||||
regexp.MustCompile(`(?i)(^|\s)(завтра|послезавтра|сегодня|вечером|утром|днём|днем|ночью)(\s|$)`),
|
||||
regexp.MustCompile(`(?i)(^|\s)(at|in)\s+\d{1,2}(:\d{2})?\s*(am|pm)?(\s|$)`),
|
||||
regexp.MustCompile(`(?i)(^|\s)(tomorrow|today|tonight)(\s|$)`),
|
||||
}
|
||||
|
||||
// reminderBody is what she says at the hour.
|
||||
//
|
||||
// The whole utterance used to be stored, so /reminders read "напомни завтра в
|
||||
// 9 утра выпить таблетки" where it should read "выпить таблетки", and the
|
||||
// agenda recited the marker back at him (Vikunja #469). The fire time is
|
||||
// already a column, and the marker is an instruction that was carried out.
|
||||
//
|
||||
// Falls back to the fuller text whenever stripping would leave nothing: an
|
||||
// empty body is a reminder that fires and says nothing, which is worse than a
|
||||
// wordy one.
|
||||
func reminderBody(utterance, text string) string {
|
||||
body := strings.TrimSpace(text)
|
||||
if body == "" {
|
||||
body = strings.TrimSpace(utterance)
|
||||
}
|
||||
stripped := reminderMarker.ReplaceAllString(body, "")
|
||||
for _, re := range reminderTimeWords {
|
||||
stripped = re.ReplaceAllString(stripped, " ")
|
||||
}
|
||||
stripped = strings.TrimSpace(strings.Join(strings.Fields(stripped), " "))
|
||||
stripped = strings.Trim(stripped, " ,;:—-")
|
||||
if stripped == "" {
|
||||
return body
|
||||
}
|
||||
return stripped
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package main
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestReminderBody(t *testing.T) {
|
||||
for _, tc := range []struct{ utterance, text, want string }{
|
||||
// The row from the QA sitting: the whole utterance was the body.
|
||||
{"напомни завтра в 9 утра выпить таблетки", "завтра в 9 утра выпить таблетки", "выпить таблетки"},
|
||||
{"напомни мне позвонить маме в семь вечера", "позвонить маме в 7 вечера", "позвонить маме"},
|
||||
{"напомни через полчаса проверить бэкап", "через полчаса проверить бэкап", "проверить бэкап"},
|
||||
{"remind me to call mom at 7pm", "to call mom at 7pm", "to call mom"},
|
||||
// Nothing left after stripping ⇒ keep what there was. A reminder that
|
||||
// fires and says nothing is worse than a wordy one.
|
||||
{"напомни завтра", "завтра", "завтра"},
|
||||
{"", "", ""},
|
||||
} {
|
||||
if got := reminderBody(tc.utterance, tc.text); got != tc.want {
|
||||
t.Errorf("reminderBody(%q, %q) = %q, want %q", tc.utterance, tc.text, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
+18
-102
@@ -2,122 +2,38 @@ package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/persona"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
|
||||
// completer is the LLM seam for the replier (subset of router.Completer).
|
||||
// *llm.Client satisfies it.
|
||||
type completer interface {
|
||||
Complete(ctx context.Context, r llm.Req) (string, error)
|
||||
}
|
||||
|
||||
// llmReplier phrases reactive confirmations with the resident model
|
||||
// (Qwen3-1.7B). Stub is the
|
||||
// floor on any error (offline-safe). Maven speaks as "she", feminine RU.
|
||||
// llmReplier is the daemon-side wiring around phraser.Replier: it owns the
|
||||
// deterministic floor, and nothing else. The phrasing itself, the prompt and the
|
||||
// output parsing live in internal/phraser so the eval can score them (#396).
|
||||
type llmReplier struct {
|
||||
c completer
|
||||
p *phraser.Replier
|
||||
stub *voice.StubReplier
|
||||
|
||||
// block renders the shared context block per turn (who he is, the time).
|
||||
// nil ⇒ the prompt stands alone.
|
||||
block func() string
|
||||
}
|
||||
|
||||
func newLLMReplier(c completer, block func() string) *llmReplier {
|
||||
return &llmReplier{c: c, stub: voice.NewStubReplier(), block: block}
|
||||
func newLLMReplier(c phraser.Completer, block func() string) *llmReplier {
|
||||
return &llmReplier{p: phraser.NewReplier(c, block), stub: voice.NewStubReplier()}
|
||||
}
|
||||
|
||||
const replySystem = `Ты — Maven, домашняя ассистентка (о себе — в женском роде). Владелец — мужчина, говоришь с ним на "ты", в единственном числе; никогда не "вы"/"ваш" и не "он"/"его". Подтверди действие РОВНО ОДНИМ коротким предложением (≤120 символов), по-русски, спокойно и без официальных формулировок. Не задавай вопросов, не повторяй слова, не добавляй ничего после точки. Отвечай ТОЛЬКО одним объектом JSON с полями "response" (текст) и "mood" (ровно одно из: neutral, happy, thinking, tired, confused).
|
||||
Пример: {"response": "Записала, что ты выпил стакан воды.", "mood": "neutral"}
|
||||
Никогда не пиши "..." в поле response.`
|
||||
|
||||
// Reply never fails: a clarify, a model error and an unusable generation all
|
||||
// answer from the stub, which is what keeps a turn from breaking on the model.
|
||||
func (r *llmReplier) Reply(d router.Decision) string {
|
||||
if d.Clarify {
|
||||
// The deck, not the stub's single sentence: a clarify she cannot turn
|
||||
// into a question is the line he hears most often when she misses him,
|
||||
// and it used to be the same words every time (Vikunja #457). Still no
|
||||
// model call — this text has to be right every time, and it is not worth
|
||||
// a generation to say something this small.
|
||||
return clarifyMissedLine(d)
|
||||
}
|
||||
out, err := r.p.PhraseReply(context.Background(), d)
|
||||
if err != nil || out == "" {
|
||||
return r.stub.Reply(d)
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
out, err := r.c.Complete(ctx, llm.Req{
|
||||
System: persona.Prepend(r.block, replySystem),
|
||||
User: replyContext(d),
|
||||
Grammar: phraser.ResponseGrammar,
|
||||
MaxTokens: 512,
|
||||
RepeatPenalty: 1.3,
|
||||
})
|
||||
if err != nil {
|
||||
return r.stub.Reply(d)
|
||||
}
|
||||
out = stripThink(out)
|
||||
if response, _ := parseResponseMood(out); response != "" {
|
||||
return response
|
||||
}
|
||||
// fallback: try plain-text parsing
|
||||
if out = firstSentence(out); out != "" {
|
||||
return out
|
||||
}
|
||||
return r.stub.Reply(d)
|
||||
}
|
||||
|
||||
// firstSentence trims the model's output to a single clean confirmation: first
|
||||
// line, first sentence, whitespace-normalized — the last-line defense against a
|
||||
// small model that rambles past the first period despite the prompt + stop.
|
||||
// stripThink removes the <think> block that Thinking-variant models emit.
|
||||
func stripThink(s string) string {
|
||||
if i := strings.LastIndex(s, "</think>"); i >= 0 {
|
||||
s = strings.TrimSpace(s[i+8:])
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func firstSentence(s string) string {
|
||||
s = strings.TrimSpace(s)
|
||||
if i := strings.IndexByte(s, '\n'); i >= 0 {
|
||||
s = s[:i]
|
||||
}
|
||||
// keep up to and including the first sentence-ending punctuation.
|
||||
if i := strings.IndexAny(s, ".!?"); i >= 0 {
|
||||
s = s[:i+1]
|
||||
}
|
||||
return strings.TrimSpace(s)
|
||||
}
|
||||
|
||||
// parseResponseMood extracts {"response","mood"} from LLM output, tolerant
|
||||
// of thinking tokens and extra text before/after the JSON block.
|
||||
func parseResponseMood(raw string) (response, mood string) {
|
||||
cleaned := strings.TrimSpace(raw)
|
||||
start := strings.Index(cleaned, "{")
|
||||
end := strings.LastIndex(cleaned, "}")
|
||||
if start < 0 || end < 0 || end <= start {
|
||||
return "", ""
|
||||
}
|
||||
var parsed struct {
|
||||
Response string `json:"response"`
|
||||
Mood string `json:"mood"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(cleaned[start:end+1]), &parsed); err != nil {
|
||||
return "", ""
|
||||
}
|
||||
return parsed.Response, parsed.Mood
|
||||
}
|
||||
|
||||
// replyContext renders the decision into a compact RU description for the model.
|
||||
func replyContext(d router.Decision) string {
|
||||
switch d.Intent {
|
||||
case router.IntentFact:
|
||||
return "записала факт: " + d.Slots.Key + " " + d.Slots.Value
|
||||
case router.IntentNote:
|
||||
return "сохранила заметку: " + d.Slots.Text
|
||||
case router.IntentReminder:
|
||||
return "поставила напоминание: " + d.Slots.Text
|
||||
default:
|
||||
return string(d.Intent) + ": " + d.Slots.Text
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -5,28 +5,22 @@ import (
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/voice"
|
||||
)
|
||||
|
||||
type mockCompleter struct {
|
||||
// The phrasing itself is tested in internal/phraser. What is left here is the
|
||||
// only thing the daemon adds: the stub floor, on the three ways a reply can
|
||||
// fail to arrive.
|
||||
type stubCompleter struct {
|
||||
out string
|
||||
err error
|
||||
}
|
||||
|
||||
func (m mockCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return m.out, m.err }
|
||||
func (s stubCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return s.out, s.err }
|
||||
|
||||
func TestLLMReplierReturnsLLMReply(t *testing.T) {
|
||||
r := newLLMReplier(mockCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
|
||||
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
||||
if got != "записала, кофе закончился" {
|
||||
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLLMReplierFallsBackToPlainText(t *testing.T) {
|
||||
r := newLLMReplier(mockCompleter{out: "записала, кофе закончился"}, nil)
|
||||
func TestLLMReplierPassesTheModelReplyThrough(t *testing.T) {
|
||||
r := newLLMReplier(stubCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
|
||||
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
||||
if got != "записала, кофе закончился" {
|
||||
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
|
||||
@@ -34,54 +28,42 @@ func TestLLMReplierFallsBackToPlainText(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestLLMReplierFallsBackToStubOnError(t *testing.T) {
|
||||
r := newLLMReplier(mockCompleter{err: errTestLLMDown}, nil)
|
||||
noteDec := router.Decision{Intent: router.IntentNote}
|
||||
got := r.Reply(noteDec)
|
||||
want := voice.NewStubReplier().Reply(noteDec)
|
||||
if got != want {
|
||||
t.Errorf("on llm error: got %q, want stub %q", got, want)
|
||||
}
|
||||
r := newLLMReplier(stubCompleter{err: errReplierTest}, nil)
|
||||
assertStub(t, r, router.Decision{Intent: router.IntentNote}, "llm error")
|
||||
}
|
||||
|
||||
func TestLLMReplierFallsBackToStubOnEmpty(t *testing.T) {
|
||||
r := newLLMReplier(mockCompleter{out: ""}, nil)
|
||||
noteDec := router.Decision{Intent: router.IntentNote}
|
||||
got := r.Reply(noteDec)
|
||||
want := voice.NewStubReplier().Reply(noteDec)
|
||||
if got != want {
|
||||
t.Errorf("on empty llm: got %q, want stub %q", got, want)
|
||||
r := newLLMReplier(stubCompleter{out: ""}, nil)
|
||||
assertStub(t, r, router.Decision{Intent: router.IntentNote}, "empty llm")
|
||||
}
|
||||
|
||||
// A clarify never reaches the model, and since Vikunja #457 it is answered from
|
||||
// the clarify deck rather than the stub's single sentence.
|
||||
func TestLLMReplierClarifyReadsTheDeck(t *testing.T) {
|
||||
r := newLLMReplier(stubCompleter{out: "я всё поняла"}, nil)
|
||||
got := r.Reply(router.Decision{Clarify: true, Utterance: "мгм"})
|
||||
if got == "я всё поняла" {
|
||||
t.Fatal("a clarify must not be phrased by the model")
|
||||
}
|
||||
if want := clarifyMissedFor("мгм"); got != want {
|
||||
t.Errorf("on clarify: got %q, want %q", got, want)
|
||||
}
|
||||
// Two different misses do not sound identical.
|
||||
if same := r.Reply(router.Decision{Clarify: true, Utterance: "а"}); same == got {
|
||||
t.Log("two utterances hashed to the same line, which is allowed but should be rare")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLLMReplierClarifyUsesStub(t *testing.T) {
|
||||
r := newLLMReplier(mockCompleter{out: "я всё поняла"}, nil)
|
||||
clarifyDec := router.Decision{Clarify: true}
|
||||
got := r.Reply(clarifyDec)
|
||||
want := voice.NewStubReplier().Reply(clarifyDec)
|
||||
func assertStub(t *testing.T, r *llmReplier, d router.Decision, what string) {
|
||||
t.Helper()
|
||||
got, want := r.Reply(d), voice.NewStubReplier().Reply(d)
|
||||
if got != want {
|
||||
t.Errorf("on clarify: got %q, want stub %q", got, want)
|
||||
t.Errorf("on %s: got %q, want stub %q", what, got, want)
|
||||
}
|
||||
}
|
||||
|
||||
var errTestLLMDown = errTest("llm down")
|
||||
var errReplierTest = errTest("llm down")
|
||||
|
||||
type errTest string
|
||||
|
||||
func (e errTest) Error() string { return string(e) }
|
||||
|
||||
// grammarRecorder captures the request so the grammar can be asserted on.
|
||||
type grammarRecorder struct{ req llm.Req }
|
||||
|
||||
func (g *grammarRecorder) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||
g.req = r
|
||||
return `{"response":"записала","mood":"neutral"}`, nil
|
||||
}
|
||||
|
||||
func TestLLMReplierCarriesTheResponseGrammar(t *testing.T) {
|
||||
rec := &grammarRecorder{}
|
||||
r := newLLMReplier(rec, nil)
|
||||
r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
||||
if rec.req.Grammar != phraser.ResponseGrammar {
|
||||
t.Errorf("grammar = %q, want phraser.ResponseGrammar", rec.req.Grammar)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1043,3 +1043,26 @@ func TestSimulatorRefusesBackwardsSteps(t *testing.T) {
|
||||
t.Errorf("the clock moved to %s on a refused step, it must stay at 09:00", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSimulatorRoutesWithTheDeployedSeeds — the scenarios must replay against
|
||||
// the classifier the deploy runs, not an empty one.
|
||||
//
|
||||
// They did not. The seed path was relative to the working directory, which is
|
||||
// cmd/mavend under `go test`, so every file failed to open and the whole
|
||||
// simulator scored three green scenarios with zero examples loaded (Vikunja
|
||||
// #465). The count is asserted rather than logged, because a silent zero is
|
||||
// exactly the failure that hid here for as long as it did.
|
||||
func TestSimulatorRoutesWithTheDeployedSeeds(t *testing.T) {
|
||||
cls := router.NewClassifier(router.NewHashEmbedder(1024))
|
||||
seedClassifier(cls)
|
||||
total := 0
|
||||
for _, intent := range cls.Intents() {
|
||||
total += len(cls.Examples(intent))
|
||||
}
|
||||
if total == 0 {
|
||||
t.Fatalf("no seed examples loaded from %s — the simulator would route on nothing", seedPath())
|
||||
}
|
||||
if len(cls.Intents()) != 7 {
|
||||
t.Fatalf("seeded %d intents, want all 7", len(cls.Intents()))
|
||||
}
|
||||
}
|
||||
|
||||
+24
-8
@@ -761,11 +761,7 @@ func (t *tickLoop) fireMorningRoutines(ctx context.Context, now time.Time, state
|
||||
facts := t.gatherMorningFacts(ctx)
|
||||
|
||||
for _, cand := range morning.Due(t.morningRoutines, facts, t.morningLast, now) {
|
||||
labels := make([]string, len(cand.Missing))
|
||||
for i, it := range cand.Missing {
|
||||
labels[i] = it.Label
|
||||
}
|
||||
body := fmt.Sprintf("%s: не сделано — %s", cand.Routine.Name, strings.Join(labels, ", "))
|
||||
body := morningNudgeBody(cand)
|
||||
pn := delivery.PhrasedNudge{
|
||||
Candidate: loop.Candidate{
|
||||
Rule: loop.Rule{Name: "morning:" + cand.Routine.Name, Severity: loop.Severity(cand.Routine.Severity)},
|
||||
@@ -781,6 +777,26 @@ func (t *tickLoop) fireMorningRoutines(ctx context.Context, now time.Time, state
|
||||
}
|
||||
}
|
||||
|
||||
// morningNudgeBody words the one message a routine gets per day. Required
|
||||
// items are what she says was not done; optional ones follow, worded as
|
||||
// something he could still do rather than something he owes (Vikunja #473).
|
||||
// Operator text, not phrased by the model, for the same reason it always was:
|
||||
// a checklist item must not be invented.
|
||||
func morningNudgeBody(cand morning.Candidate) string {
|
||||
labels := func(items []morning.Item) string {
|
||||
out := make([]string, len(items))
|
||||
for i, it := range items {
|
||||
out[i] = it.Label
|
||||
}
|
||||
return strings.Join(out, ", ")
|
||||
}
|
||||
body := fmt.Sprintf("%s: не сделано — %s", cand.Routine.Name, labels(morning.Required(cand.Missing)))
|
||||
if opt := morning.OptionalOnly(cand.Missing); len(opt) > 0 {
|
||||
body += fmt.Sprintf(". если будет время — %s", labels(opt))
|
||||
}
|
||||
return body
|
||||
}
|
||||
|
||||
// gatherMorningFacts reads the latest fact for every item's fact_key across
|
||||
// all configured morning routines. Shared by fireMorningRoutines (nudge
|
||||
// decision) and morningStatus (read-only query) so the two paths can never
|
||||
@@ -986,7 +1002,7 @@ type daemonAPI struct {
|
||||
getTrace func() *loop.TickTrace
|
||||
getMorningStatus func(ctx context.Context) []ipc.MorningRoutineStatus
|
||||
getDayPlan func(ctx context.Context) ipc.DayPlan
|
||||
chatFn func(ctx context.Context, text string) string
|
||||
chatFn func(ctx context.Context, conversation, text string) string
|
||||
getMCPServers func() []ipc.MCPServerStatus
|
||||
getEvents func(n int) []ipc.IntakeEvent
|
||||
}
|
||||
@@ -1002,11 +1018,11 @@ func (d *daemonAPI) RecentEvents(ctx context.Context, n int) ([]ipc.IntakeEvent,
|
||||
return d.getEvents(n), nil
|
||||
}
|
||||
|
||||
func (d *daemonAPI) Chat(ctx context.Context, text string) (string, error) {
|
||||
func (d *daemonAPI) Chat(ctx context.Context, conversation, text string) (string, error) {
|
||||
if d.chatFn == nil {
|
||||
return "", errors.New("mavend: chat not available")
|
||||
}
|
||||
return d.chatFn(ctx, text), nil
|
||||
return d.chatFn(ctx, conversation, text), nil
|
||||
}
|
||||
|
||||
// MCPServers — the configured MCP servers and their health (Vikunja #251).
|
||||
|
||||
+11
-4
@@ -76,6 +76,10 @@ type reactiveHandler struct {
|
||||
tts tts.Synthesizer
|
||||
router *router.Router
|
||||
embedder router.Embedder // reused for note write/query (same model as the classifier)
|
||||
// boundary — the embedded seed sets behind the personal boundary
|
||||
// (personalboundary.go). Zero value is usable and loads on first query;
|
||||
// with no embedder it never loads and the boundary uses personalMarkers.
|
||||
boundary personalBoundary
|
||||
// api — the CoreAPI the handler reads and writes through. Wired with the
|
||||
// bare store adapter and UPGRADED by main once the daemonAPI exists; see
|
||||
// upgradeAPI.
|
||||
@@ -216,9 +220,9 @@ func (h *reactiveHandler) upgradeAPI(api ipc.CoreAPI) {
|
||||
// handleText — the core reactive path without stt/tts. Used by the IPC Chat
|
||||
// endpoint (and eventually by telegram). Splits out the audio bookends from
|
||||
// HandlePushToTalk so text channels share the same routing logic.
|
||||
func (h *reactiveHandler) handleText(ctx context.Context, text string) string {
|
||||
func (h *reactiveHandler) handleText(ctx context.Context, conversation, text string) string {
|
||||
log.Printf("voice: handleText: %q", text)
|
||||
return h.runTurn(ctx, text, sourceText)
|
||||
return h.runTurn(withDialogueID(ctx, dialogueIDFor(sourceText, conversation)), text, sourceText)
|
||||
}
|
||||
|
||||
// turnSource — which channel this utterance arrived on, in the same provenance
|
||||
@@ -251,7 +255,7 @@ func (h *reactiveHandler) runTurn(ctx context.Context, text string, src turnSour
|
||||
// early. He can be asked a question, walk off, come back and say "да" to a
|
||||
// confirm that is still parked; computing the notice after that return meant
|
||||
// he answered the confirm and never heard that the older request was let go.
|
||||
expiredNotice := h.clarifyExpiredNotice()
|
||||
expiredNotice := h.clarifyExpiredNotice(ctx)
|
||||
|
||||
// 2. confirm turn — if a destructive act is parked, this utterance is its
|
||||
// y/n answer, not a fresh command. Handled before routing so "да" doesn't
|
||||
@@ -347,7 +351,10 @@ func (h *reactiveHandler) runTurn(ctx context.Context, text string, src turnSour
|
||||
// and park the request (clarify.go); otherwise the replier's canned reply
|
||||
// stands.
|
||||
if dec.Clarify {
|
||||
if question, asked := h.askClarify(dec); asked {
|
||||
if reply := h.hexisBeforeClarify(ctx, dec); reply != "" {
|
||||
return withNotice(expiredNotice, reply)
|
||||
}
|
||||
if question, asked := h.askClarify(ctx, dec); asked {
|
||||
return withNotice(expiredNotice, question)
|
||||
}
|
||||
}
|
||||
|
||||
+130
-11
@@ -48,7 +48,11 @@ type voiceWiring struct {
|
||||
// mcp — the MCP client, nil unless the `mcp` block configures an enabled
|
||||
// server (Vikunja #251). Its tools land in the same allowlist as every
|
||||
// other act, so nothing else here has to know about it.
|
||||
mcp *mcpWiring
|
||||
// pair — the workstation model with the resident one as the floor, nil
|
||||
// unless a `workstation` block names an address. Held here only so the
|
||||
// prober is stopped on shutdown; callers were handed it at build time.
|
||||
pair *llm.Pair
|
||||
mcp *mcpWiring
|
||||
// home — the Home Assistant client, nil unless the `smarthome` block is
|
||||
// enabled (Vikunja #256). Its devices land in the same allowlist as every
|
||||
// other act, so nothing else here has to know about it.
|
||||
@@ -76,6 +80,9 @@ func (w *voiceWiring) close() {
|
||||
if w.ttsClient != nil {
|
||||
_ = w.ttsClient.Close()
|
||||
}
|
||||
if w.pair != nil {
|
||||
w.pair.Stop()
|
||||
}
|
||||
w.mcp.close()
|
||||
}
|
||||
|
||||
@@ -139,6 +146,7 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
emb = router.NewHashEmbedder(1024)
|
||||
}
|
||||
w.embedder = emb
|
||||
repairFactVectors(dataStore, emb)
|
||||
checkStoredEmbedder(dataStore, emb)
|
||||
|
||||
// ----- tool executor (the enabled act allowlist, store-backed) -----
|
||||
@@ -188,6 +196,18 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
// the new llama-server when the resident model is swapped (Vikunja #250).
|
||||
llmClient = llmClientFor(lp, 60*time.Second)
|
||||
}
|
||||
// The workstation model sits above that one when it is configured and its
|
||||
// card is free. hot is what the router and the replier complete through:
|
||||
// either the pair, or the resident client alone, or nothing at all.
|
||||
hot, pair := modelSeam(cfg, llmClient)
|
||||
w.pair = pair
|
||||
// The phraser gets the same pair, which is what carries the workstation model
|
||||
// into the paths that do not go through `hot`: world questions (the naming
|
||||
// half), and the digestion worker's nudge and reminder phrasing (the silent
|
||||
// half). Wiring, so it happens once and before the voice server listens.
|
||||
if lp, ok := phr.(*phraser.LLMPhraser); ok && pair != nil {
|
||||
lp.UseRemote(pair)
|
||||
}
|
||||
// ----- router (the cascade; floor examples seed the classifier) -----
|
||||
// The act matcher's allowlist is exactly the enabled tool names — the
|
||||
// router only matches acts the executor can run (one source of truth).
|
||||
@@ -199,7 +219,7 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
// against the classifier's 50.0%, at about 1s a turn instead of 30ms (see
|
||||
// config.VoiceConfig.LLMRouter). The classifier always stays wired as the
|
||||
// fallback, so a model error never breaks a turn.
|
||||
rtr := buildRouter(emb, matcher, threshold, pickLLMRouter(cfg.Voice.UseLLMRouter(), llmClient))
|
||||
rtr := buildRouter(emb, matcher, threshold, pickLLMRouter(cfg.Voice.UseLLMRouter(), hot))
|
||||
|
||||
// ----- sessions registry (shared with voicesink) -----
|
||||
sessions := voice.NewSessions()
|
||||
@@ -218,7 +238,10 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
// ----- dialogue (multi-turn slot carry-over; 2-min follow-up window) -----
|
||||
// Store-backed when the daemon passes a store, so a restart mid-conversation
|
||||
// keeps the thread (Vikunja #363). Sessions past their TTL are dropped on
|
||||
// load, never revived. Clarify's parked question stays in memory only.
|
||||
// load, never revived. Clarify's parked question stays in memory only, and
|
||||
// that is a decision rather than an omission (Vikunja #385, docs/design.md):
|
||||
// a restart expires it, so the thread comes back and the open question does
|
||||
// not.
|
||||
var dialogueSessions *dialogue.SessionStore
|
||||
if dataStore != nil {
|
||||
dialogueSessions = dialogue.NewPersistentSessionStore(2*time.Minute, dataStore)
|
||||
@@ -233,8 +256,8 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
|
||||
// ----- replier (LLM-backed when the engine is on, Stub floor otherwise) -----
|
||||
replier := voice.Replier(voice.NewStubReplier())
|
||||
if llmClient != nil {
|
||||
replier = newLLMReplier(llmClient, contextBlockFn(cfg, time.Now))
|
||||
if hot != nil {
|
||||
replier = newLLMReplier(hot, contextBlockFn(cfg, time.Now))
|
||||
}
|
||||
|
||||
// ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) -----
|
||||
@@ -291,7 +314,42 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
// pickLLMRouter returns the LLM router when the operator asked for it and there
|
||||
// is a llama-server to talk to, and nil otherwise. nil is safe: the cascade then
|
||||
// routes with the classifier, so an unusable setting costs accuracy, not turns.
|
||||
func pickLLMRouter(enabled bool, c *llm.Client) *router.LLMRouter {
|
||||
// modelSeam builds the completion seam the hot paths use: routing and replies.
|
||||
//
|
||||
// With no `workstation` block it is the resident client and nothing probes
|
||||
// anything, which is today's deploy exactly. With one, it is an llm.Pair that
|
||||
// prefers the workstation and falls back to the resident model silently — the
|
||||
// silent half of the degradation rule (docs/offload.md), because the big model
|
||||
// is only better here and the 1.7B is today's shipping quality. He is never
|
||||
// told which of the two phrased his reply.
|
||||
//
|
||||
// A nil resident client means the phraser is not an LLM phraser. There is then
|
||||
// no floor, and a Pair with no floor is a configuration mistake rather than a
|
||||
// degraded mode, so the seam is nil and the cascade routes with the classifier.
|
||||
func modelSeam(cfg *config.Config, resident *llm.Client) (router.Completer, *llm.Pair) {
|
||||
if resident == nil {
|
||||
if cfg.Workstation != nil {
|
||||
log.Printf("voice: a workstation is configured but there is no resident model to floor it with — ignoring the block")
|
||||
}
|
||||
return nil, nil
|
||||
}
|
||||
if cfg.Workstation == nil {
|
||||
return resident, nil
|
||||
}
|
||||
ws := cfg.Workstation
|
||||
pair := llm.NewPair(
|
||||
llm.New(ws.URL, time.Duration(ws.Timeout)),
|
||||
resident,
|
||||
ws.Health,
|
||||
time.Duration(ws.Probe),
|
||||
)
|
||||
pair.Start(context.Background())
|
||||
log.Printf("voice: workstation model at %s, probed every %s, resident model as the floor",
|
||||
ws.URL, time.Duration(ws.Probe))
|
||||
return pair, pair
|
||||
}
|
||||
|
||||
func pickLLMRouter(enabled bool, c router.Completer) *router.LLMRouter {
|
||||
if !enabled {
|
||||
return nil
|
||||
}
|
||||
@@ -324,7 +382,15 @@ func buildRouter(emb router.Embedder, acts router.ActMatcher, threshold float64,
|
||||
// question and must keep reaching replySystem, while "что у меня сегодня"
|
||||
// is an agenda question and must not.
|
||||
grammars = append(grammars, router.AgendaQueryGrammars()...)
|
||||
// Same reason as the agenda rules, for the feeds: "что нового в лентах?"
|
||||
// routed system and answered "пока не умею" (Vikunja #474).
|
||||
grammars = append(grammars, router.FeedQueryGrammar())
|
||||
grammars = append(grammars, router.ReminderGrammar())
|
||||
// Last, and it matches any utterance shape — its Build is the filter. An
|
||||
// explicit capture marker beats the model, which called it an act and
|
||||
// rewrote the task text (Vikunja #467). After the rules above because a
|
||||
// marker never collides with a clock or agenda question.
|
||||
grammars = append(grammars, router.TaskCaptureGrammar())
|
||||
return router.New(router.Config{
|
||||
Grammars: grammars,
|
||||
Classifier: cls,
|
||||
@@ -338,11 +404,34 @@ func buildRouter(emb router.Embedder, acts router.ActMatcher, threshold float64,
|
||||
})
|
||||
}
|
||||
|
||||
// seedDir is the directory containing intent seed files. Each file is named
|
||||
// <intent>.txt and contains one training example per line (blank lines and
|
||||
// lines starting with # are ignored). Relative to the working directory.
|
||||
// seedDir is the directory containing intent seed files, relative to the repo
|
||||
// root. Each file is named <intent>.txt and holds one training example per
|
||||
// line (blank lines and lines starting with # are ignored).
|
||||
const seedDir = "models/seeds"
|
||||
|
||||
// seedPath resolves seedDir against the working directory, walking up until it
|
||||
// finds it. The daemon runs from the repo root and the first candidate hits.
|
||||
//
|
||||
// A test does not: `go test ./cmd/mavend/` runs with the working directory at
|
||||
// cmd/mavend, so every open failed and the simulator scenarios replayed a whole
|
||||
// scripted day against a classifier holding zero examples (Vikunja #465). They
|
||||
// passed, which is the part that matters — a green simulator was not exercising
|
||||
// the routing the deploy runs, and a regression in the seed set could not have
|
||||
// shown up there.
|
||||
//
|
||||
// Bounded at five levels, so a daemon started somewhere without the seeds logs
|
||||
// the same failure it always did rather than walking to the filesystem root.
|
||||
func seedPath() string {
|
||||
dir := seedDir
|
||||
for i := 0; i < 5; i++ {
|
||||
if st, err := os.Stat(dir); err == nil && st.IsDir() {
|
||||
return dir
|
||||
}
|
||||
dir = filepath.Join("..", dir)
|
||||
}
|
||||
return seedDir
|
||||
}
|
||||
|
||||
// seedClassifier floors the embedded examples so the cold-boot path
|
||||
// doesn't return ErrNoIntents. Loads examples from seedDir — one file per
|
||||
// intent (act.txt, reminder.txt, fact.txt, note.txt, query.txt). When the
|
||||
@@ -367,11 +456,11 @@ func seedClassifier(c *router.Classifier) {
|
||||
}
|
||||
total += n
|
||||
}
|
||||
log.Printf("voice: loaded %d seed examples from %s", total, seedDir)
|
||||
log.Printf("voice: loaded %d seed examples from %s", total, seedPath())
|
||||
}
|
||||
|
||||
func loadSeedFile(c *router.Classifier, intent router.Intent) (int, error) {
|
||||
path := filepath.Join(seedDir, string(intent)+".txt")
|
||||
path := filepath.Join(seedPath(), string(intent)+".txt")
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return 0, fmt.Errorf("open %s: %w", path, err)
|
||||
@@ -419,6 +508,36 @@ func seedTools(api ipc.CoreAPI, tools []config.ToolConfig) {
|
||||
log.Printf("voice: seeded %d act tools from config", n)
|
||||
}
|
||||
|
||||
// repairFactVectors brings stored fact vectors in line with the facts they name
|
||||
// (#493), once per box, before the embedder marker is even looked at.
|
||||
//
|
||||
// Automatic and not a flag, unlike -reembed: only voice-tapped facts are in
|
||||
// this index, so the work is tens of embeddings rather than the thousands of
|
||||
// notes that made the backfill a deliberate act. And the box that needs it is
|
||||
// broken in a way nobody can see — recall answers with the wrong text and
|
||||
// nothing logs an error — so waiting for an operator to know to run it is how
|
||||
// the defect survived four restarts in the first place.
|
||||
func repairFactVectors(dataStore *store.Store, emb router.Embedder) {
|
||||
if dataStore == nil {
|
||||
return
|
||||
}
|
||||
res, err := dataStore.RepairFactVectors(context.Background(),
|
||||
// EmbedPassage, the stored side, same as every other writer of these
|
||||
// vectors.
|
||||
func(ctx context.Context, text string) ([]float32, error) {
|
||||
return router.EmbedPassage(ctx, emb, text)
|
||||
})
|
||||
if err != nil {
|
||||
log.Printf("voice: fact vector repair failed, no marker written and nothing half-done — retried next start: %v", err)
|
||||
return
|
||||
}
|
||||
if res.Skipped || res.Rewritten+res.Dropped == 0 {
|
||||
return
|
||||
}
|
||||
log.Printf("voice: fact vector repair — %d re-embedded from the fact they name, %d dropped as voided or superseded, %d already right, took %s (#493)",
|
||||
res.Rewritten, res.Dropped, res.Kept, res.Took.Round(time.Millisecond))
|
||||
}
|
||||
|
||||
// reembedOnStart is the -reembed flag (set in run()). Opt-in on purpose: see
|
||||
// runReembed.
|
||||
var reembedOnStart bool
|
||||
|
||||
+45
-33
@@ -1,10 +1,13 @@
|
||||
// Package main — weatherq.go holds the weather-query keyword helpers: does
|
||||
// this utterance ask about weather at all, and which city (if any) did it
|
||||
// name. Both are plain substring/lookup matching, not NLU — extend this file
|
||||
// rather than voice.go for anything in that shape.
|
||||
// this utterance ask about weather at all, and which place (if any) did he
|
||||
// name. Both are plain keyword matching, not NLU — extend this file rather
|
||||
// than voice.go for anything in that shape.
|
||||
package main
|
||||
|
||||
import "strings"
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// isWeatherQuery returns true if the utterance is about weather.
|
||||
func isWeatherQuery(u string) bool {
|
||||
@@ -19,40 +22,49 @@ func isWeatherQuery(u string) bool {
|
||||
strings.Contains(lower, "temperature")
|
||||
}
|
||||
|
||||
// weatherCities — the city names an utterance may name explicitly, as
|
||||
// lowercase substrings mapped to the provider's spelling. This is a
|
||||
// convenience for "какая погода в Лондоне", NOT a source of default truth:
|
||||
// nothing here is used unless he actually said it.
|
||||
var weatherCities = map[string]string{
|
||||
"москв": "Moscow",
|
||||
"moscow": "Moscow",
|
||||
"питер": "Saint Petersburg",
|
||||
"spb": "Saint Petersburg",
|
||||
"петербур": "Saint Petersburg",
|
||||
"лондон": "London",
|
||||
"london": "London",
|
||||
"париж": "Paris",
|
||||
"paris": "Paris",
|
||||
"берлин": "Berlin",
|
||||
"berlin": "Berlin",
|
||||
"нью-йорк": "New York",
|
||||
"new york": "New York",
|
||||
// weatherPlace — the place he named, after "в"/"во"/"in". One or two words,
|
||||
// letters and dashes only, so "в Нижнем Новгороде" and "in New York" both
|
||||
// come through whole and "в 5 утра" does not.
|
||||
var weatherPlace = regexp.MustCompile(`(?i)(?:^|\s)(?:в|во|in)\s+([\p{L}-]+(?:\s+[\p{L}-]+)?)`)
|
||||
|
||||
// weatherNonPlaces — words that follow "в" in a weather question and are not
|
||||
// cities. "какая погода в доме" is the smart-home sensor, not Open-Meteo, and
|
||||
// "тепло в комнате" is the same question about the same room.
|
||||
var weatherNonPlaces = map[string]bool{
|
||||
"доме": true, "квартире": true, "комнате": true, "спальне": true,
|
||||
"гостиной": true, "кухне": true, "гараже": true, "офисе": true,
|
||||
"выходные": true, "субботу": true, "воскресенье": true, "понедельник": true,
|
||||
"вторник": true, "среду": true, "четверг": true, "пятницу": true,
|
||||
"обед": true, "обеде": true, "утро": true, "утром": true, "вечер": true,
|
||||
"вечером": true, "ночь": true, "ночью": true, "целом": true, "принципе": true,
|
||||
}
|
||||
|
||||
// extractWeatherLocation returns the city he named, or the configured default
|
||||
// extractWeatherLocation returns the place he named, or the configured default
|
||||
// when he named none. It returns "" when he named none AND no default is
|
||||
// configured — the caller must then say it does not know.
|
||||
//
|
||||
// It used to return "Moscow" in that case. That is a made-up answer presented
|
||||
// as fact: reading out Moscow's temperature to someone who is not in Moscow is
|
||||
// wrong in exactly the way maven must never be wrong. voice.weather
|
||||
// .default_location is the only source of an unstated location.
|
||||
// It used to be a hand-written table of six cities in two spellings each
|
||||
// (Vikunja #421). Anything outside it — Kazan, Tbilisi — was dropped silently
|
||||
// and answered for the default location, which reads as a correct answer about
|
||||
// the wrong place. There is a geocoder behind this now: internal/weather
|
||||
// already calls Open-Meteo's geocoding endpoint for every lookup, so any place
|
||||
// it knows is a place he can ask about, and the table bought nothing.
|
||||
//
|
||||
// A named place that the geocoder cannot resolve is the caller's problem to
|
||||
// report, not this function's to hide.
|
||||
//
|
||||
// It used to return "Moscow" when he named nothing. That is a made-up answer
|
||||
// presented as fact. voice.weather.default_location is the only source of an
|
||||
// unstated location.
|
||||
func extractWeatherLocation(u, defaultLoc string) string {
|
||||
lower := strings.ToLower(u)
|
||||
for substr, name := range weatherCities {
|
||||
if strings.Contains(lower, substr) {
|
||||
return name
|
||||
}
|
||||
m := weatherPlace.FindStringSubmatch(u)
|
||||
if m == nil {
|
||||
return defaultLoc
|
||||
}
|
||||
return defaultLoc
|
||||
place := strings.TrimSpace(m[1])
|
||||
first := strings.ToLower(strings.Fields(place)[0])
|
||||
if weatherNonPlaces[first] {
|
||||
return defaultLoc
|
||||
}
|
||||
return place
|
||||
}
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// TestExtractWeatherLocation — any place he names comes through, not just the
|
||||
// six that used to be in a table (Vikunja #421).
|
||||
func TestExtractWeatherLocation(t *testing.T) {
|
||||
cases := []struct {
|
||||
utterance string
|
||||
def string
|
||||
want string
|
||||
}{
|
||||
// The cities the table had, and the ones it silently dropped.
|
||||
{"какая погода в Москве", "Berlin", "Москве"},
|
||||
{"какая погода в Казани", "Berlin", "Казани"},
|
||||
{"погода в Тбилиси?", "Berlin", "Тбилиси"},
|
||||
{"what's the weather in New York", "Berlin", "New York"},
|
||||
{"тепло в Нижнем Новгороде?", "Berlin", "Нижнем Новгороде"},
|
||||
// He named nothing: the configured default, and nothing at all when
|
||||
// there is no default.
|
||||
{"какая сегодня погода", "Berlin", "Berlin"},
|
||||
{"какая сегодня погода", "", ""},
|
||||
// "в" followed by something that is not a place stays the default —
|
||||
// the house sensors and the day words answer elsewhere.
|
||||
{"тепло в комнате?", "Berlin", "Berlin"},
|
||||
{"какая погода в выходные", "Berlin", "Berlin"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := extractWeatherLocation(c.utterance, c.def); got != c.want {
|
||||
t.Errorf("extractWeatherLocation(%q, %q) = %q, want %q", c.utterance, c.def, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestCalendarStepsAsideForWeather — the defect (Vikunja #474). "какая сегодня
|
||||
// погода в Москве?" answered "на 02.08.2026 ничего нет.": the calendar matches
|
||||
// on a day word alone, and it sits above the weather source.
|
||||
func TestCalendarStepsAsideForWeather(t *testing.T) {
|
||||
h, api := contQueryHandler()
|
||||
for _, u := range []string{
|
||||
"какая сегодня погода в Москве?",
|
||||
"будет дождь завтра?",
|
||||
"сколько градусов сегодня?",
|
||||
} {
|
||||
if reply, ok := h.queryCalendar(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Intent: router.IntentQuery, Utterance: u},
|
||||
}); ok {
|
||||
t.Errorf("the calendar claimed %q with %q", u, reply)
|
||||
}
|
||||
}
|
||||
if api.events != 0 {
|
||||
t.Errorf("CalendarEvents called %d times for weather questions, want 0", api.events)
|
||||
}
|
||||
// The agenda question it exists for still reaches it.
|
||||
if _, ok := h.queryCalendar(context.Background(), &queryTurn{
|
||||
dec: router.Decision{Intent: router.IntentQuery, Utterance: "что у меня сегодня?"},
|
||||
}); !ok {
|
||||
t.Fatal("the calendar stopped answering the agenda question")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
)
|
||||
|
||||
// worldPhraser — the naming half of the degradation rule (docs/offload.md), as
|
||||
// the query sources see it. Only *phraser.LLMPhraser implements it, so the
|
||||
// Stub and every test double stay exactly as they are.
|
||||
type worldPhraser interface {
|
||||
PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error)
|
||||
}
|
||||
|
||||
// worldGap — what he hears when the question is about the world, the workstation
|
||||
// model is the one configured to answer it, and that machine is not answering.
|
||||
//
|
||||
// It says the true thing. The resident 1.7B is not a worse answer here, it is an
|
||||
// invented one: "Война и мир" came back with Левитан as its author, and a
|
||||
// question about his meeting came back as a swimming competition in Nottingham.
|
||||
// Naming the gap is the rule CLAUDE.md already applies to a sibling service
|
||||
// being down.
|
||||
const worldGap = "сейчас не могу ответить — большая модель недоступна, а придумывать не хочу."
|
||||
|
||||
// phraseWorld asks the world model, or reports the gap.
|
||||
//
|
||||
// The three outcomes come straight from LLMPhraser.PhraseWorld: no workstation
|
||||
// configured means the resident model answers as it always has, a workstation
|
||||
// that is up answers, and a workstation that is down returns
|
||||
// phraser.ErrNoWorldModel. A phraser that has no world seam at all — the Stub,
|
||||
// and the doubles in the tests — is the first of those three.
|
||||
func (h *reactiveHandler) phraseWorld(ctx context.Context, utterance string, sources []string) (string, error) {
|
||||
if h.phraser == nil {
|
||||
return "", phraser.ErrNoWorldModel
|
||||
}
|
||||
if w, ok := h.phraser.(worldPhraser); ok {
|
||||
return w.PhraseWorld(ctx, utterance, sources)
|
||||
}
|
||||
return h.phraser.PhraseQuery(ctx, utterance, sources)
|
||||
}
|
||||
|
||||
// phraseSource asks the world model to answer from a passage someone already
|
||||
// fetched — a live search result, a ZIM article, a page he named. It returns ""
|
||||
// rather than the gap phrase, because these callers hold something better than a
|
||||
// gap: the passage itself, which their own floor reads back to him. Nothing is
|
||||
// invented either way, and a real quote beats "не могу сейчас".
|
||||
func (h *reactiveHandler) phraseSource(ctx context.Context, name, utterance string, sources []string) string {
|
||||
reply, err := h.phraseWorld(ctx, utterance, sources)
|
||||
switch {
|
||||
case errors.Is(err, phraser.ErrNoWorldModel):
|
||||
log.Printf("voice: %s: no world model, reading the source back instead", name)
|
||||
return ""
|
||||
case err != nil:
|
||||
log.Printf("voice: %s: phrase: %v", name, err)
|
||||
}
|
||||
return reply
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// gapPhraser — a phraser whose world model is configured and asleep, which is
|
||||
// the state the naming half exists for.
|
||||
type gapPhraser struct {
|
||||
*phraser.Stub
|
||||
worldCalls int
|
||||
}
|
||||
|
||||
func (g *gapPhraser) PhraseWorld(context.Context, string, []string) (string, error) {
|
||||
g.worldCalls++
|
||||
return "", phraser.ErrNoWorldModel
|
||||
}
|
||||
|
||||
func worldTurn(utterance string) *queryTurn {
|
||||
return &queryTurn{dec: router.Decision{Intent: router.IntentQuery, Utterance: utterance}}
|
||||
}
|
||||
|
||||
// A world question with the workstation asleep says so. The resident model is
|
||||
// not asked, because what it produces here is an invention with no signal that
|
||||
// it is one.
|
||||
func TestQueryGeneralNamesTheGap(t *testing.T) {
|
||||
g := &gapPhraser{Stub: phraser.NewStub()}
|
||||
h := &reactiveHandler{phraser: g}
|
||||
reply, ok := h.queryGeneral(context.Background(), worldTurn("почему небо голубое"))
|
||||
if !ok {
|
||||
t.Fatal("queryGeneral passed on the last source in the chain")
|
||||
}
|
||||
if reply != worldGap {
|
||||
t.Fatalf("reply = %q, want the named gap", reply)
|
||||
}
|
||||
if g.worldCalls != 1 {
|
||||
t.Fatalf("PhraseWorld called %d times, want 1", g.worldCalls)
|
||||
}
|
||||
}
|
||||
|
||||
// A phraser with no world seam at all — the Stub, and every box with no
|
||||
// `workstation` block — answers exactly as it did before this seam existed.
|
||||
func TestQueryGeneralWithoutAWorldModelIsUnchanged(t *testing.T) {
|
||||
h := &reactiveHandler{phraser: phraser.NewStub()}
|
||||
reply, ok := h.queryGeneral(context.Background(), worldTurn("почему небо голубое"))
|
||||
if !ok {
|
||||
t.Fatal("queryGeneral passed on the last source in the chain")
|
||||
}
|
||||
if reply != "не знаю." {
|
||||
t.Fatalf("reply = %q, want the Stub's answer", reply)
|
||||
}
|
||||
}
|
||||
|
||||
// The gap is spoken aloud by a Russian voice, so it is Russian, feminine and
|
||||
// informal. "не хочу" and "не могу" are her own verbs; there is no "вы" and no
|
||||
// English in it.
|
||||
func TestWorldGapIsInPersona(t *testing.T) {
|
||||
for _, bad := range []string{"вы", "ваш", "рад ", "дорогой", "милый"} {
|
||||
if strings.Contains(worldGap, bad) {
|
||||
t.Errorf("the gap phrase contains %q: %s", bad, worldGap)
|
||||
}
|
||||
}
|
||||
if strings.ContainsAny(worldGap, "abcdefghijklmnopqrstuvwxyz") {
|
||||
t.Errorf("the gap phrase has Latin letters in it: %s", worldGap)
|
||||
}
|
||||
}
|
||||
|
||||
// The sources that hold a passage read it back rather than name a gap. He gets a
|
||||
// real quote instead of "не могу сейчас", and nothing is invented either way.
|
||||
func TestASourceWithAPassageReadsItBackInsteadOfNamingTheGap(t *testing.T) {
|
||||
g := &gapPhraser{Stub: phraser.NewStub()}
|
||||
h := &reactiveHandler{phraser: g}
|
||||
if got := h.phraseSource(context.Background(), "search", "почему небо голубое",
|
||||
[]string{"Рэлеевское рассеяние."}); got != "" {
|
||||
t.Fatalf("phraseSource = %q, want \"\" so the caller's own floor reads the passage back", got)
|
||||
}
|
||||
if g.worldCalls != 1 {
|
||||
t.Fatalf("PhraseWorld called %d times, want 1", g.worldCalls)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// The card is an AMD 7900 GRE with 16GB, driven by amdgpu and ROCm. Everything
|
||||
// here reads sysfs and forks nothing: rocm-smi is not even installed on the
|
||||
// workstation, and a poll that costs a subprocess every second is a poll that
|
||||
// gets tuned down until it is useless.
|
||||
|
||||
// gpuProc — one process holding the compute engine.
|
||||
type gpuProc struct {
|
||||
PID int
|
||||
Comm string
|
||||
VRAM int64 // bytes, as the kernel accounts them to this process
|
||||
}
|
||||
|
||||
// probe reads the two sysfs trees the supervisor decides from.
|
||||
//
|
||||
// kfdRoot is /sys/class/kfd/kfd/proc, one directory per ROCm process. The
|
||||
// directory appears when the process initialises HIP, which is well before it
|
||||
// allocates anything large. That is the whole reason this works: the job that
|
||||
// is about to want the card announces itself while it is still starting up,
|
||||
// so we see the contender rather than only the winner of an allocation race.
|
||||
//
|
||||
// drmDev is /sys/class/drm/cardN/device, which reports total and used VRAM for
|
||||
// the card as a whole.
|
||||
type probe struct {
|
||||
kfdRoot string
|
||||
drmDev string
|
||||
}
|
||||
|
||||
// foreign lists every ROCm process that is not ours. selfPID is the supervisor's
|
||||
// llama-server child, or 0 when it is not running.
|
||||
//
|
||||
// An unreadable kfd tree returns no processes and no error. That is deliberate
|
||||
// and it is the safe direction only because startVRAM also has to agree before
|
||||
// anything launches: a supervisor that cannot see the KFD never sees free VRAM
|
||||
// either, because the CPT run holding the card shows up in the drm totals.
|
||||
func (p probe) foreign(selfPID int) []gpuProc {
|
||||
entries, err := os.ReadDir(p.kfdRoot)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var out []gpuProc
|
||||
for _, e := range entries {
|
||||
pid, err := strconv.Atoi(e.Name())
|
||||
if err != nil || pid == selfPID {
|
||||
continue
|
||||
}
|
||||
out = append(out, gpuProc{
|
||||
PID: pid,
|
||||
Comm: readComm(pid),
|
||||
VRAM: p.procVRAM(e.Name()),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// procVRAM sums the per-node vram_* files under one process directory. The
|
||||
// suffix is the KFD topology node id (vram_35881 on this card), so it is
|
||||
// globbed rather than named, and a machine with two cards sums both.
|
||||
func (p probe) procVRAM(pid string) int64 {
|
||||
matches, err := filepath.Glob(filepath.Join(p.kfdRoot, pid, "vram_*"))
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
var total int64
|
||||
for _, m := range matches {
|
||||
total += readInt(m)
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// freeVRAM reports the bytes the card has left. Used only to decide whether to
|
||||
// start: a shortfall here means llama-server would refuse to load anyway. It is
|
||||
// never used to decide to stop, because by the time free VRAM has dropped the
|
||||
// other job has already failed its allocation, which is exactly the outcome
|
||||
// yielding exists to prevent.
|
||||
func (p probe) freeVRAM() int64 {
|
||||
total := readInt(filepath.Join(p.drmDev, "mem_info_vram_total"))
|
||||
used := readInt(filepath.Join(p.drmDev, "mem_info_vram_used"))
|
||||
if total <= 0 {
|
||||
return 0
|
||||
}
|
||||
if free := total - used; free > 0 {
|
||||
return free
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
func readInt(path string) int64 {
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
n, err := strconv.ParseInt(strings.TrimSpace(string(b)), 10, 64)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// readComm names the contender for the log. The log is the instrument for the
|
||||
// open question in Vikunja #488: whether a process can want this card without
|
||||
// ever registering on the KFD, which a Vulkan or video-decode job would.
|
||||
func readComm(pid int) string {
|
||||
b, err := os.ReadFile(filepath.Join("/proc", strconv.Itoa(pid), "comm"))
|
||||
if err != nil {
|
||||
return "?"
|
||||
}
|
||||
return strings.TrimSpace(string(b))
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// fakeKFD builds the sysfs shape the workstation actually has: one directory
|
||||
// per ROCm process, each holding a vram_<node> file. Sampled from the live box
|
||||
// on 02-08-2026, where the CPT run appeared as proc/478104/vram_35881.
|
||||
func fakeKFD(t *testing.T, vramByPID map[int]int64) string {
|
||||
t.Helper()
|
||||
root := t.TempDir()
|
||||
for pid, vram := range vramByPID {
|
||||
dir := filepath.Join(root, strconv.Itoa(pid))
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f := filepath.Join(dir, "vram_35881")
|
||||
if err := os.WriteFile(f, []byte(strconv.FormatInt(vram, 10)+"\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
return root
|
||||
}
|
||||
|
||||
func TestForeignExcludesOurChild(t *testing.T) {
|
||||
root := fakeKFD(t, map[int]int64{478104: 12791693312, 999: 4096})
|
||||
p := probe{kfdRoot: root}
|
||||
|
||||
all := p.foreign(0)
|
||||
if len(all) != 2 {
|
||||
t.Fatalf("with no child running, both processes are foreign, got %d", len(all))
|
||||
}
|
||||
|
||||
ours := p.foreign(999)
|
||||
if len(ours) != 1 || ours[0].PID != 478104 {
|
||||
t.Fatalf("our own llama-server must not count as a contender, got %+v", ours)
|
||||
}
|
||||
if ours[0].VRAM != 12791693312 {
|
||||
t.Errorf("per-process VRAM = %d, want the value from vram_35881", ours[0].VRAM)
|
||||
}
|
||||
}
|
||||
|
||||
// An empty KFD tree is the state that permits a start, so it must read as empty
|
||||
// rather than as an error the caller has to interpret.
|
||||
func TestForeignEmptyAndMissing(t *testing.T) {
|
||||
if got := (probe{kfdRoot: t.TempDir()}).foreign(0); len(got) != 0 {
|
||||
t.Errorf("empty kfd tree: got %d processes, want 0", len(got))
|
||||
}
|
||||
if got := (probe{kfdRoot: "/nonexistent"}).foreign(0); got != nil {
|
||||
t.Errorf("missing kfd tree: got %+v, want nil", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFreeVRAM(t *testing.T) {
|
||||
dev := t.TempDir()
|
||||
write := func(name, v string) {
|
||||
if err := os.WriteFile(filepath.Join(dev, name), []byte(v), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
// The live numbers from the workstation while the CPT run held the card.
|
||||
write("mem_info_vram_total", "17163091968\n")
|
||||
write("mem_info_vram_used", "13396389888\n")
|
||||
p := probe{drmDev: dev}
|
||||
if got, want := p.freeVRAM(), int64(3766702080); got != want {
|
||||
t.Errorf("freeVRAM = %d, want %d", got, want)
|
||||
}
|
||||
if got := (probe{drmDev: "/nonexistent"}).freeVRAM(); got != 0 {
|
||||
t.Errorf("unreadable card reports %d free, want 0 so nothing starts", got)
|
||||
}
|
||||
}
|
||||
|
||||
// With no model loaded the supervisor must still answer, and it must answer 503
|
||||
// rather than hanging or proxying into a closed port. Maven reads this endpoint
|
||||
// on a timer forever, including while the workstation is busy.
|
||||
func TestHealthAndProxyRefuseWhenNotReady(t *testing.T) {
|
||||
s := &supervisor{run: newRunner("/bin/true", nil, "")}
|
||||
h := s.handler(mustURL(t, "http://127.0.0.1:1"))
|
||||
|
||||
for _, path := range []string{"/health", "/v1/chat/completions"} {
|
||||
w := httptest.NewRecorder()
|
||||
h.ServeHTTP(w, httptest.NewRequest(http.MethodGet, path, nil))
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Errorf("%s with no model: got %d, want 503", path, w.Code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func mustURL(t *testing.T, s string) *url.URL {
|
||||
t.Helper()
|
||||
u, err := url.Parse(s)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return u
|
||||
}
|
||||
@@ -0,0 +1,247 @@
|
||||
// mavgpud — the workstation's GPU supervisor.
|
||||
//
|
||||
// It runs on the workstation (an AMD 7900 GRE, 16GB), not on homesrv, and it is
|
||||
// deployed separately from the Maven daemons. Maven does not participate in any
|
||||
// of this and never asks for a start: it reads /health through internal/llm.Pair
|
||||
// and either gets the big model or falls back to the resident 1.7B.
|
||||
//
|
||||
// The rule, from Vikunja #488: keep llama-server loaded whenever the card is
|
||||
// free, unload it when it has been idle too long or when another process needs
|
||||
// the card. Not on demand, because a 7-14B takes tens of seconds to load and a
|
||||
// world question would be answered by a gap every time the card had been quiet.
|
||||
// Not always on, because that holds 16GB against the owner's own jobs.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"flag"
|
||||
"log"
|
||||
"net/http"
|
||||
"net/http/httputil"
|
||||
"net/url"
|
||||
"os"
|
||||
"os/signal"
|
||||
"sync/atomic"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
type config struct {
|
||||
Listen string `json:"listen"` // what Maven talks to
|
||||
LlamaAddr string `json:"llama_addr"` // where llama-server binds
|
||||
LlamaBin string `json:"llama_bin"`
|
||||
// LlamaArgs must include the flags that bind LlamaAddr. They are passed
|
||||
// through untouched so the model, context size and layer count stay the
|
||||
// owner's business and not this daemon's schema.
|
||||
LlamaArgs []string `json:"llama_args"`
|
||||
|
||||
KFDRoot string `json:"kfd_root"`
|
||||
DRMDevice string `json:"drm_device"`
|
||||
|
||||
Poll duration `json:"poll"`
|
||||
IdleTimeout duration `json:"idle_timeout"`
|
||||
StopGrace duration `json:"stop_grace"`
|
||||
MinFreeVRAM int64 `json:"min_free_vram_bytes"`
|
||||
// EvictAfter and StartAfter are counted in polls, not seconds. Both exist
|
||||
// to damp flapping: a one-tick blip from a short-lived rocm process must
|
||||
// not evict the model, and a card that has just been released must not be
|
||||
// grabbed before the previous job has finished unmapping.
|
||||
EvictAfter int `json:"evict_after_polls"`
|
||||
StartAfter int `json:"start_after_polls"`
|
||||
}
|
||||
|
||||
func defaults() config {
|
||||
return config{
|
||||
Listen: ":8080",
|
||||
LlamaAddr: "127.0.0.1:8081",
|
||||
KFDRoot: "/sys/class/kfd/kfd/proc",
|
||||
DRMDevice: "/sys/class/drm/card1/device",
|
||||
Poll: duration(time.Second),
|
||||
IdleTimeout: duration(15 * time.Minute),
|
||||
StopGrace: duration(20 * time.Second),
|
||||
MinFreeVRAM: 15 << 30,
|
||||
EvictAfter: 2,
|
||||
StartAfter: 5,
|
||||
}
|
||||
}
|
||||
|
||||
// duration lets the config file say "15m" instead of counting nanoseconds.
|
||||
type duration time.Duration
|
||||
|
||||
func (d *duration) UnmarshalJSON(b []byte) error {
|
||||
var s string
|
||||
if err := json.Unmarshal(b, &s); err != nil {
|
||||
return err
|
||||
}
|
||||
v, err := time.ParseDuration(s)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
*d = duration(v)
|
||||
return nil
|
||||
}
|
||||
|
||||
func main() {
|
||||
path := flag.String("config", "/etc/mavgpud.json", "config file")
|
||||
flag.Parse()
|
||||
|
||||
cfg := defaults()
|
||||
b, err := os.ReadFile(*path)
|
||||
if err != nil {
|
||||
log.Fatalf("mavgpud: read config: %v", err)
|
||||
}
|
||||
if err := json.Unmarshal(b, &cfg); err != nil {
|
||||
log.Fatalf("mavgpud: parse config: %v", err)
|
||||
}
|
||||
if cfg.LlamaBin == "" {
|
||||
log.Fatal("mavgpud: llama_bin is required")
|
||||
}
|
||||
|
||||
base := "http://" + cfg.LlamaAddr
|
||||
run := newRunner(cfg.LlamaBin, cfg.LlamaArgs, base+"/health")
|
||||
sup := &supervisor{
|
||||
cfg: cfg,
|
||||
probe: probe{kfdRoot: cfg.KFDRoot, drmDev: cfg.DRMDevice},
|
||||
run: run,
|
||||
}
|
||||
sup.touch()
|
||||
|
||||
ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
|
||||
defer cancel()
|
||||
|
||||
target, err := url.Parse(base)
|
||||
if err != nil {
|
||||
log.Fatalf("mavgpud: llama_addr: %v", err)
|
||||
}
|
||||
srv := &http.Server{Addr: cfg.Listen, Handler: sup.handler(target)}
|
||||
go func() {
|
||||
log.Printf("mavgpud: listening on %s, model %s", cfg.Listen, cfg.LlamaBin)
|
||||
if err := srv.ListenAndServe(); err != nil && err != http.ErrServerClosed {
|
||||
log.Fatalf("mavgpud: listen: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
sup.loop(ctx)
|
||||
|
||||
// The card must come back before we do. A supervisor that exits leaving
|
||||
// llama-server holding 14GB is worse than one that never ran.
|
||||
shut, done := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
defer done()
|
||||
_ = srv.Shutdown(shut)
|
||||
run.stop(time.Duration(cfg.StopGrace))
|
||||
}
|
||||
|
||||
type supervisor struct {
|
||||
cfg config
|
||||
probe probe
|
||||
run *runner
|
||||
|
||||
lastReq atomic.Int64 // unix nanos of the last request Maven sent
|
||||
|
||||
foreignStreak int
|
||||
clearStreak int
|
||||
}
|
||||
|
||||
func (s *supervisor) touch() { s.lastReq.Store(time.Now().UnixNano()) }
|
||||
|
||||
func (s *supervisor) idle() time.Duration {
|
||||
return time.Since(time.Unix(0, s.lastReq.Load()))
|
||||
}
|
||||
|
||||
// handler serves the two things the workstation exposes.
|
||||
//
|
||||
// /health is answered locally and always, with no GPU cost and no round trip,
|
||||
// because it is the only thing Maven reads and Maven reads it on a timer
|
||||
// forever. Everything else is llama-server's API, reverse-proxied. Proxying
|
||||
// rather than pointing Maven straight at llama-server is what makes the idle
|
||||
// window measurable: the supervisor cannot otherwise know when the model was
|
||||
// last used.
|
||||
func (s *supervisor) handler(target *url.URL) http.Handler {
|
||||
proxy := httputil.NewSingleHostReverseProxy(target)
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/health", func(w http.ResponseWriter, r *http.Request) {
|
||||
if !s.run.isReady() {
|
||||
http.Error(w, "model not loaded", http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"status":"ok"}`))
|
||||
})
|
||||
mux.HandleFunc("/", func(w http.ResponseWriter, r *http.Request) {
|
||||
if !s.run.isReady() {
|
||||
http.Error(w, "model not loaded", http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
s.touch()
|
||||
proxy.ServeHTTP(w, r)
|
||||
})
|
||||
return mux
|
||||
}
|
||||
|
||||
func (s *supervisor) loop(ctx context.Context) {
|
||||
t := time.NewTicker(time.Duration(s.cfg.Poll))
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
s.tick(ctx)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// tick is the whole decision. Yielding is checked before starting, and presence
|
||||
// on the KFD is what triggers it — not a VRAM threshold. A ROCm process
|
||||
// registers under /sys/class/kfd/kfd/proc when it initialises HIP, before it
|
||||
// allocates, so we see a contender during its startup rather than after it has
|
||||
// already failed to get the memory it wanted.
|
||||
func (s *supervisor) tick(ctx context.Context) {
|
||||
others := s.probe.foreign(s.run.pid())
|
||||
if len(others) > 0 {
|
||||
s.foreignStreak++
|
||||
s.clearStreak = 0
|
||||
} else {
|
||||
s.foreignStreak = 0
|
||||
s.clearStreak++
|
||||
}
|
||||
|
||||
if s.run.running() {
|
||||
s.run.refreshReady(ctx)
|
||||
switch {
|
||||
case s.foreignStreak >= s.cfg.EvictAfter:
|
||||
log.Printf("mavgpud: yielding the card to %s", describe(others))
|
||||
s.run.stop(time.Duration(s.cfg.StopGrace))
|
||||
case s.idle() > time.Duration(s.cfg.IdleTimeout):
|
||||
log.Printf("mavgpud: idle for %s, unloading", s.idle().Round(time.Second))
|
||||
s.run.stop(time.Duration(s.cfg.StopGrace))
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
if s.clearStreak < s.cfg.StartAfter {
|
||||
return
|
||||
}
|
||||
if free := s.probe.freeVRAM(); free < s.cfg.MinFreeVRAM {
|
||||
return
|
||||
}
|
||||
s.touch() // the idle clock starts at load, not at the last request before it
|
||||
if err := s.run.start(); err != nil {
|
||||
log.Printf("mavgpud: start llama-server: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// describe names the contenders in the log. This log is the instrument for the
|
||||
// open question in #488: whether polling the KFD misses a job that wants the
|
||||
// card without registering there.
|
||||
func describe(procs []gpuProc) string {
|
||||
out := ""
|
||||
for i, p := range procs {
|
||||
if i > 0 {
|
||||
out += ", "
|
||||
}
|
||||
out += p.Comm
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"net/http"
|
||||
"os/exec"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// runner owns one llama-server process. Owning it is the point of the daemon:
|
||||
// the workstation cannot keep a 7-14B resident, because that holds 16GB against
|
||||
// the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that
|
||||
// stays up is this, which costs no VRAM, and the model comes and goes under it.
|
||||
type runner struct {
|
||||
bin string
|
||||
args []string
|
||||
// ready is llama-server's own /health, which answers "is a model loaded".
|
||||
// Loading a 7-14B takes tens of seconds, so started is not ready.
|
||||
readyURL string
|
||||
|
||||
mu sync.Mutex
|
||||
cmd *exec.Cmd
|
||||
ready bool
|
||||
// yielding — stop() has sent the signal and the exit that follows is ours.
|
||||
// llama-server aborts on SIGTERM (its static teardown throws, upstream
|
||||
// ggml-org/llama.cpp), so a routine yield and a real crash produce the same
|
||||
// "signal: aborted" and used to log identically (Vikunja #491).
|
||||
yielding bool
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
func newRunner(bin string, args []string, readyURL string) *runner {
|
||||
return &runner{
|
||||
bin: bin, args: args, readyURL: readyURL,
|
||||
http: &http.Client{Timeout: 2 * time.Second},
|
||||
}
|
||||
}
|
||||
|
||||
// pid is the child's, or 0. The GPU probe needs it to tell our own model apart
|
||||
// from a contender.
|
||||
func (r *runner) pid() int {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
if r.cmd == nil || r.cmd.Process == nil {
|
||||
return 0
|
||||
}
|
||||
return r.cmd.Process.Pid
|
||||
}
|
||||
|
||||
func (r *runner) running() bool { return r.pid() != 0 }
|
||||
|
||||
// isReady reports the cached readiness. The supervisor loop refreshes it; the
|
||||
// health handler only reads, so answering /health never costs a round trip.
|
||||
func (r *runner) isReady() bool {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
return r.ready
|
||||
}
|
||||
|
||||
// start launches llama-server. It returns as soon as the process exists, not
|
||||
// when the model is loaded.
|
||||
func (r *runner) start() error {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
if r.cmd != nil {
|
||||
return nil
|
||||
}
|
||||
cmd := exec.Command(r.bin, r.args...)
|
||||
// Own process group, so stop kills anything llama-server spawned rather
|
||||
// than leaving it holding VRAM after we have declared the card yielded.
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
|
||||
if err := cmd.Start(); err != nil {
|
||||
return err
|
||||
}
|
||||
r.cmd, r.ready, r.yielding = cmd, false, false
|
||||
log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid)
|
||||
go func() {
|
||||
err := cmd.Wait()
|
||||
r.mu.Lock()
|
||||
yielded := r.yielding
|
||||
r.cmd, r.ready, r.yielding = nil, false, false
|
||||
r.mu.Unlock()
|
||||
if yielded {
|
||||
log.Printf("mavgpud: llama-server stopped, card yielded (%v)", err)
|
||||
return
|
||||
}
|
||||
log.Printf("mavgpud: llama-server exited: %v", err)
|
||||
}()
|
||||
return nil
|
||||
}
|
||||
|
||||
// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so
|
||||
// it unmaps cleanly, SIGKILL after the grace window. Returning before the
|
||||
// process is gone would let the supervisor report a free card while 14GB is
|
||||
// still mapped, which is the one lie that would make yielding useless.
|
||||
func (r *runner) stop(grace time.Duration) {
|
||||
r.mu.Lock()
|
||||
cmd := r.cmd
|
||||
r.ready = false
|
||||
if cmd != nil && cmd.Process != nil {
|
||||
// The exit that follows is ours, not a crash.
|
||||
r.yielding = true
|
||||
}
|
||||
r.mu.Unlock()
|
||||
if cmd == nil || cmd.Process == nil {
|
||||
return
|
||||
}
|
||||
pgid := -cmd.Process.Pid
|
||||
_ = syscall.Kill(pgid, syscall.SIGTERM)
|
||||
deadline := time.Now().Add(grace)
|
||||
for time.Now().Before(deadline) {
|
||||
if !r.running() {
|
||||
return
|
||||
}
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
}
|
||||
log.Printf("mavgpud: llama-server did not exit in %s, killing", grace)
|
||||
_ = syscall.Kill(pgid, syscall.SIGKILL)
|
||||
}
|
||||
|
||||
// refreshReady asks llama-server whether the model is loaded. Called once per
|
||||
// supervisor tick, never per request.
|
||||
func (r *runner) refreshReady(ctx context.Context) {
|
||||
if !r.running() {
|
||||
return
|
||||
}
|
||||
ok := false
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, r.readyURL, nil)
|
||||
if err == nil {
|
||||
resp, err := r.http.Do(req)
|
||||
if err == nil {
|
||||
ok = resp.StatusCode == http.StatusOK
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
r.mu.Lock()
|
||||
was := r.ready
|
||||
r.ready = ok
|
||||
r.mu.Unlock()
|
||||
if ok && !was {
|
||||
log.Printf("mavgpud: model ready")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// fakeServer writes an executable standing in for llama-server: it ignores
|
||||
// SIGTERM the way the real one effectively does — by dying messily rather than
|
||||
// cleanly — and reports a non-zero status.
|
||||
func fakeServer(t *testing.T, body string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), "fake-llama-server")
|
||||
if err := os.WriteFile(path, []byte("#!/bin/sh\n"+body+"\n"), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
|
||||
// A deliberate stop is a yield, and the log has to say so.
|
||||
//
|
||||
// llama-server aborts inside its own static teardown on SIGTERM, so the exit
|
||||
// status of a routine yield is identical to that of a real crash. Reading the
|
||||
// mavgpud log, the two were indistinguishable (Vikunja #491).
|
||||
func TestStopMarksTheExitAsAYield(t *testing.T) {
|
||||
r := newRunner(fakeServer(t, "while : ; do sleep 1 ; done"), nil, "")
|
||||
if err := r.start(); err != nil {
|
||||
t.Fatalf("start: %v", err)
|
||||
}
|
||||
r.mu.Lock()
|
||||
if r.yielding {
|
||||
t.Error("a freshly started server is already marked as yielding")
|
||||
}
|
||||
r.mu.Unlock()
|
||||
|
||||
r.stop(2 * time.Second)
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if !r.running() {
|
||||
return
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("the child outlived stop")
|
||||
}
|
||||
|
||||
// Stopping when nothing is running must not arm the flag for the next child.
|
||||
// The next exit after that would be a real crash logged as a yield.
|
||||
func TestStopWithNoChildDoesNotArmTheFlag(t *testing.T) {
|
||||
r := newRunner("/nonexistent", nil, "")
|
||||
r.stop(10 * time.Millisecond)
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
if r.yielding {
|
||||
t.Error("stop armed the yield flag with no child running")
|
||||
}
|
||||
}
|
||||
@@ -54,7 +54,9 @@ type fakeCore struct {
|
||||
revertErr error
|
||||
|
||||
// for handleNotifications tests
|
||||
nudgesErr error
|
||||
nudgesErr error
|
||||
attempts []ipc.DeliveryAttempt
|
||||
attemptStatus string
|
||||
|
||||
// for handleHistory tests
|
||||
historyFacts []ipc.Fact
|
||||
@@ -77,7 +79,7 @@ func (f *fakeCore) MCPServers(context.Context) ([]ipc.MCPServerStatus, error) {
|
||||
return f.mcpServers, f.mcpErr
|
||||
}
|
||||
|
||||
func (f *fakeCore) Chat(_ context.Context, text string) (string, error) {
|
||||
func (f *fakeCore) Chat(_ context.Context, _, text string) (string, error) {
|
||||
f.chatText = text
|
||||
if f.chatErr != nil {
|
||||
return "", f.chatErr
|
||||
@@ -1251,3 +1253,35 @@ func TestHandleWS_AssertedSession_PassesGate(t *testing.T) {
|
||||
t.Fatalf("status = 403 on an asserted session; body=%s", rr.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeCore) DeliveryAttempts(_ context.Context, status string, _ int) ([]ipc.DeliveryAttempt, error) {
|
||||
f.attemptStatus = status
|
||||
return f.attempts, nil
|
||||
}
|
||||
|
||||
// TestHandleNotifications_ShowsTheOutbox — the outbox was written and never
|
||||
// read, so a dropped or failed send was invisible (Vikunja #390).
|
||||
func TestHandleNotifications_ShowsTheOutbox(t *testing.T) {
|
||||
done := time.Date(2026, 8, 4, 9, 0, 30, 0, time.UTC)
|
||||
core := &fakeCore{
|
||||
attempts: []ipc.DeliveryAttempt{
|
||||
{Kind: "nudge", Rule: "care-check", Channel: "telegram", Status: "dropped",
|
||||
Created: done.Add(-30 * time.Second), Completed: &done},
|
||||
{Kind: "reminder", ReminderID: 7, Channel: "voice", Status: "pending", Created: done},
|
||||
},
|
||||
}
|
||||
rr := httptest.NewRecorder()
|
||||
handleNotifications(rr, httptest.NewRequest(http.MethodGet, "/notifications?status=dropped", nil), core)
|
||||
if rr.Code != http.StatusOK {
|
||||
t.Fatalf("status = %d, want 200; body=%s", rr.Code, rr.Body.String())
|
||||
}
|
||||
if core.attemptStatus != "dropped" {
|
||||
t.Errorf("status filter = %q, want it passed through", core.attemptStatus)
|
||||
}
|
||||
body := rr.Body.String()
|
||||
for _, want := range []string{"care-check", "dropped", "reminder #7", "Delivery outbox"} {
|
||||
if !strings.Contains(body, want) {
|
||||
t.Errorf("rendered outbox missing %q", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+101
-3
@@ -893,12 +893,104 @@ func handleNotifications(w http.ResponseWriter, r *http.Request, core ipc.CoreAP
|
||||
http.Error(w, "notifications error: "+err.Error(), http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
// The outbox, on the page that already answers "what did she send".
|
||||
// A failed or dropped attempt is why she went quiet, and until now it was
|
||||
// recorded and unreadable (Vikunja #390). Filter with ?status=dropped.
|
||||
status := r.URL.Query().Get("status")
|
||||
attempts, err := core.DeliveryAttempts(ctx, status, 50)
|
||||
if err != nil {
|
||||
// The nudge list is still worth showing, so this is a note on the page
|
||||
// rather than a dead page.
|
||||
log.Printf("notifications: delivery attempts: %v", err)
|
||||
}
|
||||
w.Header().Set("Content-Type", "text/html; charset=utf-8")
|
||||
if err := notificationsTmpl.Execute(w, map[string]any{"Nudges": nudges}); err != nil {
|
||||
if err := notificationsTmpl.Execute(w, map[string]any{
|
||||
"Nudges": nudges,
|
||||
"Attempts": deliveryRows(attempts),
|
||||
"Status": status,
|
||||
}); err != nil {
|
||||
log.Printf("notifications template: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// deliveryRow is one outbox line, with every timestamp already formatted so
|
||||
// the template holds no date logic — same shape as taskRow.
|
||||
type deliveryRow struct {
|
||||
Kind string
|
||||
Target string
|
||||
Channel string
|
||||
Status string
|
||||
Created string
|
||||
Completed string
|
||||
}
|
||||
|
||||
func deliveryRows(as []ipc.DeliveryAttempt) []deliveryRow {
|
||||
out := make([]deliveryRow, 0, len(as))
|
||||
for _, a := range as {
|
||||
target := a.Rule
|
||||
if target == "" && a.ReminderID != 0 {
|
||||
target = "reminder #" + strconv.FormatInt(a.ReminderID, 10)
|
||||
}
|
||||
row := deliveryRow{
|
||||
Kind: a.Kind,
|
||||
Target: target,
|
||||
Channel: a.Channel,
|
||||
Status: a.Status,
|
||||
Created: a.Created.Format("02.01 15:04"),
|
||||
}
|
||||
if a.Completed != nil {
|
||||
row.Completed = a.Completed.Format("15:04")
|
||||
}
|
||||
out = append(out, row)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// reminderRow is one line on /reminders, with the payload unwrapped and both
|
||||
// timestamps already in his clock.
|
||||
//
|
||||
// The page rendered `{{.Payload}}` and the UTC instant, so a reminder read
|
||||
// `{"text":"выпить таблетки"}` and fired an hour off what he was told
|
||||
// (Vikunja #469). Neither is a formatting nicety: the envelope is an internal
|
||||
// shape he never chose, and a time on a page he reads is the time on his wall.
|
||||
type reminderRow struct {
|
||||
Created string
|
||||
Fires string
|
||||
Status string
|
||||
Text string
|
||||
}
|
||||
|
||||
// reminderText unwraps the {"text":...} payload the router writes.
|
||||
//
|
||||
// A copy of store.ReminderText rather than a call to it, because mavweb is one
|
||||
// of the pure-Go daemons and internal/store carries the CGO sqlite driver. The
|
||||
// ipc DTO is decoupled from the store on purpose, so the unwrap belongs to
|
||||
// whoever renders it. Payload that is not that shape is shown as he said it.
|
||||
func reminderText(payload string) string {
|
||||
var m map[string]any
|
||||
if err := json.Unmarshal([]byte(payload), &m); err == nil {
|
||||
if t, ok := m["text"]; ok {
|
||||
if s, isStr := t.(string); isStr && s != "" {
|
||||
return s
|
||||
}
|
||||
}
|
||||
}
|
||||
return strings.TrimSpace(payload)
|
||||
}
|
||||
|
||||
func reminderRows(rs []ipc.Reminder) []reminderRow {
|
||||
out := make([]reminderRow, 0, len(rs))
|
||||
for _, r := range rs {
|
||||
out = append(out, reminderRow{
|
||||
Created: r.CreatedTs.Local().Format("02 Jan 15:04"),
|
||||
Fires: r.FireTs.Local().Format("02 Jan 15:04"),
|
||||
Status: r.Status,
|
||||
Text: reminderText(r.Payload),
|
||||
})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func handleReminders(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
||||
if core == nil {
|
||||
http.Error(w, "reminders disabled (no -core)", http.StatusServiceUnavailable)
|
||||
@@ -912,7 +1004,7 @@ func handleReminders(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "text/html; charset=utf-8")
|
||||
if err := remindersTmpl.Execute(w, map[string]any{"Reminders": reminders}); err != nil {
|
||||
if err := remindersTmpl.Execute(w, map[string]any{"Reminders": reminderRows(reminders)}); err != nil {
|
||||
log.Printf("reminders template: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -1678,7 +1770,13 @@ func handleChatAPI(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI, ses
|
||||
http.Redirect(w, r, "/chat", http.StatusSeeOther)
|
||||
return
|
||||
}
|
||||
reply, err := core.Chat(r.Context(), text)
|
||||
// One conversation id for the whole web chat, and a different one from
|
||||
// telegram or the mic. A parked question belongs to the reach that was
|
||||
// asked; before this, a clarify nobody answered on the web ate the next
|
||||
// utterance spoken at the mic (Vikunja #466). This server has no
|
||||
// per-browser session, so every browser tab is the same conversation —
|
||||
// which is right for a single-owner box.
|
||||
reply, err := core.Chat(r.Context(), "web", text)
|
||||
if err != nil {
|
||||
log.Printf("chat api: %v", err)
|
||||
http.Redirect(w, r, "/chat", http.StatusSeeOther)
|
||||
|
||||
@@ -14,5 +14,27 @@
|
||||
<div>no notifications yet</div>
|
||||
<div class=hint>check back later or ask maven a question</div>
|
||||
</div>{{end}}
|
||||
<h2>Delivery outbox</h2>
|
||||
<p class=hint>
|
||||
every send is recorded before it leaves, so a failure is visible rather than silent.
|
||||
<a href="/notifications">all</a> ·
|
||||
<a href="/notifications?status=dropped">dropped</a> ·
|
||||
<a href="/notifications?status=failed">failed</a> ·
|
||||
<a href="/notifications?status=pending">pending</a> ·
|
||||
<a href="/notifications?status=unknown">unknown</a>
|
||||
</p>
|
||||
{{if .Attempts}}<div class=scroll><table>
|
||||
<tr><th>started</th><th>kind</th><th>rule</th><th>channel</th><th>status</th><th>finished</th></tr>
|
||||
{{range .Attempts}}<tr>
|
||||
<td class=hint>{{.Created}}</td>
|
||||
<td>{{.Kind}}</td>
|
||||
<td class=key>{{.Target}}</td>
|
||||
<td><span class=badge>{{.Channel}}</span></td>
|
||||
<td class={{.Status}}>{{.Status}}</td>
|
||||
<td class=hint>{{.Completed}}</td>
|
||||
</tr>{{end}}</table></div>
|
||||
{{else}}<div class=empty>
|
||||
<div>no delivery attempts{{if .Status}} with status {{.Status}}{{end}}</div>
|
||||
</div>{{end}}
|
||||
{{template "shellBottom"}}
|
||||
</html>
|
||||
|
||||
@@ -3,10 +3,10 @@
|
||||
{{if .Reminders}}<div class=scroll><table>
|
||||
<tr><th>created</th><th>fires</th><th>status</th><th>what</th></tr>
|
||||
{{range .Reminders}}<tr>
|
||||
<td class=hint>{{.CreatedTs.Format "02 Jan 15:04"}}</td>
|
||||
<td>{{.FireTs.Format "02 Jan 15:04"}}</td>
|
||||
<td class=hint>{{.Created}}</td>
|
||||
<td>{{.Fires}}</td>
|
||||
<td><span class="badge {{.Status}}">{{.Status}}</span></td>
|
||||
<td class=text-max>{{.Payload}}</td>
|
||||
<td class=text-max>{{.Text}}</td>
|
||||
</tr>{{end}}</table></div>
|
||||
{{else}}<div class=empty>
|
||||
<svg class=icon width="24" height="24"><use href="/ethos-icons.svg#i-calendar"/></svg>
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
)
|
||||
|
||||
// The page showed the storage envelope and the UTC instant (Vikunja #469).
|
||||
func TestReminderRowsUnwrapAndLocalise(t *testing.T) {
|
||||
fire := time.Date(2026, 8, 4, 18, 30, 0, 0, time.UTC)
|
||||
rows := reminderRows([]ipc.Reminder{{
|
||||
CreatedTs: fire.Add(-time.Hour),
|
||||
FireTs: fire,
|
||||
Status: "pending",
|
||||
Payload: `{"text":"выпить таблетки"}`,
|
||||
}})
|
||||
if len(rows) != 1 {
|
||||
t.Fatalf("rows = %d, want 1", len(rows))
|
||||
}
|
||||
if rows[0].Text != "выпить таблетки" {
|
||||
t.Errorf("Text = %q, want the words without the envelope", rows[0].Text)
|
||||
}
|
||||
if want := fire.Local().Format("02 Jan 15:04"); rows[0].Fires != want {
|
||||
t.Errorf("Fires = %q, want %q", rows[0].Fires, want)
|
||||
}
|
||||
if strings.Contains(rows[0].Text, "{") {
|
||||
t.Errorf("Text still carries JSON: %q", rows[0].Text)
|
||||
}
|
||||
}
|
||||
|
||||
// A payload that is not the envelope is his own words, so it is shown as it is.
|
||||
func TestReminderTextKeepsPlainPayload(t *testing.T) {
|
||||
for _, tc := range []struct{ in, want string }{
|
||||
{`{"text":"позвонить маме"}`, "позвонить маме"},
|
||||
{" полить цветы ", "полить цветы"},
|
||||
{`{"body":"nope"}`, `{"body":"nope"}`},
|
||||
{"", ""},
|
||||
} {
|
||||
if got := reminderText(tc.in); got != tc.want {
|
||||
t.Errorf("reminderText(%q) = %q, want %q", tc.in, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -20,6 +20,7 @@
|
||||
"bin_path": "llama-server",
|
||||
"n_gpu_layers": 99,
|
||||
"n_ctx": 4096,
|
||||
"cache_ram_mib": 512,
|
||||
"timeout": "60s",
|
||||
"llm_nudges": false
|
||||
},
|
||||
@@ -39,6 +40,23 @@
|
||||
"proxy": "socks5://192.168.240.1:10808"
|
||||
},
|
||||
|
||||
"//workstation": [
|
||||
"The big model on the desk PC (workpc, 7900 GRE 16GB), fronted by",
|
||||
"mavgpud on port 8080. It runs gemma-4-12b and it is preferred over the",
|
||||
"resident Qwen3-1.7B for routing and replies whenever the card is free.",
|
||||
"The machine is never assumed up: it sleeps, and the card is often held by",
|
||||
"a CPT run, in which case mavgpud answers 503 and Maven falls back to the",
|
||||
"resident model without saying so. Deleting this block restores exactly",
|
||||
"the behaviour homesrv had before it existed.",
|
||||
"Addressed by LAN address, not container name: mavgpud runs on another",
|
||||
"machine and there is no shared docker network to name it on."
|
||||
],
|
||||
"workstation": {
|
||||
"url": "http://192.168.1.105:8080",
|
||||
"probe": "15s",
|
||||
"timeout": "90s"
|
||||
},
|
||||
|
||||
"//search": [
|
||||
"The live web, searched after his own notes and before Kiwix. Only the",
|
||||
"query string leaves the box — never a note, a fact, the persona block or",
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
{
|
||||
"listen": ":8080",
|
||||
"llama_addr": "127.0.0.1:10000",
|
||||
"llama_bin": "llama-server",
|
||||
"llama_args": [
|
||||
"-m", "/mnt/D/AI/gemma4/gemma-4-12B-it-qat-UD-Q4_K_XL.gguf",
|
||||
"-md", "/mnt/D/AI/gemma4/mtp-gemma-4-12B-it-BF16.gguf",
|
||||
"-ngl", "99",
|
||||
"-fa", "on",
|
||||
"-np", "1",
|
||||
"--host", "127.0.0.1",
|
||||
"--port", "10000",
|
||||
"--ctx-size", "32768",
|
||||
"--threads", "6",
|
||||
"--batch-size", "2048",
|
||||
"--ubatch-size", "512",
|
||||
"--jinja",
|
||||
"--chat-template-kwargs", "{\"enable_thinking\":false}",
|
||||
"--spec-type", "draft-mtp",
|
||||
"--spec-draft-n-max", "2"
|
||||
],
|
||||
|
||||
"kfd_root": "/sys/class/kfd/kfd/proc",
|
||||
"drm_device": "/sys/class/drm/card1/device",
|
||||
|
||||
"poll": "1s",
|
||||
"idle_timeout": "15m",
|
||||
"stop_grace": "20s",
|
||||
"min_free_vram_bytes": 10737418240,
|
||||
"evict_after_polls": 2,
|
||||
"start_after_polls": 5
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
[Unit]
|
||||
# Runs on the workstation (bugmachine), not on homesrv. Install as a systemd
|
||||
# user unit and turn on lingering, so the card is supervised after a reboot
|
||||
# with nobody logged in:
|
||||
#
|
||||
# scp mavgpud workpc:~/.local/bin/mavgpud
|
||||
# scp deploy/mavgpud.json workpc:~/.config/mavgpud.json
|
||||
# scp deploy/mavgpud.service workpc:~/.config/systemd/user/mavgpud.service
|
||||
# ssh workpc 'systemctl --user daemon-reload && systemctl --user enable --now mavgpud'
|
||||
# sudo loginctl enable-linger kami
|
||||
Description=Maven GPU supervisor (holds llama-server while the card is free)
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
ExecStart=%h/.local/bin/mavgpud -config %h/.config/mavgpud.json
|
||||
Restart=always
|
||||
RestartSec=5
|
||||
# The card must come back when the supervisor goes down. mavgpud stops
|
||||
# llama-server on SIGTERM, so give it longer than stop_grace to do that.
|
||||
KillSignal=SIGTERM
|
||||
TimeoutStopSec=60
|
||||
# llama-server aborts inside its own static teardown on SIGTERM, so every
|
||||
# routine yield used to write a multi-gigabyte core into systemd-coredump
|
||||
# (Vikunja #491). Yielding is meant to happen several times a day.
|
||||
LimitCORE=0
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -223,6 +223,30 @@ Not alternatives — layers:
|
||||
Router contract: `[{"intent":<enum>, key?, value?, text?, verb?}, ...]` over
|
||||
7 intents (`fact, reminder, note, query, act, chat, system`).
|
||||
|
||||
#### A restart expires a parked question
|
||||
|
||||
Decided 2026-08-04 (Vikunja #385). The follow-up dialogue session survives a
|
||||
restart; the clarify question parked behind it does not, and neither do the
|
||||
three yes/no confirms in `voice.go`. `ClarifyStore` stays in memory.
|
||||
|
||||
Three reasons, in the order they settle it:
|
||||
|
||||
- The clock stops meaning anything. A parked question carries a 90s TTL and an
|
||||
attempt count. A restart is a gap of unknown length, so a restored question is
|
||||
either already dead or pretending to be young.
|
||||
- Restoring the question restores the request behind it. He asked for something,
|
||||
she asked back, and then the daemon went away. Acting on that minutes later,
|
||||
against words he has probably given up on, is the misroute the stage 3 gate
|
||||
exists to avoid.
|
||||
- She does not announce it either. The expiry notice needs to know a question
|
||||
was parked, and knowing that across a restart means storing it. One sentence,
|
||||
in the rare window where he speaks within 90s of a restart, does not pay for a
|
||||
marker that outlives the thing it describes. His next words route fresh, which
|
||||
is the correct answer with or without the notice.
|
||||
|
||||
So the notice stays what it is: the in-process TTL case, where she really did
|
||||
wait and really did let go.
|
||||
|
||||
### save-where — the two-memory routing axis
|
||||
|
||||
One discriminator: **does the loop evaluate a predicate against it?**
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
# gemma-4-12b on the workstation, against the resident Qwen3-1.7B
|
||||
|
||||
Measured 2026-08-02 on the fixtures as they stand. Dated file: it is not edited
|
||||
after today, and a newer number is a new file.
|
||||
|
||||
Vikunja #485's first assumption was that a 7-14B measurably beats Qwen3-1.7B on
|
||||
the 77-case RU routing fixture and the 27-case talk fixture. It does, on both,
|
||||
and it is also faster.
|
||||
|
||||
## The setup
|
||||
|
||||
`gemma-4-12B-it-qat-UD-Q4_K_XL` with the `mtp-gemma-4-12B-it-BF16` draft model,
|
||||
served by `llama-server` b10220 on bugmachine (AMD 7900 GRE, 16GB), fronted by
|
||||
`mavgpud` on `192.168.1.105:8080`. Thinking is off through
|
||||
`--chat-template-kwargs '{"enable_thinking":false}'`, speculative decoding is
|
||||
`--spec-type draft-mtp --spec-draft-n-max 2`, context 32768. The exact line is
|
||||
`deploy/mavgpud.json`.
|
||||
|
||||
Every number below crossed the LAN from homesrv. Note the trap: homesrv's shell
|
||||
exports `HTTP_PROXY`, Go honours it, and the runs need
|
||||
`env -u HTTP_PROXY -u HTTPS_PROXY -u http_proxy -u https_proxy`.
|
||||
|
||||
## Routing, 77-case RU fixture
|
||||
|
||||
| | full | intent-only | p50 | p95 |
|
||||
|---|---|---|---|---|
|
||||
| classifier alone (02-08) | 68.8% | — | 16.6µs | — |
|
||||
| Qwen3-1.7B through the cascade (31-07, 02-08) | 72.7% | 77.9% | 0.80-1.04s | — |
|
||||
| **gemma-4-12b through the cascade** | **84.4%** | **93.5%** | **329ms** | 429ms |
|
||||
| gemma-4-12b alone, no cascade | 55.8% | 85.7% | 335ms | 436ms |
|
||||
|
||||
The workstation buys 11.7 points of full accuracy over the resident model. It
|
||||
buys 15.6 points of intent-only, at a third of the latency. The router's p50 was
|
||||
never the model's fault, which the 02-08 contention finding already said. A 12B
|
||||
on a free 16GB card answers a routing turn in a third of a second.
|
||||
|
||||
Two things the table hides.
|
||||
|
||||
The alone-versus-cascade gap is slots, not intents. gemma reads the intent right
|
||||
85.7% of the time on its own. It loses full accuracy on seven fact keys
|
||||
(`вода` instead of `water`, `ужин` instead of `meal`) and on six reminder times
|
||||
with no time slot. Stage 0 and the daemon's own extractor repair
|
||||
both, which is why the cascade is 28 points higher. The lesson is that the
|
||||
cascade earns its keep even under a much better model, not that it is scaffolding
|
||||
to remove.
|
||||
|
||||
`errors: 6` in the alone row are declines on single-token and ambiguous
|
||||
utterances, all of which the cascade caught. The remaining defects through the
|
||||
cascade are three `query→fact` confusions, one `chat→query`, and one false
|
||||
clarify.
|
||||
|
||||
## Talk, 27-case conversational fixture
|
||||
|
||||
| | pass | notes |
|
||||
|---|---|---|
|
||||
| Qwen3-1.7B (31-07) | 20/27 | 11-17/27 for the 0.8B before it |
|
||||
| **gemma-4-12b** | **25/27 (92.6%)** | chat 8/9, knowledge 9/9, query 8/9 |
|
||||
|
||||
Knowledge is the interesting column: 9/9, in Russian, with real answers about
|
||||
Rayleigh scattering, SSD versus HDD and thunder delay. That is the case the
|
||||
1.7B cannot do at all and the reason the naming half of the degradation rule
|
||||
exists.
|
||||
|
||||
Two failures, and one of them is the persona defect the CPT (#122) targets:
|
||||
`query-notes-do-not-answer` wrote `заплатил` where Maven needs the feminine
|
||||
form. The other is `chat-joke`, where the model told a joke without using any of
|
||||
the words the check looks for. Run-to-run variance is about one case: a second
|
||||
run scored 24/27 with `chat-followup-server` also off-topic.
|
||||
|
||||
## Nudge phrasing, 15-case fixture
|
||||
|
||||
15/15, every check, no errors. `mood`, `lang`, `length`, `feminine`,
|
||||
`hisgender`, `address`, `cringe` and `ontopic` all clean.
|
||||
|
||||
## What this settles and what it does not
|
||||
|
||||
Settled: the size question. A 12B on the workstation beats the resident model on
|
||||
every fixture we have, and it is faster. The offload argument holds.
|
||||
|
||||
Not settled: how often the card is free. That is #485's second assumption and
|
||||
only the `mavgpud` log answers it, after a week of the owner's normal work. A
|
||||
model that is better whenever it is up is worth little if it is never up.
|
||||
@@ -0,0 +1,84 @@
|
||||
# Where the resident model's 7.9GB of RSS goes (2026-08-03, homesrv)
|
||||
|
||||
Measured for Vikunja #499. The deployed llama-server held 7.9GB RSS for a 1.1GB
|
||||
model file. Half a gigabyte of it was in swap, on a box that also runs
|
||||
whisper.cpp, piper and the embedder.
|
||||
|
||||
## Method
|
||||
|
||||
`maven-mavend-1` was stopped for the measurement, with the owner's approval.
|
||||
Its own binary then ran on the host with the exact deployed command line. That
|
||||
binary is `/opt/maven/bin/llama-server`, version `1 (4c65955)`, a Vulkan build.
|
||||
|
||||
```sh
|
||||
llama-server -m /mnt/hdd1/llms/qwen3/Qwen3-1.7B-UD-Q4_K_XL.gguf \
|
||||
--host 127.0.0.1 --port 18099 -c 4096 -ngl 99 --no-webui
|
||||
```
|
||||
|
||||
RSS was read from `/proc/<pid>/status` after load and after each of 8 distinct
|
||||
1521-token prompts. `smaps` of the deployed process was read first, from inside
|
||||
the container, since the host user cannot read another user's maps.
|
||||
|
||||
## The cause: the prompt cache, not the weights and not the offload
|
||||
|
||||
The startup log says it outright:
|
||||
|
||||
```text
|
||||
srv load_model: prompt cache is enabled, size limit: 8192 MiB
|
||||
srv llama_server: n_parallel is set to auto, using n_parallel = 4 and kv_unified = true
|
||||
```
|
||||
|
||||
The server saves the full KV state of every idle slot it evicts. It keeps up to
|
||||
8GiB of those states in host RAM (llama.cpp PR 16391). One saved prompt of 1521
|
||||
tokens costs 166.377 MiB. That is 112 kiB per token, exactly Qwen3-1.7B's KV
|
||||
footprint (28 layers x 2 x 1024 dims x 2 bytes).
|
||||
|
||||
RSS at rest, and per distinct prompt:
|
||||
|
||||
| Prompts served | RSS, default | RSS, `--cache-ram 512` |
|
||||
|---|---|---|
|
||||
| 0 (just loaded) | 443 MB | 411 MB |
|
||||
| 1 | 445 MB | 411 MB |
|
||||
| 4 | 958 MB | 929 MB |
|
||||
| 8 | 1641 MB | 932 MB |
|
||||
|
||||
Uncapped, RSS climbs about 170MB per distinct prompt and does not stop until
|
||||
the 8GiB limit. Capped at 512 MiB it plateaus at 932MB from the fourth prompt
|
||||
on, with the cache holding steady at `3 prompts, 499.132 MiB` and evicting.
|
||||
|
||||
The 7.9GB on the running daemon was that climb, weeks of it. Its `smaps` showed
|
||||
one 6.03GB anonymous mapping at 5.32GB resident plus a 1.45GB mapping at 1.27GB
|
||||
resident, and only 30MB of file-backed RSS.
|
||||
|
||||
## The task's leading guess was wrong
|
||||
|
||||
`-ngl 99` on the Vega iGPU costs almost no process RSS. A freshly loaded server
|
||||
has 95MB of anonymous RSS in total. RADV allocates device memory through the
|
||||
kernel, outside the process, and the log sees 8202 MiB free on `Vulkan0`. The
|
||||
weights are mmapped and file-backed, so they are evictable and do not pin RSS. The logit buffer is not visible in the numbers above at all.
|
||||
|
||||
## Decision
|
||||
|
||||
`--cache-ram 512` is now the default, wired as `phraser.cache_ram_mib` and set
|
||||
in `deploy/mavend.json`. 512 MiB caps total RSS near 1GB, an eighth of what the
|
||||
box carried. It still holds three of the 1521-token probes above. Maven's real
|
||||
routing and phrasing prompts are much shorter, so it holds more of those than
|
||||
the table suggests. `-c 4096` is untouched, as #499
|
||||
required. A negative `cache_ram_mib` passes no flag, for a llama-server too old
|
||||
to know it.
|
||||
|
||||
Not changed: `n_parallel = 4`. With `kv_unified = true` the four slots share one
|
||||
4096-token KV cache, so they do not multiply it.
|
||||
|
||||
The other half of #499 was that none of these lines were reachable. mavend
|
||||
scraped llama-server's stderr for the listen line and discarded it, and never
|
||||
piped stdout at all. Both streams now go to mavend's log with a `llama:` prefix.
|
||||
The last 12 startup lines go into the error when the server dies before it
|
||||
listens.
|
||||
|
||||
## Deployed
|
||||
|
||||
The `mavenai:latest` image was rebuilt and `maven-mavend-1` recreated the same
|
||||
day. The daemon's own log now carries the child's startup, it reads
|
||||
`prompt cache is enabled, size limit: 512 MiB`, and the resident server sat at
|
||||
439MB RSS after load and 613MB after one served turn.
|
||||
@@ -0,0 +1,46 @@
|
||||
# Personal boundary, seed scoring vs possession markers, 2026-08-03
|
||||
|
||||
Vikunja #495. `что я говорил про бэкапы?` walked past the personal boundary into
|
||||
SearXNG and came back answered from a Habr article. The boundary matched
|
||||
possession words only, so a first-person speech verb was not a personal
|
||||
question.
|
||||
|
||||
## What changed
|
||||
|
||||
The boundary now scores the turn's query vector against two frozen seed sets.
|
||||
It claims the turn when the personal side is nearer than the world side. Seeds
|
||||
and code are in `cmd/mavend/personalboundary.go`. The possession markers stay as
|
||||
the offline floor for a handler with no embedder.
|
||||
|
||||
A regex speech class was written first and dropped. Russian gives every verb a
|
||||
dozen surface forms, and the "как я говорил, ..." preamble list has no end. Each
|
||||
form the lexicon missed was one more question reaching the world.
|
||||
|
||||
## Measurement
|
||||
|
||||
Embedder: multilingual-e5-small int8, the one homesrv runs. Both sides are
|
||||
embedded on the query side. Cases are held out, none of them a seed. `make test`
|
||||
runs the offline part. The scored part is opt-in through `MAVEN_ONNX_LIB`, like
|
||||
`TestONNXRecall`.
|
||||
|
||||
19/19 held-out utterances correct (TestONNXPersonalBoundary)
|
||||
|
||||
true positive margins +0.014 to +0.089
|
||||
nearest true negative -0.005 ("кто такой гагарин")
|
||||
|
||||
One case missed during the first pass and is not held out any more: `as i said,
|
||||
what is the population of india`, +0.008 to the personal side. It is a world seed
|
||||
now.
|
||||
|
||||
The gate is the sign of the difference and nothing tighter. The margins are too
|
||||
thin for a threshold. The asymmetry favours claiming: a false claim costs one
|
||||
honest "не знаю", a false pass sends his life to an upstream engine.
|
||||
|
||||
`make eval-recall` unchanged, 18/27 answered at gate 0.55. Recall does not touch
|
||||
this path.
|
||||
|
||||
## Not verified
|
||||
|
||||
The live probe on the deployed box. The daemon was not rebuilt in this session.
|
||||
The reply to `что я говорил про бэкапы?` with no matching note is still untested
|
||||
against a real SearXNG.
|
||||
@@ -0,0 +1,65 @@
|
||||
# Recall topic veto, what it costs and what it buys, 2026-08-03
|
||||
|
||||
Vikunja #496. The task asked for a cross-language fix. Skip the topic veto in
|
||||
`memory.RecallAllowed` when the question and the hit are in different scripts.
|
||||
An English question would then stop losing a Russian note.
|
||||
|
||||
No such case exists. No fixture case puts the question and its wanted note in
|
||||
different scripts. The case the task named is not one either.
|
||||
|
||||
en-hard-024
|
||||
query "what fixed the screen problem"
|
||||
note "the flicker went away once i swapped the display cable"
|
||||
|
||||
Both are English. It is a paraphrase failure, not a language failure. A script
|
||||
test would not have changed a single case, and neither would a bilingual stem
|
||||
map.
|
||||
|
||||
## What the veto is worth today
|
||||
|
||||
Measured with the real embedder, multilingual-e5-small int8, gate 0.55, margin
|
||||
0.008. The first row is the veto as it ships. The second is `RecallAllowed`
|
||||
forced to true.
|
||||
|
||||
| | cases passing | answered | false recall | silenced by gate |
|
||||
|---|---|---|---|---|
|
||||
| veto on | 22/32 | 17/27 | 0/5 | 2 |
|
||||
| veto off | 22/32 | 18/27 | 1/5 | 1 |
|
||||
|
||||
The pass count does not move. The veto trades one true recall for one false one.
|
||||
It costs `en-hard-024` and it buys `ru-silent-029`:
|
||||
|
||||
ru-silent-029
|
||||
query "во сколько отходит поезд"
|
||||
note "погулял вдоль реки" 0.835, margin 0.019
|
||||
|
||||
The second case counted as silenced by the gate is `ru-home-026` at margin
|
||||
0.001, which the margin gate stops. The veto has nothing to do with it.
|
||||
|
||||
## Why no lexical rule separates the two
|
||||
|
||||
`en-hard-024` and `ru-silent-029` are in the same lexical class. Both questions
|
||||
share zero content words with their hit, and neither carries a first-person
|
||||
marker. The scores sit on top of each other, 0.826 against 0.835, and so do the
|
||||
margins, 0.023 against 0.019. Only one thing separates them. A screen problem
|
||||
and a swapped display cable are the same event. A train and a river walk are
|
||||
not. The embedder scores that difference at nine thousandths.
|
||||
|
||||
So the signal is semantic and the gate is lexical. Any rule cheap enough to sit
|
||||
in `RecallAllowed` and strong enough to recover `en-hard-024` also re-admits
|
||||
`ru-silent-029`, which puts false recall back to 1/5.
|
||||
|
||||
One near-miss rule was tried on paper and rejected: let the veto pass when the
|
||||
hit itself is first person. It works on these two, because the English note says
|
||||
"i swapped" and the Russian note says only "погулял". It is backwards as a
|
||||
principle. A first-person note is exactly the personal note the veto keeps away
|
||||
from a world question. The rule would weaken the veto where it was designed to
|
||||
bite. It survives here only because Russian drops the pronoun.
|
||||
|
||||
## Decision
|
||||
|
||||
Accept the loss. `en-hard-024` stays silenced and false recall stays 0/5.
|
||||
|
||||
The way out is a reranker, not a longer word list. Recall@3 is 85.2% against
|
||||
recall@1 at 70.4%, so the right note is usually in the returned set and ranked
|
||||
wrong. That is where the remaining points are, and it is not this task.
|
||||
+77
-12
@@ -1,6 +1,6 @@
|
||||
# Offloading model work to the workstation
|
||||
|
||||
*Last verified: 2026-08-02 @ 5c05163. Living doc: correct it in place, do not append.*
|
||||
*Last verified: 2026-08-03 @ 12530c8. Living doc: correct it in place, do not append.*
|
||||
|
||||
Owner's call, 2026-08-02. Vikunja #483 is the umbrella. Tasks #484 to #487 are the
|
||||
work, and this file holds the shape and the rules all four must obey.
|
||||
@@ -47,6 +47,24 @@ service being down.
|
||||
|
||||
Nothing in between. A turn never breaks on the workstation being asleep.
|
||||
|
||||
Both halves are wired, 03-08-2026. `LLMPhraser.PhraseWorld`
|
||||
(`internal/phraser/world.go`) is the naming half and has three outcomes, not two:
|
||||
|
||||
| State | What he hears |
|
||||
|---|---|
|
||||
| no `workstation` block | the resident model answers, exactly as before the seam existed |
|
||||
| configured, card free | the workstation answers |
|
||||
| configured, asleep or busy | the gap, `worldGap` in `cmd/mavend/worldmodel.go` |
|
||||
|
||||
The first row is the one worth stating. Naming a gap requires a gap. On a box with
|
||||
no second model the 1.7B is the whole product. Refusing every world question there
|
||||
would remove a capability the owner has today.
|
||||
|
||||
A source holding a passage is on the naming half too: a live search, a ZIM
|
||||
article, a page he named. None of them says "не могу сейчас". They read the
|
||||
passage back, which is what `phraseSource` returning `""` selects. A real quote
|
||||
beats a gap, and neither path invents.
|
||||
|
||||
## Admission control, not a scheduler
|
||||
|
||||
There is no GPU arbiter. That is a service with its own failure modes, and nothing
|
||||
@@ -58,6 +76,34 @@ jobs.
|
||||
|
||||
The caller must be able to ask "is this peer usable right now" without a turn
|
||||
hanging on a timeout. A dead remote is a normal state, not an error state.
|
||||
`internal/llm.Pair` is that check on the Maven side. A prober caches the answer,
|
||||
so `Available()` is an atomic read and no turn pays for a health check.
|
||||
|
||||
llama-server does not stay up on the workstation. It cannot: a resident 7-14B
|
||||
would hold 16GB against the owner's CPT runs. So a supervisor there owns its
|
||||
lifecycle, keeps it loaded while the card is free, and unloads it on idle or
|
||||
when another process needs the card (owner's call, 2026-08-02, Vikunja #488).
|
||||
|
||||
That supervisor is still not a scheduler, and the distinction is worth holding.
|
||||
It arbitrates nothing between callers. It reports whether it can take work and
|
||||
manages one process to back that answer. Maven never asks it to start anything
|
||||
and never learns that it did.
|
||||
|
||||
Contention is decided by presence under `/sys/class/kfd/kfd/proc`, not by a VRAM
|
||||
threshold. A ROCm process registers there when it initialises HIP, before it
|
||||
allocates anything. So the supervisor sees a contender during that job's startup,
|
||||
and yields before the job loses the memory it asked for. A
|
||||
threshold reads the card too late. By the time free VRAM has dropped, the other
|
||||
job has already lost the allocation race. Free VRAM is still read, but only as a
|
||||
precondition for loading, never as the eviction signal. One blind spot is known.
|
||||
A job can take the card without registering on the KFD, as a Vulkan or a
|
||||
video-decode job would. `describe()` logs every contender's comm, and that log is
|
||||
how we find out whether the blind spot is real.
|
||||
|
||||
`mavgpud` runs from a systemd unit on the workstation with
|
||||
`deploy/mavgpud.json` as its config, and `llama_args` is passed to llama-server
|
||||
untouched. The model, the context size, the layer count and the MTP flags are the
|
||||
owner's business and not this daemon's schema.
|
||||
|
||||
## What stays on homesrv, permanently
|
||||
|
||||
@@ -76,17 +122,29 @@ it buys nothing. Four callers:
|
||||
|
||||
## Inventory: what runs a model on homesrv today
|
||||
|
||||
The **resident model** is one llama-server with seven callers:
|
||||
The **resident model** is one llama-server with seven callers, and 03-08-2026 is
|
||||
the date each of them stopped or did not stop being resident-only:
|
||||
|
||||
| Caller | What for |
|
||||
|---|---|
|
||||
| `cmd/mavend/voicewire.go` | routing |
|
||||
| `cmd/mavend/replier_llm.go` | replies |
|
||||
| `cmd/mavend/tick.go` | digestion worker: `PhraseNudge`, `PhraseReminder` |
|
||||
| `cmd/mavend/capture.go` | capture summarisation (unreachable, see #480) |
|
||||
| `cmd/mavend/mail.go` | mail extraction (off, no IMAP) |
|
||||
| `cmd/mavend/kiwixwire.go` | answering from a Kiwix, search or crawl passage |
|
||||
| `memoryeval.go`, `modelswap.go` | admin and evals |
|
||||
| Caller | What for | Offloaded |
|
||||
|---|---|---|
|
||||
| `cmd/mavend/voicewire.go` | routing | silently, through `hot` |
|
||||
| `cmd/mavend/replier_llm.go` | replies | silently, through `hot` |
|
||||
| `cmd/mavend/tick.go` | digestion worker: `PhraseNudge`, `PhraseReminder` | silently, inside the phraser |
|
||||
| `cmd/mavend/actions_query.go` | world questions, and any fetched passage | names the gap |
|
||||
| `cmd/mavend/capture.go` | capture summarisation (unreachable, see #480) | no, holds its own client |
|
||||
| `cmd/mavend/mail.go` | mail extraction (off, no IMAP) | no, holds its own client |
|
||||
| `memoryeval.go`, `modelswap.go` | admin and evals | no, and deliberately |
|
||||
|
||||
The last three rows are resident-only on purpose. `memoryeval.go` and
|
||||
`modelswap.go` measure and swap the resident model, so sending their work
|
||||
elsewhere would measure the wrong thing. `capture.go` and `mail.go` are
|
||||
background jobs that hold a gated background client (`llmBackgroundClientFor`),
|
||||
and that priority has no equivalent on the remote yet. Both are also unreachable
|
||||
on this deploy, so wiring them would ship an untestable path.
|
||||
|
||||
The `tick.go` row needs one caveat. `phraser.llm_nudges` is `false` in deploy, so
|
||||
nudges come from templates and the seam under them changes nothing until that
|
||||
flips. It is wired anyway: `PhraseReminder` is on the same transport and is on.
|
||||
|
||||
Then the embedder above, **whisper.cpp** in `mavsttd`, and **piper** in `mavttsd`.
|
||||
`mavwaked` uses no model at all: an energy-threshold VAD over 30ms frames.
|
||||
@@ -97,7 +155,14 @@ Then the embedder above, **whisper.cpp** in `mavsttd`, and **piper** in `mavttsd
|
||||
`internal/netaddr` landed in PR #92. A seam address now carries its own scheme,
|
||||
and a scheme-less one is still unix. A tcp seam requires a shared token, because
|
||||
the filesystem permission that authenticated the unix socket is gone.
|
||||
2. **The resident model** (#485). Biggest quality delta. A 16GB card runs a 7-14B,
|
||||
2. **The resident model** (#485, #490). Wired. A `workstation` block builds an
|
||||
`llm.Pair` in `modelSeam` (`cmd/mavend/voicewire.go`), routing and replies
|
||||
complete through it, and the phraser holds the same pair (`UseRemote`). Both
|
||||
halves of the rule are live: see the table above for which caller gets which.
|
||||
Measured, `docs/evals/2026-08-02-workstation-gemma4-12b.md`: gemma-4-12b
|
||||
through the cascade scores 84.4% full accuracy at p50 329ms. The resident
|
||||
model scores 72.7% at p50 0.80-1.04s. On the talk fixture it is 25/27
|
||||
against 20/27. Biggest quality delta. A 16GB card runs a 7-14B,
|
||||
which fixes what the 1.7B gets wrong: world knowledge, and the persona the CPT
|
||||
targets. The degradation path is already written and measured, since the
|
||||
classifier scores 68.8% full accuracy at p50 16.6µs on its own.
|
||||
|
||||
+15
-6
@@ -498,12 +498,21 @@ rejects `https://api.openai.com`, and forget really deletes
|
||||
(`internal/store/memory.go:145` is a real `DELETE`, not a tombstone). Vision is
|
||||
19/19, speaker 22/22, media 16/16.
|
||||
|
||||
**470 got worse.** Both poisoned facts show `voided` on `/history`, and the
|
||||
defect survives. Re-measured at 15:42, after four restarts: `почему небо синее?`
|
||||
still answers `какая последняя версия языка Go?` with no `search:` line. What
|
||||
comes back is the question he typed, not the value the fact held. So the poison
|
||||
is a vector in the memory index, and `revert` does not remove it. There is
|
||||
currently no documented way to repair a poisoned box.
|
||||
**470 got worse, then closed.** Both poisoned facts showed `voided` on
|
||||
`/history` and the defect survived. Re-measured at 15:42, after four restarts:
|
||||
`почему небо синее?` still answered `какая последняя версия языка Go?` with no
|
||||
`search:` line. What came back was the question he typed, not the value the fact
|
||||
held. So the poison was a vector in the memory index, and `revert` did not
|
||||
remove it.
|
||||
|
||||
Repaired in two parts. 470 stopped the writes: a question is never a fact, and a
|
||||
void drops the key's vectors. 493 fixed what the index holds. A fact is indexed
|
||||
as the fact and not as the utterance, and a correction drops its superseded
|
||||
vector too.
|
||||
|
||||
A poisoned box now repairs itself on the next start. `RepairFactVectors`
|
||||
re-embeds every fact vector from the fact it names, and deletes the voided and
|
||||
superseded ones. It runs once, guarded by a marker, and logs what it did.
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -311,7 +311,7 @@ func TestGate_IpcServer_CheckWiredThroughSocket(t *testing.T) {
|
||||
if fake.writes != 0 {
|
||||
t.Errorf("auth refused but CoreAPI was called %d time(s); refused calls must not reach CoreAPI", fake.writes)
|
||||
}
|
||||
_, err = cli.Chat(context.Background(), "привет")
|
||||
_, err = cli.Chat(context.Background(), "web", "привет")
|
||||
if !errors.Is(err, ipc.ErrForbidden) {
|
||||
t.Errorf("wire: chat from unenrolled uid = %v; want ipc.ErrForbidden", err)
|
||||
}
|
||||
@@ -344,7 +344,7 @@ func TestGate_IpcServer_ChatAllowedForEnrolledCaller(t *testing.T) {
|
||||
t.Fatalf("dial: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { _ = cli.Close() })
|
||||
reply, err := cli.Chat(context.Background(), "привет")
|
||||
reply, err := cli.Chat(context.Background(), "web", "привет")
|
||||
if err != nil {
|
||||
t.Fatalf("Chat: %v", err)
|
||||
}
|
||||
@@ -373,7 +373,7 @@ func (r *recordingAPI) WriteFact(_ context.Context, _ ipc.WriteFactReq) (int64,
|
||||
return int64(r.writes), nil
|
||||
}
|
||||
|
||||
func (r *recordingAPI) Chat(_ context.Context, text string) (string, error) {
|
||||
func (r *recordingAPI) Chat(_ context.Context, _, text string) (string, error) {
|
||||
r.chats++
|
||||
return "echo: " + text, nil
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// Fact sources. A calendar event reaches the store as a
|
||||
@@ -153,14 +154,20 @@ func Overlapping(events []Event, from, to time.Time) []Event {
|
||||
return out
|
||||
}
|
||||
|
||||
// safeKey makes a summary safe to use inside a fact key (ASCII alphanumerics
|
||||
// and dashes). Non-Latin summaries collapse to their punctuation, which is why
|
||||
// the day prefix carries the identity and this only disambiguates within a day.
|
||||
// safeKey makes a summary safe to use inside a fact key: letters and digits in
|
||||
// any script, plus dashes, with space and underscore folded to a dash.
|
||||
//
|
||||
// It kept ASCII only until 04-08-2026, and dropped everything else. His
|
||||
// calendar is Russian, so "Встреча с Аней" and "Обед с мамой" both reduced to
|
||||
// "--" and produced the same key on the same day — the second event of the day
|
||||
// silently overwrote the first (Vikunja #443). Letting the letters through is
|
||||
// what makes the key identify the event. Migration #18 drops the keys written
|
||||
// under the old rule; they are re-derived on the next poll.
|
||||
func safeKey(s string) string {
|
||||
var b strings.Builder
|
||||
for _, r := range s {
|
||||
switch {
|
||||
case (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z') || (r >= '0' && r <= '9') || r == '-':
|
||||
case unicode.IsLetter(r) || unicode.IsDigit(r) || r == '-':
|
||||
b.WriteRune(r)
|
||||
case r == ' ' || r == '_':
|
||||
b.WriteRune('-')
|
||||
|
||||
@@ -139,6 +139,9 @@ func TestSafeKey(t *testing.T) {
|
||||
{"Hello_World", "Hello-World"},
|
||||
{"special@#$chars!!", "specialchars"},
|
||||
{"ALL_CAPS_123", "ALL-CAPS-123"},
|
||||
// His calendar is Russian. These reduced to "--" and "--" (Vikunja #443).
|
||||
{"Встреча с Аней", "Встреча-с-Аней"},
|
||||
{"Обед с мамой", "Обед-с-мамой"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
if got := safeKey(tt.in); got != tt.want {
|
||||
@@ -263,3 +266,19 @@ func TestSourceTrust(t *testing.T) {
|
||||
t.Errorf("Sources() = %v", Sources())
|
||||
}
|
||||
}
|
||||
|
||||
// Two Russian events on one day must not share a key. They did: safeKey kept
|
||||
// ASCII only, so both summaries collapsed to their spaces and the second event
|
||||
// overwrote the first in the store (Vikunja #443).
|
||||
func TestFactKeyDistinguishesRussianEventsOnOneDay(t *testing.T) {
|
||||
day := time.Date(2026, 8, 4, 0, 0, 0, 0, time.UTC)
|
||||
a := Event{Summary: "Встреча с Аней", Start: day.Add(10 * time.Hour), End: day.Add(11 * time.Hour)}
|
||||
b := Event{Summary: "Обед с мамой", Start: day.Add(13 * time.Hour), End: day.Add(14 * time.Hour)}
|
||||
if FactKeyIn(a, time.UTC) == FactKeyIn(b, time.UTC) {
|
||||
t.Fatalf("both events keyed as %q", FactKeyIn(a, time.UTC))
|
||||
}
|
||||
// The day prefix still has to survive, because the store range-scans on it.
|
||||
if !strings.HasPrefix(FactKeyIn(a, time.UTC), KeyPrefixForDay(day)) {
|
||||
t.Fatalf("key %q lost the day prefix %q", FactKeyIn(a, time.UTC), KeyPrefixForDay(day))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,6 +224,11 @@ type Config struct {
|
||||
// See SearchConfig.
|
||||
Search *SearchConfig `json:"search,omitempty"`
|
||||
|
||||
// Workstation — the big model on the owner's desktop, preferred over the
|
||||
// resident one when its GPU is free. nil / absent / url empty ⇒ homesrv
|
||||
// behaves exactly as it does today. See WorkstationConfig.
|
||||
Workstation *WorkstationConfig `json:"workstation,omitempty"`
|
||||
|
||||
// Praxis — the ecosystem attention-state service. When configured, maven
|
||||
// calls the Praxis HTTP tools API for attention listing and item lifecycle.
|
||||
// Maven never touches Praxis's database directly (ecosystem invariant: no
|
||||
@@ -596,6 +601,9 @@ type MorningRoutineItemConfig struct {
|
||||
Key string `json:"key"`
|
||||
FactKey string `json:"fact_key"`
|
||||
Label string `json:"label"`
|
||||
// Optional — this one being skipped does not earn a nudge. Default false,
|
||||
// so a routine written before 04-08-2026 keeps behaving as it did.
|
||||
Optional bool `json:"optional,omitempty"`
|
||||
}
|
||||
|
||||
// QuietHoursConfig — a recurring daily quiet-window. Times are local to the
|
||||
@@ -1105,6 +1113,44 @@ const (
|
||||
DefaultKiwixSnippetRunes = 1500
|
||||
)
|
||||
|
||||
// WorkstationConfig — the big model on the owner's desktop (workpc, a
|
||||
// 7900 GRE with 16GB), fronted by mavgpud.
|
||||
//
|
||||
// homesrv cannot grow a GPU, so the resident Qwen3-1.7B is the floor and this
|
||||
// is the preferred model above it (owner's call, 2026-08-02, docs/offload.md).
|
||||
// The workstation is never assumed up: its card is often held by a CPT run and
|
||||
// the machine sleeps. No block, or an empty URL, and homesrv behaves exactly as
|
||||
// it does today.
|
||||
//
|
||||
// Only the prompt crosses the LAN, and the workstation is not "the box". The
|
||||
// rules in CLAUDE.md about what may leave still apply.
|
||||
type WorkstationConfig struct {
|
||||
// URL — where mavgpud listens, e.g. "http://192.168.1.105:8080". Empty ⇒
|
||||
// the whole block is normalised to nil and nothing probes anything.
|
||||
URL string `json:"url,omitempty"`
|
||||
|
||||
// Health — the admission endpoint. Empty ⇒ URL + "/health", which is what
|
||||
// mavgpud serves. It answers 503 while the card is held, and that is the
|
||||
// signal, so it must be the supervisor's endpoint and not llama-server's.
|
||||
Health string `json:"health,omitempty"`
|
||||
|
||||
// Probe — how often admission is re-checked. 0 ⇒ DefaultWorkstationProbe.
|
||||
// Nothing on the hot path waits for it: the answer is cached and read
|
||||
// atomically, so this only sets how late Maven notices the card came back.
|
||||
Probe Duration `json:"probe,omitempty"`
|
||||
|
||||
// Timeout — the per-request budget for a completion on the workstation.
|
||||
// 0 ⇒ DefaultWorkstationTimeout. A big model on a LAN host is slower than
|
||||
// the resident one, and a request that overruns falls back to the floor.
|
||||
Timeout Duration `json:"timeout,omitempty"`
|
||||
}
|
||||
|
||||
// Workstation defaults, applied in Normalise.
|
||||
const (
|
||||
DefaultWorkstationProbe = 15 * time.Second
|
||||
DefaultWorkstationTimeout = 90 * time.Second
|
||||
)
|
||||
|
||||
// SearchConfig — the self-hosted SearXNG instance she searches with.
|
||||
//
|
||||
// External search is allowed and off unless configured (CLAUDE.md). Configuring
|
||||
@@ -1233,6 +1279,12 @@ type PhraserConfig struct {
|
||||
NCtx int `json:"n_ctx,omitempty"`
|
||||
Timeout Duration `json:"timeout,omitempty"`
|
||||
|
||||
// CacheRAMMiB bounds llama-server's prompt cache. Omitted ⇒ 512 MiB, which
|
||||
// is what keeps the resident model near 1 GB of RSS instead of the 7.9 GB
|
||||
// measured on 2026-08-03. Set it to -1 to pass no flag at all and let the
|
||||
// server apply its own 8 GiB default. See phraser.Config.CacheRAMMiB.
|
||||
CacheRAMMiB int `json:"cache_ram_mib,omitempty"`
|
||||
|
||||
// LLMNudges — let the model word nudges again. Off by default: nudges are
|
||||
// worded from hand-written Russian templates now (the model broke the
|
||||
// persona and invented units). Chat, query and reminder phrasing always go
|
||||
@@ -1500,6 +1552,24 @@ func (c *Config) applyDefaults() {
|
||||
}
|
||||
}
|
||||
|
||||
// No address, no preferred model. An unconfigured workstation is the
|
||||
// default deploy and must be indistinguishable from today.
|
||||
if c.Workstation != nil && strings.TrimSpace(c.Workstation.URL) == "" {
|
||||
c.Workstation = nil
|
||||
}
|
||||
if c.Workstation != nil {
|
||||
w := c.Workstation
|
||||
if strings.TrimSpace(w.Health) == "" {
|
||||
w.Health = strings.TrimRight(w.URL, "/") + "/health"
|
||||
}
|
||||
if w.Probe <= 0 {
|
||||
w.Probe = Duration(DefaultWorkstationProbe)
|
||||
}
|
||||
if w.Timeout <= 0 {
|
||||
w.Timeout = Duration(DefaultWorkstationTimeout)
|
||||
}
|
||||
}
|
||||
|
||||
if c.Voice != nil {
|
||||
if c.Voice.RouterThreshold <= 0 {
|
||||
c.Voice.RouterThreshold = DefaultRouterThreshold
|
||||
@@ -1667,7 +1737,7 @@ func morningRoutinesFromConfig(mc []MorningRoutineConfig) []morning.Routine {
|
||||
for i, r := range mc {
|
||||
items := make([]morning.Item, len(r.Items))
|
||||
for j, it := range r.Items {
|
||||
items[j] = morning.Item{Key: it.Key, FactKey: it.FactKey, Label: it.Label}
|
||||
items[j] = morning.Item{Key: it.Key, FactKey: it.FactKey, Label: it.Label, Optional: it.Optional}
|
||||
}
|
||||
weekdays := make([]time.Weekday, len(r.Weekdays))
|
||||
for j, w := range r.Weekdays {
|
||||
|
||||
@@ -413,3 +413,56 @@ func TestNormaliseFillsKiwixDefaults(t *testing.T) {
|
||||
t.Error("rewrite: false was not honoured")
|
||||
}
|
||||
}
|
||||
|
||||
// A workstation with no address is not a workstation. The unconfigured deploy
|
||||
// must be indistinguishable from today, so the block is dropped rather than
|
||||
// left to fail one probe at a time.
|
||||
func TestNormaliseDropsAddresslessWorkstation(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
in *WorkstationConfig
|
||||
}{
|
||||
{"no url", &WorkstationConfig{Probe: Duration(time.Second)}},
|
||||
{"blank url", &WorkstationConfig{URL: " "}},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
c := &Config{Workstation: tc.in}
|
||||
c.applyDefaults()
|
||||
if c.Workstation != nil {
|
||||
t.Errorf("kept an unusable workstation block: %+v", c.Workstation)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The health endpoint defaults to the supervisor's, not llama-server's: mavgpud
|
||||
// answers 503 while the card is held, and that refusal is the whole signal.
|
||||
func TestNormaliseFillsWorkstationDefaults(t *testing.T) {
|
||||
c := &Config{Workstation: &WorkstationConfig{URL: "http://192.168.1.105:8080/"}}
|
||||
c.applyDefaults()
|
||||
if c.Workstation == nil {
|
||||
t.Fatal("dropped a usable workstation block")
|
||||
}
|
||||
if got, want := c.Workstation.Health, "http://192.168.1.105:8080/health"; got != want {
|
||||
t.Errorf("Health = %q, want %q", got, want)
|
||||
}
|
||||
if time.Duration(c.Workstation.Probe) != DefaultWorkstationProbe {
|
||||
t.Errorf("Probe = %s, want %s", time.Duration(c.Workstation.Probe), DefaultWorkstationProbe)
|
||||
}
|
||||
if time.Duration(c.Workstation.Timeout) != DefaultWorkstationTimeout {
|
||||
t.Errorf("Timeout = %s, want %s", time.Duration(c.Workstation.Timeout), DefaultWorkstationTimeout)
|
||||
}
|
||||
}
|
||||
|
||||
// An explicit health URL is left alone: the supervisor may sit behind something
|
||||
// that does not put /health at the root.
|
||||
func TestNormaliseKeepsExplicitWorkstationHealth(t *testing.T) {
|
||||
c := &Config{Workstation: &WorkstationConfig{
|
||||
URL: "http://192.168.1.105:8080",
|
||||
Health: "http://192.168.1.105:9000/ready",
|
||||
}}
|
||||
c.applyDefaults()
|
||||
if got, want := c.Workstation.Health, "http://192.168.1.105:9000/ready"; got != want {
|
||||
t.Errorf("Health = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,6 +57,14 @@ func (q *PendingQuestion) CanAsk() bool {
|
||||
|
||||
// ClarifyStore holds the parked questions. Same shape and locking as
|
||||
// SessionStore: keyed by dialogue id, expired entries dropped on read.
|
||||
//
|
||||
// Memory only, deliberately, unlike SessionStore — a restart expires every
|
||||
// parked question and she does not announce that it happened (Vikunja #385,
|
||||
// written down in docs/design.md). The 90s TTL and the attempt count measure a
|
||||
// pause in one conversation, and a restart is a gap of unknown length, so a
|
||||
// restored question would either be dead already or lying about its age. His
|
||||
// next words route fresh, which is the right answer with or without a notice.
|
||||
// Do not give this store a persister without re-arguing that.
|
||||
type ClarifyStore struct {
|
||||
mu sync.RWMutex
|
||||
questions map[string]*PendingQuestion
|
||||
|
||||
+35
-2
@@ -60,6 +60,19 @@ type Nudge struct {
|
||||
OutcomeTs *int64 `json:"outcome_ts,omitempty"`
|
||||
}
|
||||
|
||||
// DeliveryAttempt — one row of the delivery outbox. Times are formatted by the
|
||||
// reader; Completed is nil while the attempt is still pending.
|
||||
type DeliveryAttempt struct {
|
||||
ID int64 `json:"id"`
|
||||
Kind string `json:"kind"`
|
||||
Rule string `json:"rule,omitempty"`
|
||||
ReminderID int64 `json:"reminder_id,omitempty"`
|
||||
Channel string `json:"channel"`
|
||||
Status string `json:"status"`
|
||||
Created time.Time `json:"created"`
|
||||
Completed *time.Time `json:"completed,omitempty"`
|
||||
}
|
||||
|
||||
// Note — a recall/preference item; ranked by embedding cosine on query.
|
||||
// Score is set by QueryNotes (0 on the write path).
|
||||
type Note struct {
|
||||
@@ -521,6 +534,12 @@ type outcomesReq struct {
|
||||
type nReq struct {
|
||||
N int `json:"n"`
|
||||
}
|
||||
|
||||
// deliveryAttemptsReq — the outbox read. Status is empty for every status.
|
||||
type deliveryAttemptsReq struct {
|
||||
Status string `json:"status,omitempty"`
|
||||
N int `json:"n"`
|
||||
}
|
||||
type kindNReq struct {
|
||||
Kind string `json:"kind"`
|
||||
N int `json:"n"`
|
||||
@@ -593,8 +612,14 @@ type MCPServerStatus struct {
|
||||
}
|
||||
|
||||
// chatReq / chatResp — text chat round-trip for the IPC Chat method.
|
||||
//
|
||||
// Conversation names the thread this utterance belongs to: a mavweb session, a
|
||||
// telegram chat. It is opaque to the daemon and only has to be stable for one
|
||||
// conversation and distinct across them. Empty is allowed and means "the
|
||||
// unattributed text tap", which is what an old client sends.
|
||||
type chatReq struct {
|
||||
Text string `json:"text"`
|
||||
Text string `json:"text"`
|
||||
Conversation string `json:"conversation,omitempty"`
|
||||
}
|
||||
type chatResp struct {
|
||||
Reply string `json:"reply"`
|
||||
@@ -679,6 +704,9 @@ type CoreAPI interface {
|
||||
RecentActiveFactsByKind(ctx context.Context, kind string, n int) ([]Fact, error)
|
||||
CalendarEvents(ctx context.Context, from, to time.Time) ([]Fact, error)
|
||||
RecentNudges(ctx context.Context, n int) ([]Nudge, error)
|
||||
// DeliveryAttempts reads the outbox, newest first. An empty status means
|
||||
// every status (Vikunja #390).
|
||||
DeliveryAttempts(ctx context.Context, status string, n int) ([]DeliveryAttempt, error)
|
||||
|
||||
// RecentEcosystemTraces reads the ecosystem call log, which lives in its
|
||||
// own table so machine-rate traces never crowd out human-rate facts.
|
||||
@@ -759,7 +787,12 @@ type CoreAPI interface {
|
||||
// Chat routes a text utterance through the reactive handler's core path
|
||||
// (router → dialogue → action → replier) and returns the reply text.
|
||||
// No audio or stt/tts — for text channels (mavweb, telegram).
|
||||
Chat(ctx context.Context, text string) (string, error)
|
||||
//
|
||||
// conversation names the thread. A parked clarifying question is held per
|
||||
// conversation, so an unanswered question on one reach cannot eat the next
|
||||
// utterance from another (Vikunja #466). Empty means the unattributed text
|
||||
// tap and is still one conversation of its own, separate from the mic.
|
||||
Chat(ctx context.Context, conversation, text string) (string, error)
|
||||
|
||||
// RecentEvents returns the daemon's unified intake journal, newest first
|
||||
// (Vikunja #283) — one envelope per thing that arrived, whatever direction
|
||||
|
||||
+11
-2
@@ -68,6 +68,7 @@ var readOnlyMethods = map[Method]bool{
|
||||
MethodRecentActiveFacts: true,
|
||||
MethodCalendarEvents: true,
|
||||
MethodRecentNudges: true,
|
||||
MethodDeliveryAttempts: true,
|
||||
MethodRecentEcoTraces: true,
|
||||
MethodQueryNotes: true,
|
||||
MethodRecentNotes: true,
|
||||
@@ -373,6 +374,14 @@ func (c *Client) RecentEcosystemTraces(ctx context.Context, n int) ([]EcosystemT
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *Client) DeliveryAttempts(ctx context.Context, status string, n int) ([]DeliveryAttempt, error) {
|
||||
var out []DeliveryAttempt
|
||||
if err := c.call(ctx, MethodDeliveryAttempts, deliveryAttemptsReq{Status: status, N: n}, &out); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *Client) RecentNudges(ctx context.Context, n int) ([]Nudge, error) {
|
||||
var out []Nudge
|
||||
if err := c.call(ctx, MethodRecentNudges, nReq{N: n}, &out); err != nil {
|
||||
@@ -623,9 +632,9 @@ func (c *Client) AcceptProposedRoutine(ctx context.Context, id int64) error {
|
||||
return c.call(ctx, MethodAcceptProposedRoutine, acceptProposedRoutineReq{ID: id}, nil)
|
||||
}
|
||||
|
||||
func (c *Client) Chat(ctx context.Context, text string) (string, error) {
|
||||
func (c *Client) Chat(ctx context.Context, conversation, text string) (string, error) {
|
||||
var r chatResp
|
||||
if err := c.call(ctx, MethodChat, chatReq{Text: text}, &r); err != nil {
|
||||
if err := c.call(ctx, MethodChat, chatReq{Text: text, Conversation: conversation}, &r); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return r.Reply, nil
|
||||
|
||||
@@ -401,7 +401,7 @@ func TestChatViaClient(t *testing.T) {
|
||||
}
|
||||
t.Cleanup(func() { _ = cli.Close() })
|
||||
|
||||
reply, err := cli.Chat(context.Background(), "привет")
|
||||
reply, err := cli.Chat(context.Background(), "web", "привет")
|
||||
if err != nil {
|
||||
t.Fatalf("Chat: %v", err)
|
||||
}
|
||||
@@ -417,7 +417,7 @@ type chatTestAPI struct {
|
||||
UnimplementedCoreAPI
|
||||
}
|
||||
|
||||
func (a *chatTestAPI) Chat(ctx context.Context, text string) (string, error) {
|
||||
func (a *chatTestAPI) Chat(ctx context.Context, _, text string) (string, error) {
|
||||
if text == "привет" {
|
||||
return "и тебе привет!", nil
|
||||
}
|
||||
|
||||
+31
-2
@@ -173,6 +173,25 @@ func (a *storeAPI) RecentNudges(ctx context.Context, n int) ([]Nudge, error) {
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (a *storeAPI) DeliveryAttempts(ctx context.Context, status string, n int) ([]DeliveryAttempt, error) {
|
||||
as, err := a.s.ListDeliveryAttempts(ctx, status, n)
|
||||
if err != nil {
|
||||
return nil, mapErr(err)
|
||||
}
|
||||
out := make([]DeliveryAttempt, len(as))
|
||||
for i, at := range as {
|
||||
out[i] = DeliveryAttempt{
|
||||
ID: at.ID, Kind: at.Kind, Rule: at.Rule, ReminderID: at.ReminderID,
|
||||
Channel: at.Channel, Status: at.Status, Created: at.Created,
|
||||
}
|
||||
if at.HasComplete {
|
||||
t := at.Completed
|
||||
out[i].Completed = &t
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (a *storeAPI) WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error) {
|
||||
id, err := a.s.WriteNote(ctx, ts, text, embedding, source)
|
||||
return id, mapErr(err)
|
||||
@@ -240,7 +259,7 @@ func (a *storeAPI) RevertFact(ctx context.Context, key string) (int64, error) {
|
||||
return newID, mapErr(err)
|
||||
}
|
||||
|
||||
func (a *storeAPI) Chat(ctx context.Context, text string) (string, error) {
|
||||
func (a *storeAPI) Chat(ctx context.Context, conversation, text string) (string, error) {
|
||||
return "", errors.New("store: chat not available via direct store API")
|
||||
}
|
||||
|
||||
@@ -863,6 +882,16 @@ var methodTable = map[Method]handlerFunc{
|
||||
}
|
||||
return out, nil
|
||||
}),
|
||||
MethodDeliveryAttempts: withParams(func(ctx context.Context, api CoreAPI, p deliveryAttemptsReq) ([]DeliveryAttempt, error) {
|
||||
out, err := api.DeliveryAttempts(ctx, p.Status, p.N)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if out == nil {
|
||||
out = []DeliveryAttempt{}
|
||||
}
|
||||
return out, nil
|
||||
}),
|
||||
MethodRecentNudges: withParams(func(ctx context.Context, api CoreAPI, p nReq) ([]Nudge, error) {
|
||||
out, err := api.RecentNudges(ctx, p.N)
|
||||
if err != nil {
|
||||
@@ -974,7 +1003,7 @@ var methodTable = map[Method]handlerFunc{
|
||||
return map[string]int64{"new_id": newID}, nil
|
||||
}),
|
||||
MethodChat: withParams(func(ctx context.Context, api CoreAPI, p chatReq) (chatResp, error) {
|
||||
reply, err := api.Chat(ctx, p.Text)
|
||||
reply, err := api.Chat(ctx, p.Conversation, p.Text)
|
||||
return chatResp{Reply: reply}, err
|
||||
}),
|
||||
MethodTickTrace: withoutParams(func(ctx context.Context, api CoreAPI) (TickTrace, error) {
|
||||
|
||||
@@ -68,6 +68,9 @@ func (UnimplementedCoreAPI) RecentActiveFactsByKind(ctx context.Context, kind st
|
||||
func (UnimplementedCoreAPI) CalendarEvents(ctx context.Context, from, to time.Time) ([]Fact, error) {
|
||||
return nil, ErrNotImplemented
|
||||
}
|
||||
func (UnimplementedCoreAPI) DeliveryAttempts(ctx context.Context, status string, n int) ([]DeliveryAttempt, error) {
|
||||
return nil, ErrNotImplemented
|
||||
}
|
||||
func (UnimplementedCoreAPI) RecentNudges(ctx context.Context, n int) ([]Nudge, error) {
|
||||
return nil, ErrNotImplemented
|
||||
}
|
||||
@@ -141,6 +144,6 @@ func (UnimplementedCoreAPI) MCPServers(ctx context.Context) ([]MCPServerStatus,
|
||||
func (UnimplementedCoreAPI) DayPlan(ctx context.Context) (DayPlan, error) {
|
||||
return DayPlan{}, ErrNotImplemented
|
||||
}
|
||||
func (UnimplementedCoreAPI) Chat(ctx context.Context, text string) (string, error) {
|
||||
func (UnimplementedCoreAPI) Chat(ctx context.Context, conversation, text string) (string, error) {
|
||||
return "", ErrNotImplemented
|
||||
}
|
||||
|
||||
@@ -28,6 +28,7 @@ const (
|
||||
MethodRecentActiveFacts Method = "recent_active_facts_by_kind"
|
||||
MethodCalendarEvents Method = "calendar_events"
|
||||
MethodRecentNudges Method = "recent_nudges"
|
||||
MethodDeliveryAttempts Method = "delivery_attempts"
|
||||
MethodRecentEcoTraces Method = "recent_ecosystem_traces"
|
||||
MethodWriteNote Method = "write_note"
|
||||
MethodQueryNotes Method = "query_notes"
|
||||
|
||||
@@ -129,6 +129,11 @@ type Req struct {
|
||||
RepeatPenalty float64
|
||||
// Stop — sequences that end generation early (e.g. newline for a one-liner).
|
||||
Stop []string
|
||||
// Temperature — 0 (the zero value) is greedy decoding, and greedy is what
|
||||
// every caller here wanted before this field existed. It is set only by the
|
||||
// phraser, whose own transport has always sampled at 0.7: routing a phrasing
|
||||
// call through this client must not quietly change how it decodes.
|
||||
Temperature float64
|
||||
}
|
||||
|
||||
type msg struct {
|
||||
@@ -176,7 +181,7 @@ func (c *Client) Complete(ctx context.Context, r Req) (string, error) {
|
||||
Messages: []msg{{Role: "system", Content: r.System}, {Role: "user", Content: r.User}},
|
||||
MaxTokens: r.MaxTokens,
|
||||
Grammar: r.Grammar,
|
||||
Temp: 0,
|
||||
Temp: r.Temperature,
|
||||
RepeatPenalty: r.RepeatPenalty,
|
||||
Stop: r.Stop,
|
||||
})
|
||||
|
||||
@@ -0,0 +1,193 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Pair — a preferred model on another host, with the resident one as the floor.
|
||||
//
|
||||
// homesrv cannot grow a GPU and the workstation has 16GB of VRAM, so the big
|
||||
// model runs there and the resident Qwen3-1.7B stays here. See docs/offload.md.
|
||||
// The workstation is never assumed up: its GPU is often busy with CPT runs and
|
||||
// the manga-recap pipeline, and the machine sleeps. So the remote is preferred,
|
||||
// never required, and Pair is what makes "preferred" mean something precise.
|
||||
//
|
||||
// This is admission control, not a scheduler. There is no arbiter deciding who
|
||||
// gets the card. A prober asks the remote whether it will take work, caches the
|
||||
// answer, and every request reads that cached answer in nanoseconds. Routing
|
||||
// sits on the hot path at p50 825ms and must never wait on a machine that may
|
||||
// be asleep, so no request ever pays for a health check itself.
|
||||
//
|
||||
// Pair satisfies nothing by itself. Callers pick a method by which half of the
|
||||
// degradation rule they live under:
|
||||
//
|
||||
// - Complete falls back silently. For routing, replies, and nudge phrasing,
|
||||
// where the big model is only better and the 1.7B is today's shipping
|
||||
// quality. He is not told which model phrased his reply.
|
||||
// - CompleteRemote returns ErrRemoteUnavailable instead of falling back. For
|
||||
// a world question, or a long Kiwix or search passage, where a 1.7B
|
||||
// confabulates rather than summarises. A named gap beats an invented
|
||||
// answer.
|
||||
type Pair struct {
|
||||
remote *Client
|
||||
floor *Client
|
||||
|
||||
// up — the cached admission answer, written only by the prober goroutine
|
||||
// and read by every request. Atomic so the read costs nanoseconds and no
|
||||
// request ever contends with the prober.
|
||||
up atomic.Bool
|
||||
|
||||
health string
|
||||
interval time.Duration
|
||||
http *http.Client
|
||||
stop chan struct{}
|
||||
}
|
||||
|
||||
// ErrRemoteUnavailable — the workstation model was required and is not
|
||||
// answering. Callers on the naming half of the degradation rule turn this into
|
||||
// a gap in the reply ("не могу сейчас"), never into a guess from the floor.
|
||||
var ErrRemoteUnavailable = errors.New("llm: workstation model unavailable")
|
||||
|
||||
// ErrNoFloor — a Pair was built with no resident model to fall back to. A
|
||||
// configuration mistake: the floor is the whole point.
|
||||
var ErrNoFloor = errors.New("llm: no floor client")
|
||||
|
||||
// NewPair builds the two-model arrangement. remote may be nil, which is the
|
||||
// unconfigured deploy and must behave exactly as the box behaves today: every
|
||||
// call goes to the floor and nothing probes anything.
|
||||
//
|
||||
// health is the URL the prober asks. llama-server's /health answers "is a model
|
||||
// loaded and ready", which is the useful signal here, because llama-server
|
||||
// refuses to load at all when VRAM is short. That makes a busy card detectable
|
||||
// without any cooperation from the owner's other jobs.
|
||||
func NewPair(remote, floor *Client, health string, interval time.Duration) *Pair {
|
||||
p := &Pair{
|
||||
remote: remote,
|
||||
floor: floor,
|
||||
health: health,
|
||||
interval: interval,
|
||||
http: &http.Client{Timeout: probeTimeout},
|
||||
stop: make(chan struct{}),
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
// probeTimeout — a remote that cannot answer /health this fast is not going to
|
||||
// serve a turn either. Short on purpose: the prober runs on its own goroutine,
|
||||
// but a slow probe still delays the moment Maven notices the card came back.
|
||||
const probeTimeout = 2 * time.Second
|
||||
|
||||
// Start begins probing. It returns immediately, and the first probe runs before
|
||||
// the first tick so a remote that is already up is used on the first turn
|
||||
// rather than after one interval of falling back. Safe to call with a nil
|
||||
// remote; it does nothing.
|
||||
func (p *Pair) Start(ctx context.Context) {
|
||||
if p.remote == nil || p.health == "" {
|
||||
return
|
||||
}
|
||||
go func() {
|
||||
p.probe(ctx)
|
||||
t := time.NewTicker(p.interval)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-p.stop:
|
||||
return
|
||||
case <-t.C:
|
||||
p.probe(ctx)
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// Stop ends the prober. Idempotent.
|
||||
func (p *Pair) Stop() {
|
||||
select {
|
||||
case <-p.stop:
|
||||
default:
|
||||
close(p.stop)
|
||||
}
|
||||
}
|
||||
|
||||
// Available reports whether the workstation will take work right now. It reads
|
||||
// a cached flag, so it is safe to call per turn on the hot path. A false answer
|
||||
// is never stale in the direction that matters: the worst case is that Maven
|
||||
// falls back for up to one probe interval after the card frees up.
|
||||
func (p *Pair) Available() bool {
|
||||
return p.remote != nil && p.up.Load()
|
||||
}
|
||||
|
||||
func (p *Pair) probe(ctx context.Context) {
|
||||
ctx, cancel := context.WithTimeout(ctx, probeTimeout)
|
||||
defer cancel()
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, p.health, nil)
|
||||
if err != nil {
|
||||
p.set(false)
|
||||
return
|
||||
}
|
||||
resp, err := p.http.Do(req)
|
||||
if err != nil {
|
||||
p.set(false)
|
||||
return
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
p.set(resp.StatusCode == http.StatusOK)
|
||||
}
|
||||
|
||||
// set records the admission answer and logs only the transitions. A machine
|
||||
// that sleeps every night would otherwise write one line per interval forever.
|
||||
func (p *Pair) set(up bool) {
|
||||
if p.up.Swap(up) == up {
|
||||
return
|
||||
}
|
||||
if up {
|
||||
log.Printf("llm: workstation model available at %s", p.health)
|
||||
} else {
|
||||
log.Printf("llm: workstation model unavailable, falling back to the resident model")
|
||||
}
|
||||
}
|
||||
|
||||
// Complete runs r on the workstation when it will take work, and on the
|
||||
// resident model otherwise. A remote that fails mid-request falls back too: the
|
||||
// admission answer is a cache and can be one interval out of date, so an error
|
||||
// here is expected rather than exceptional.
|
||||
//
|
||||
// This is the silent half of the degradation rule. It must be indistinguishable
|
||||
// from today's behaviour when the workstation is down.
|
||||
func (p *Pair) Complete(ctx context.Context, r Req) (string, error) {
|
||||
if p.floor == nil {
|
||||
return "", ErrNoFloor
|
||||
}
|
||||
if p.Available() {
|
||||
out, err := p.remote.Complete(ctx, r)
|
||||
if err == nil {
|
||||
return out, nil
|
||||
}
|
||||
// The cached answer was wrong. Correct it now rather than sending the
|
||||
// next request into the same hole, then fall back.
|
||||
p.set(false)
|
||||
}
|
||||
return p.floor.Complete(ctx, r)
|
||||
}
|
||||
|
||||
// CompleteRemote runs r on the workstation or refuses. It never falls back,
|
||||
// because for a world question the resident 1.7B does not answer worse, it
|
||||
// invents. Callers turn ErrRemoteUnavailable into a named gap.
|
||||
func (p *Pair) CompleteRemote(ctx context.Context, r Req) (string, error) {
|
||||
if !p.Available() {
|
||||
return "", ErrRemoteUnavailable
|
||||
}
|
||||
out, err := p.remote.Complete(ctx, r)
|
||||
if err != nil {
|
||||
p.set(false)
|
||||
return "", errors.Join(ErrRemoteUnavailable, err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,210 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// completionServer stands in for a llama-server. It counts what reached it, so
|
||||
// a test can say which of the two models answered.
|
||||
func completionServer(t *testing.T, reply string, hits *atomic.Int64) *httptest.Server {
|
||||
t.Helper()
|
||||
s := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hits.Add(1)
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"` + reply + `"}}]}`))
|
||||
}))
|
||||
t.Cleanup(s.Close)
|
||||
return s
|
||||
}
|
||||
|
||||
func healthServer(t *testing.T, ok *atomic.Bool) *httptest.Server {
|
||||
t.Helper()
|
||||
s := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if !ok.Load() {
|
||||
w.WriteHeader(http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}))
|
||||
t.Cleanup(s.Close)
|
||||
return s
|
||||
}
|
||||
|
||||
// waitFor polls until cond holds or the deadline passes. The prober runs on its
|
||||
// own goroutine, so a test has to wait for it rather than assume it has run.
|
||||
func waitFor(t *testing.T, cond func() bool) bool {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if cond() {
|
||||
return true
|
||||
}
|
||||
time.Sleep(5 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// The unconfigured deploy. No remote, no probing, every call to the floor —
|
||||
// exactly what the box does today.
|
||||
func TestNoRemoteGoesToTheFloor(t *testing.T) {
|
||||
var floorHits atomic.Int64
|
||||
floor := completionServer(t, "floor", &floorHits)
|
||||
|
||||
p := NewPair(nil, New(floor.URL, time.Second), "", time.Second)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
|
||||
if p.Available() {
|
||||
t.Fatal("a Pair with no remote reports available")
|
||||
}
|
||||
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
||||
if err != nil {
|
||||
t.Fatalf("complete: %v", err)
|
||||
}
|
||||
if out != "floor" || floorHits.Load() != 1 {
|
||||
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// The workstation is up, so it answers and the resident model is not touched.
|
||||
func TestAvailableRemoteAnswers(t *testing.T) {
|
||||
var remoteHits, floorHits atomic.Int64
|
||||
remote := completionServer(t, "remote", &remoteHits)
|
||||
floor := completionServer(t, "floor", &floorHits)
|
||||
up := &atomic.Bool{}
|
||||
up.Store(true)
|
||||
health := healthServer(t, up)
|
||||
|
||||
p := NewPair(New(remote.URL, time.Second), New(floor.URL, time.Second), health.URL, 20*time.Millisecond)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
if !waitFor(t, p.Available) {
|
||||
t.Fatal("prober never saw the remote come up")
|
||||
}
|
||||
|
||||
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
||||
if err != nil {
|
||||
t.Fatalf("complete: %v", err)
|
||||
}
|
||||
if out != "remote" || floorHits.Load() != 0 {
|
||||
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// The card is busy, so /health refuses and Complete degrades silently. This is
|
||||
// the constraint from 483: the workstation being down is indistinguishable from
|
||||
// today's behaviour.
|
||||
func TestBusyCardFallsBackSilently(t *testing.T) {
|
||||
var remoteHits, floorHits atomic.Int64
|
||||
remote := completionServer(t, "remote", &remoteHits)
|
||||
floor := completionServer(t, "floor", &floorHits)
|
||||
health := healthServer(t, &atomic.Bool{}) // never ok
|
||||
|
||||
p := NewPair(New(remote.URL, time.Second), New(floor.URL, time.Second), health.URL, 20*time.Millisecond)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
time.Sleep(60 * time.Millisecond)
|
||||
|
||||
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
||||
if err != nil {
|
||||
t.Fatalf("complete: %v", err)
|
||||
}
|
||||
if out != "floor" || remoteHits.Load() != 0 {
|
||||
t.Fatalf("out = %q, remote hits = %d", out, remoteHits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// The cached admission answer can be one interval out of date, so a remote that
|
||||
// dies between probes must still not break the turn.
|
||||
func TestRemoteErrorMidRequestFallsBack(t *testing.T) {
|
||||
var floorHits atomic.Int64
|
||||
dead := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusInternalServerError)
|
||||
}))
|
||||
defer dead.Close()
|
||||
floor := completionServer(t, "floor", &floorHits)
|
||||
up := &atomic.Bool{}
|
||||
up.Store(true)
|
||||
health := healthServer(t, up)
|
||||
|
||||
p := NewPair(New(dead.URL, time.Second), New(floor.URL, time.Second), health.URL, time.Hour)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
if !waitFor(t, p.Available) {
|
||||
t.Fatal("prober never saw the remote come up")
|
||||
}
|
||||
|
||||
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
||||
if err != nil {
|
||||
t.Fatalf("complete: %v", err)
|
||||
}
|
||||
if out != "floor" || floorHits.Load() != 1 {
|
||||
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
||||
}
|
||||
// The failed request must have corrected the cached answer, so the next
|
||||
// one does not walk into the same hole.
|
||||
if p.Available() {
|
||||
t.Fatal("a failed remote request left the admission answer up")
|
||||
}
|
||||
}
|
||||
|
||||
// The naming half of the degradation rule. A world question must not be handed
|
||||
// to the resident model, because it answers by inventing.
|
||||
func TestCompleteRemoteNamesTheGap(t *testing.T) {
|
||||
var floorHits atomic.Int64
|
||||
floor := completionServer(t, "floor", &floorHits)
|
||||
health := healthServer(t, &atomic.Bool{}) // never ok
|
||||
|
||||
p := NewPair(New("http://127.0.0.1:1", time.Second), New(floor.URL, time.Second), health.URL, 20*time.Millisecond)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
time.Sleep(60 * time.Millisecond)
|
||||
|
||||
if _, err := p.CompleteRemote(context.Background(), Req{User: "почему небо голубое"}); !errors.Is(err, ErrRemoteUnavailable) {
|
||||
t.Fatalf("err = %v, want ErrRemoteUnavailable", err)
|
||||
}
|
||||
if floorHits.Load() != 0 {
|
||||
t.Fatalf("CompleteRemote fell back to the floor %d times", floorHits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// Routing sits on the hot path and must never pay for a health check. Available
|
||||
// reads a cached flag, so it costs no network at all.
|
||||
func TestAvailableDoesNotProbe(t *testing.T) {
|
||||
var probes atomic.Int64
|
||||
health := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
probes.Add(1)
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}))
|
||||
defer health.Close()
|
||||
|
||||
p := NewPair(New("http://127.0.0.1:1", time.Second), New("http://127.0.0.1:1", time.Second), health.URL, time.Hour)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
if !waitFor(t, p.Available) {
|
||||
t.Fatal("prober never ran")
|
||||
}
|
||||
|
||||
before := probes.Load()
|
||||
for range 1000 {
|
||||
p.Available()
|
||||
}
|
||||
if got := probes.Load(); got != before {
|
||||
t.Fatalf("1000 Available calls made %d probes", got-before)
|
||||
}
|
||||
}
|
||||
|
||||
// A Pair with no floor is a configuration mistake, and it must say so rather
|
||||
// than silently having nowhere to degrade to.
|
||||
func TestNoFloorIsAnError(t *testing.T) {
|
||||
p := NewPair(nil, nil, "", time.Second)
|
||||
if _, err := p.Complete(context.Background(), Req{User: "привет"}); !errors.Is(err, ErrNoFloor) {
|
||||
t.Fatalf("err = %v, want ErrNoFloor", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,134 @@
|
||||
package memory
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
// stopwords — words that carry no topic. A question and a note that share only
|
||||
// these share nothing: "почему небо синее" and "сеть какая-то медленная" both
|
||||
// contain "какая"-shaped filler and are about different worlds.
|
||||
var stopwords = map[string]bool{
|
||||
// interrogatives and demonstratives
|
||||
"что": true, "чего": true, "какой": true, "какая": true, "какое": true,
|
||||
"какие": true, "каких": true, "кто": true, "кого": true, "кому": true,
|
||||
"почему": true, "зачем": true, "где": true, "куда": true, "откуда": true,
|
||||
"когда": true, "сколько": true, "как": true, "то": true, "это": true,
|
||||
"этот": true, "тот": true, "там": true, "тут": true, "такой": true,
|
||||
// pronouns — every sentence he says is about him, so "я" is not a topic
|
||||
"я": true, "меня": true, "мне": true, "мой": true, "моя": true, "мои": true,
|
||||
"ты": true, "тебя": true, "тебе": true, "твой": true, "он": true, "она": true,
|
||||
"они": true, "мы": true, "себя": true, "свой": true,
|
||||
// prepositions, conjunctions, particles, copulas
|
||||
"в": true, "во": true, "на": true, "с": true, "со": true, "у": true,
|
||||
"о": true, "об": true, "про": true, "за": true, "из": true, "по": true,
|
||||
"до": true, "от": true, "для": true, "над": true, "под": true, "при": true,
|
||||
"и": true, "а": true, "но": true, "или": true, "же": true, "ли": true,
|
||||
"не": true, "ни": true, "бы": true, "был": true, "была": true, "было": true,
|
||||
"быть": true, "есть": true, "был-ли": true, "уже": true, "ещё": true,
|
||||
"еще": true, "так": true, "вот": true, "там-же": true,
|
||||
// English filler, for the mixed utterances he does say
|
||||
"the": true, "a": true, "an": true, "is": true, "are": true, "was": true,
|
||||
"were": true, "be": true, "of": true, "in": true, "on": true, "at": true,
|
||||
"to": true, "for": true, "about": true, "and": true, "or": true, "not": true,
|
||||
"what": true, "who": true, "why": true, "when": true, "where": true,
|
||||
"which": true, "how": true, "i": true, "my": true, "me": true, "it": true,
|
||||
"this": true, "that": true,
|
||||
}
|
||||
|
||||
// firstPerson — the words that make an utterance a question about his own
|
||||
// life. Not possession only: "как я восстановил конфиги" owns nothing and is
|
||||
// still about him.
|
||||
var firstPerson = map[string]bool{
|
||||
"я": true, "меня": true, "мне": true, "мной": true, "мой": true,
|
||||
"моя": true, "моё": true, "мое": true, "мои": true, "моего": true,
|
||||
"моей": true, "моих": true, "моим": true, "себя": true, "свой": true,
|
||||
"своя": true, "свои": true, "своего": true, "мною": true,
|
||||
"i": true, "me": true, "my": true, "mine": true, "myself": true,
|
||||
}
|
||||
|
||||
// RecallAllowed is the second half of the recall gate (#470). A hit that
|
||||
// cleared the score and margin gate may still be about something else
|
||||
// entirely: the held-out fixture puts the right note at 0.791-0.890 and the
|
||||
// must-be-silent cases at 0.795-0.835, so no threshold sits between them, and
|
||||
// a note about his slow network answered "почему небо синее?".
|
||||
//
|
||||
// The veto applies only to a question that mentions nothing of his. That
|
||||
// restriction is what keeps the fix from costing more than it saves: recall
|
||||
// exists to find the note whose words he no longer remembers, and demanding a
|
||||
// shared word of every recall silenced four true recalls on the fixture to
|
||||
// kill one false one. A question about his own life keeps the embedder alone
|
||||
// as its judge. A question about the world has to name something the memory
|
||||
// actually mentions.
|
||||
//
|
||||
// The veto's price was re-measured on 2026-08-03 (#496,
|
||||
// docs/evals/2026-08-03-recall-topic-veto.md). It costs one true recall and
|
||||
// buys one false one, and the fixture pass count is the same either way. The
|
||||
// lost case is an English paraphrase, not the cross-language loss it was
|
||||
// reported as, and the fixture has no cross-language case at all. Do not add a
|
||||
// script test or a bilingual stem map for it — both are no-ops here. The
|
||||
// separating signal is semantic and belongs in a reranker, not in this file.
|
||||
func RecallAllowed(query, text string) bool {
|
||||
if mentionsHim(query) {
|
||||
return true
|
||||
}
|
||||
return SharesContentWord(query, text)
|
||||
}
|
||||
|
||||
func mentionsHim(query string) bool {
|
||||
for _, w := range strings.FieldsFunc(strings.ToLower(query), func(r rune) bool {
|
||||
return !unicode.IsLetter(r) && !unicode.IsDigit(r)
|
||||
}) {
|
||||
if firstPerson[w] {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// SharesContentWord reports whether query and text have at least one topic
|
||||
// word in common, after dropping the words that carry no topic. Stems are
|
||||
// compared, so the note and the question do not have to inflect alike.
|
||||
func SharesContentWord(query, text string) bool {
|
||||
q := contentWords(query)
|
||||
if len(q) == 0 {
|
||||
// Nothing to compare — a question made entirely of filler. The score
|
||||
// gate is then the only judge it can have.
|
||||
return true
|
||||
}
|
||||
t := contentWords(text)
|
||||
for _, a := range q {
|
||||
for _, b := range t {
|
||||
if a == b || sameStem(a, b) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func contentWords(s string) []string {
|
||||
var out []string
|
||||
for _, w := range strings.FieldsFunc(strings.ToLower(s), func(r rune) bool {
|
||||
return !unicode.IsLetter(r) && !unicode.IsDigit(r)
|
||||
}) {
|
||||
if !stopwords[w] {
|
||||
out = append(out, w)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// sameStem is inflection and derivation tolerance: Russian marks case and
|
||||
// tense on the ending, and the note and the question rarely use the same form.
|
||||
// "воду" and "вода" are the same water, "кормить" and "корм" the same feeding.
|
||||
// All but the last rune of the shorter word must match, and never fewer than
|
||||
// three, which is what keeps "сеть" clear of "сеанс".
|
||||
func sameStem(a, b string) bool {
|
||||
ar, br := []rune(a), []rune(b)
|
||||
n := min(len(ar), len(br)) - 1
|
||||
if n < 3 {
|
||||
return false
|
||||
}
|
||||
return string(ar[:n]) == string(br[:n])
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
package memory
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestRecallAllowed(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
query, text string
|
||||
want bool
|
||||
}{
|
||||
// The #470 shape: a world question and a note about his box.
|
||||
{"world question, unrelated note", "почему небо синее", "сеть какая-то медленная", false},
|
||||
{"world question, unrelated fact", "какая столица Франции", "какая последняя версия языка Go", false},
|
||||
{"silent fixture case", "во сколько отходит поезд", "бэкап запускается в три ночи", false},
|
||||
|
||||
// A world question that does name the topic keeps its answer.
|
||||
{"world question, same topic", "какой поезд идёт в Минск", "поезда в Минск ходят утром", true},
|
||||
|
||||
// A question about his own life is judged by the embedder alone,
|
||||
// because recall exists for words he no longer remembers.
|
||||
{"about him, no shared word", "во сколько я обычно засыпаю", "ложусь около одиннадцати", true},
|
||||
{"about him, english", "which colour scheme do i like", "тёмная тема везде", true},
|
||||
|
||||
// Inflection must not break a match.
|
||||
{"inflected", "чем кормить кота", "корм для кота в шкафу", true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := RecallAllowed(c.query, c.text); got != c.want {
|
||||
t.Errorf("%s: RecallAllowed(%q, %q) = %v, want %v", c.name, c.query, c.text, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The known cost of the veto and the thing that pays for it, both measured on
|
||||
// the held-out fixture with the real embedder (#496,
|
||||
// docs/evals/2026-08-03-recall-topic-veto.md). The two are one lexical class:
|
||||
// zero shared content words, no first-person marker, scores 0.826 against 0.835
|
||||
// and margins 0.023 against 0.019. Recovering the first re-admits the second,
|
||||
// which puts false recall back to 1/5. Anyone loosening the veto has to move
|
||||
// the first line without moving the second.
|
||||
func TestRecallVetoTradeIsPinned(t *testing.T) {
|
||||
if RecallAllowed("what fixed the screen problem", "the flicker went away once i swapped the display cable") {
|
||||
t.Error("en-hard-024 is expected to stay vetoed — if this passes now, re-measure false recall before celebrating")
|
||||
}
|
||||
if RecallAllowed("во сколько отходит поезд", "погулял вдоль реки") {
|
||||
t.Error("ru-silent-029 must stay vetoed — this is the false recall the veto exists to stop")
|
||||
}
|
||||
}
|
||||
|
||||
// A question made only of filler has no topic word to match on, and the score
|
||||
// gate is then the only judge it can have.
|
||||
func TestRecallAllowedFallsBackWhenNothingToCompare(t *testing.T) {
|
||||
if !RecallAllowed("что это", "сеть какая-то медленная") {
|
||||
t.Error("a question with no content word must not be vetoed")
|
||||
}
|
||||
}
|
||||
@@ -90,9 +90,44 @@ func Load() (Fixture, error) {
|
||||
if len(f.Cases) == 0 {
|
||||
return Fixture{}, fmt.Errorf("fixture has no cases")
|
||||
}
|
||||
if err := checkIDs(f); err != nil {
|
||||
return Fixture{}, err
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// checkIDs refuses a fixture where a case note and a filler note share an id.
|
||||
//
|
||||
// Every case is scored over its own notes plus the whole filler set, and the
|
||||
// two stores disagree about what a repeated id means: the sqlite store upserts
|
||||
// on it, the in-memory store appends. So one collision makes a case score
|
||||
// differently on the two backends, and it reads as an embedder or gate
|
||||
// difference, which is the one thing this harness exists to measure (Vikunja
|
||||
// #386). It was dodged once by hand during #373 by renaming two ids.
|
||||
//
|
||||
// Checked in Load rather than in the test, so every caller of the fixture is
|
||||
// covered and not only the one that remembers to look.
|
||||
func checkIDs(f Fixture) error {
|
||||
filler := make(map[string]bool, len(f.Filler))
|
||||
for _, n := range f.Filler {
|
||||
if n.ID == "" {
|
||||
return fmt.Errorf("filler note with an empty id")
|
||||
}
|
||||
if filler[n.ID] {
|
||||
return fmt.Errorf("duplicate filler note id %q", n.ID)
|
||||
}
|
||||
filler[n.ID] = true
|
||||
}
|
||||
for _, c := range f.Cases {
|
||||
for _, n := range c.Notes {
|
||||
if filler[n.ID] {
|
||||
return fmt.Errorf("case %s: note id %q collides with a filler note", c.ID, n.ID)
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// NewStore builds an empty store for one case, plus a function to release it.
|
||||
// A factory rather than a store because every case needs a clean index — notes
|
||||
// from case A must not be visible to case B's query.
|
||||
@@ -378,7 +413,7 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
|
||||
if len(hits) > 1 {
|
||||
o.Margin = hits[0].Score - hits[1].Score
|
||||
}
|
||||
o.Recalled = bestRecall(hits, minScore, minMargin)
|
||||
o.Recalled = bestRecall(c.Query, hits, minScore, minMargin)
|
||||
}
|
||||
for i, h := range hits {
|
||||
if h.ID != c.Want {
|
||||
@@ -424,11 +459,19 @@ func rankNote(inTop3 bool) string {
|
||||
// is not importable; recalleval_test.go asserts the two agree in behaviour.
|
||||
// The daemon returns the whole hit (a note and a fact are said differently);
|
||||
// the harness only scores what came back, so it keeps returning the text.
|
||||
func bestRecall(results []memory.Result, minScore, minMargin float64) string {
|
||||
// bestRecall mirrors the daemon's gate in cmd/mavend/recall.go, including the
|
||||
// topic veto added for #470: a score that clears the gate still has to be
|
||||
// about what he asked. Keep the two in step — a fixture that measures a
|
||||
// weaker gate than the daemon runs flatters it.
|
||||
func bestRecall(query string, results []memory.Result, minScore, minMargin float64) string {
|
||||
if !memory.Confident(results, minScore, minMargin) {
|
||||
return ""
|
||||
}
|
||||
return results[0].Meta["text"]
|
||||
text := results[0].Meta["text"]
|
||||
if !memory.RecallAllowed(query, text) {
|
||||
return ""
|
||||
}
|
||||
return text
|
||||
}
|
||||
|
||||
func bump(m map[string]TagStat, key string, pass bool) {
|
||||
|
||||
@@ -140,21 +140,22 @@ func words(s string) []string {
|
||||
|
||||
// TestBestRecallMatchesDaemon — the harness duplicates bestRecall from
|
||||
// cmd/mavend/recall.go (package main is not importable). This pins the copy to
|
||||
// the original's three rules: no hits, below the gate, or no text ⇒ silence.
|
||||
// the original's rules: no hits, below the gate, no text, or no shared topic
|
||||
// word ⇒ silence.
|
||||
func TestBestRecallMatchesDaemon(t *testing.T) {
|
||||
if got := bestRecall(nil, 0.55, 0); got != "" {
|
||||
if got := bestRecall("чай", nil, 0.55, 0); got != "" {
|
||||
t.Errorf("no hits: got %q, want silence", got)
|
||||
}
|
||||
low := []memory.Result{{ID: "a", Score: 0.4, Meta: map[string]string{"text": "чай"}}}
|
||||
if got := bestRecall(low, 0.55, 0); got != "" {
|
||||
if got := bestRecall("чай", low, 0.55, 0); got != "" {
|
||||
t.Errorf("below gate: got %q, want silence", got)
|
||||
}
|
||||
noText := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{}}}
|
||||
if got := bestRecall(noText, 0.55, 0); got != "" {
|
||||
if got := bestRecall("чай", noText, 0.55, 0); got != "" {
|
||||
t.Errorf("no text: got %q, want silence", got)
|
||||
}
|
||||
ok := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{"text": "чай"}}}
|
||||
if got := bestRecall(ok, 0.55, 0); got != "чай" {
|
||||
if got := bestRecall("чай", ok, 0.55, 0); got != "чай" {
|
||||
t.Errorf("above gate: got %q, want %q", got, "чай")
|
||||
}
|
||||
// Margin: a close runner-up means the embedder cannot tell the two apart,
|
||||
@@ -163,17 +164,23 @@ func TestBestRecallMatchesDaemon(t *testing.T) {
|
||||
{ID: "a", Score: 0.86, Meta: map[string]string{"text": "чай"}},
|
||||
{ID: "b", Score: 0.85, Meta: map[string]string{"text": "кофе"}},
|
||||
}
|
||||
if got := bestRecall(close, 0.55, 0.03); got != "" {
|
||||
if got := bestRecall("чай", close, 0.55, 0.03); got != "" {
|
||||
t.Errorf("thin margin: got %q, want silence", got)
|
||||
}
|
||||
if got := bestRecall(close, 0.55, 0); got != "чай" {
|
||||
if got := bestRecall("чай", close, 0.55, 0); got != "чай" {
|
||||
t.Errorf("margin off: got %q, want %q", got, "чай")
|
||||
}
|
||||
// The topic veto (#470): the score is fine and the note is about
|
||||
// something else.
|
||||
offTopic := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{"text": "сеть какая-то медленная"}}}
|
||||
if got := bestRecall("почему небо синее", offTopic, 0.55, 0); got != "" {
|
||||
t.Errorf("off topic: got %q, want silence", got)
|
||||
}
|
||||
clear := []memory.Result{
|
||||
{ID: "a", Score: 0.86, Meta: map[string]string{"text": "чай"}},
|
||||
{ID: "b", Score: 0.70, Meta: map[string]string{"text": "кофе"}},
|
||||
}
|
||||
if got := bestRecall(clear, 0.55, 0.03); got != "чай" {
|
||||
if got := bestRecall("чай", clear, 0.55, 0.03); got != "чай" {
|
||||
t.Errorf("wide margin: got %q, want %q", got, "чай")
|
||||
}
|
||||
}
|
||||
@@ -330,3 +337,28 @@ func marginSweep(t *testing.T, emb router.Embedder, f Fixture) string {
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// TestFillerIDCollisionIsRefused — the guard that keeps a fixture edit from
|
||||
// looking like a backend difference (Vikunja #386).
|
||||
func TestFillerIDCollisionIsRefused(t *testing.T) {
|
||||
f := Fixture{
|
||||
SchemaVersion: SchemaVersion,
|
||||
Cases: []Case{{ID: "ru-001", Notes: []StoredNote{{ID: "f1", Text: "..."}}}},
|
||||
Filler: []StoredNote{{ID: "f1", Text: "..."}},
|
||||
}
|
||||
if err := checkIDs(f); err == nil {
|
||||
t.Fatal("a case note reusing a filler id must be refused")
|
||||
}
|
||||
f.Filler = append(f.Filler, StoredNote{ID: "f1", Text: "..."})
|
||||
if err := checkIDs(Fixture{SchemaVersion: SchemaVersion, Filler: f.Filler}); err == nil {
|
||||
t.Fatal("a duplicate filler id must be refused")
|
||||
}
|
||||
ok := Fixture{
|
||||
SchemaVersion: SchemaVersion,
|
||||
Cases: []Case{{ID: "ru-001", Notes: []StoredNote{{ID: "n1", Text: "..."}}}},
|
||||
Filler: []StoredNote{{ID: "f1", Text: "..."}},
|
||||
}
|
||||
if err := checkIDs(ok); err != nil {
|
||||
t.Fatalf("a clean fixture must pass: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,6 +30,18 @@ type Item struct {
|
||||
Key string
|
||||
FactKey string
|
||||
Label string // RU text surfaced when this item is still missing.
|
||||
// Optional — a missing one is not worth a nudge on its own.
|
||||
//
|
||||
// Every item was implicitly required until 04-08-2026, because there was
|
||||
// no field, so a skipped stretch read exactly like skipped medication and
|
||||
// #280's first behaviour could not hold (Vikunja #473). A checklist where
|
||||
// everything is mandatory is a checklist he learns to ignore.
|
||||
//
|
||||
// It changes two things and nothing else: an all-optional routine never
|
||||
// nudges, and a nudge that does fire names the optional stragglers after
|
||||
// the required ones, in softer words. Evidence, the window and the day
|
||||
// plan treat both kinds alike — a missing optional item is still missing.
|
||||
Optional bool
|
||||
}
|
||||
|
||||
// Routine — one daily checklist. WindowStart/WindowEnd are "HH:MM" local
|
||||
@@ -60,12 +72,37 @@ type Status struct {
|
||||
}
|
||||
|
||||
// Candidate — a routine that's due for its one-per-day nag: the window has
|
||||
// reached NudgeAt and at least one item is still unevidenced.
|
||||
// reached NudgeAt and at least one REQUIRED item is still unevidenced. Missing
|
||||
// carries the optional stragglers too, so the one message she is allowed per
|
||||
// day per routine can mention them; they never cause it.
|
||||
type Candidate struct {
|
||||
Routine Routine
|
||||
Missing []Item
|
||||
}
|
||||
|
||||
// Required reports the missing items that are not optional. The nudge fires on
|
||||
// these; the rest ride along.
|
||||
func Required(missing []Item) []Item {
|
||||
var out []Item
|
||||
for _, it := range missing {
|
||||
if !it.Optional {
|
||||
out = append(out, it)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// OptionalOnly is the other half of Required.
|
||||
func OptionalOnly(missing []Item) []Item {
|
||||
var out []Item
|
||||
for _, it := range missing {
|
||||
if it.Optional {
|
||||
out = append(out, it)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Validate reports the first structural problem with a routine set: missing
|
||||
// name/items, an unparseable HH:MM, an inverted window, a duplicate item key
|
||||
// within a routine, or an out-of-range weekday. Called at config load so a
|
||||
@@ -191,7 +228,11 @@ func Due(routines []Routine, facts map[string]store.Fact, last map[string]time.T
|
||||
missing = append(missing, it)
|
||||
}
|
||||
}
|
||||
if len(missing) == 0 {
|
||||
// A day where only the optional items were skipped is a fine day, and
|
||||
// nagging about it is what teaches him to stop listening (Vikunja
|
||||
// #473). The optional ones still travel in Missing so the message can
|
||||
// mention them when it is being sent anyway.
|
||||
if len(Required(missing)) == 0 {
|
||||
continue
|
||||
}
|
||||
if prev, seen := last[r.Name]; seen && sameDay(prev, now) {
|
||||
|
||||
@@ -182,3 +182,40 @@ func TestDueRespectsExplicitNudgeAt(t *testing.T) {
|
||||
t.Fatalf("expected candidate at explicit nudge_at, got %d", len(out))
|
||||
}
|
||||
}
|
||||
|
||||
// TestOptionalItemsDoNotEarnANudge — behaviour 1 of #280, which could not hold
|
||||
// while every item was implicitly required (Vikunja #473).
|
||||
func TestOptionalItemsDoNotEarnANudge(t *testing.T) {
|
||||
r := Routine{
|
||||
Name: "утро",
|
||||
WindowStart: "07:00",
|
||||
WindowEnd: "10:00",
|
||||
Items: []Item{
|
||||
{Key: "meds", FactKey: "meds", Label: "таблетки"},
|
||||
{Key: "stretch", FactKey: "stretch", Label: "растяжка", Optional: true},
|
||||
},
|
||||
}
|
||||
now := time.Date(2026, 8, 4, 10, 0, 0, 0, time.UTC)
|
||||
took := map[string]store.Fact{"meds": {Key: "meds", Ts: now.Add(-2 * time.Hour)}}
|
||||
|
||||
// Only the stretch was skipped: nothing to say.
|
||||
if due := Due([]Routine{r}, took, map[string]time.Time{}, now); len(due) != 0 {
|
||||
t.Fatalf("an optional item alone must not nudge, got %+v", due)
|
||||
}
|
||||
// The medication was skipped: she says so, and mentions the stretch too.
|
||||
due := Due([]Routine{r}, map[string]store.Fact{}, map[string]time.Time{}, now)
|
||||
if len(due) != 1 {
|
||||
t.Fatalf("a missing required item must nudge, got %+v", due)
|
||||
}
|
||||
if got := Required(due[0].Missing); len(got) != 1 || got[0].Key != "meds" {
|
||||
t.Fatalf("Required = %+v, want the meds item alone", got)
|
||||
}
|
||||
if got := OptionalOnly(due[0].Missing); len(got) != 1 || got[0].Key != "stretch" {
|
||||
t.Fatalf("OptionalOnly = %+v, want the stretch item alone", got)
|
||||
}
|
||||
// The window still reports it as missing — optional is not invisible.
|
||||
st := Evaluate(r, map[string]store.Fact{}, now.Add(-time.Hour))
|
||||
if len(st.Missing) != 2 {
|
||||
t.Fatalf("Evaluate must still list both, got %+v", st.Missing)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -53,9 +53,31 @@ const MinOnPatternFraction = 0.7
|
||||
// a repeat. False negatives cost one more observation and nothing else.
|
||||
const MinEvents = 4
|
||||
|
||||
// MinIntervalDays — the fastest rhythm that may be called a routine. Two
|
||||
// hours.
|
||||
//
|
||||
// Without a floor, four taps of the same key minutes apart give intervals near
|
||||
// 0.002 days. They all sit inside the ±50% band by construction, so the
|
||||
// detector proposed a routine and PhraseRoutine worded it as "каждый день"
|
||||
// (Vikunja #468). The damage outlives the mistake: UNIQUE(action, object)
|
||||
// means dismissing the bogus proposal burns that pair permanently, so the real
|
||||
// routine behind it can never be proposed again.
|
||||
//
|
||||
// Two hours rather than a day, because a genuine habit can run several times a
|
||||
// day — meals, water, a break. Anything faster than that is not a habit she
|
||||
// should be proposing to remind him about; the loop rules already cover that
|
||||
// range, and they are rules, not guesses. It is checked against the median, so
|
||||
// one quick repeat inside a real rhythm still counts.
|
||||
//
|
||||
// The other half of this is that hand-QA of the detector was unsafe: seeding a
|
||||
// pattern the obvious way, four chat turns in a row, poisoned the very pair
|
||||
// being tested.
|
||||
const MinIntervalDays = 2.0 / 24.0
|
||||
|
||||
// Detect checks whether a sequence of events for the same action+object
|
||||
// forms a stable recurring pattern. Returns a ProposedRoutine when:
|
||||
// - At least MinEvents events exist (≥3 intervals)
|
||||
// - The median interval is at least MinIntervalDays
|
||||
// - At least MinOnPatternFraction of the intervals sit within
|
||||
// MaxIntervalRatio of the median interval
|
||||
//
|
||||
@@ -88,8 +110,8 @@ func Detect(events []Event) (*ProposedRoutine, error) {
|
||||
}
|
||||
|
||||
center := medianFloat(intervals)
|
||||
if center <= 0 {
|
||||
return nil, nil
|
||||
if center <= 0 || center < MinIntervalDays {
|
||||
return nil, nil // a burst, not a rhythm — see MinIntervalDays
|
||||
}
|
||||
|
||||
// Keep the intervals that sit inside the band around the median. The
|
||||
|
||||
@@ -216,3 +216,46 @@ func TestDetectMedianBandNotExtremes(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A burst is not a habit. Four taps of the same key minutes apart give
|
||||
// intervals near 0.002 days, all inside the ±50% band by construction, so the
|
||||
// detector called it a daily routine (Vikunja #468). Dismissing that proposal
|
||||
// burns the action+object pair permanently, which also made hand-QA of the
|
||||
// detector unsafe.
|
||||
func TestDetectRejectsABurst(t *testing.T) {
|
||||
base := time.Date(2026, 8, 4, 9, 0, 0, 0, time.UTC)
|
||||
var events []Event
|
||||
for i := 0; i < 4; i++ {
|
||||
events = append(events, Event{
|
||||
Action: "refill", Object: "cat_water",
|
||||
Ts: base.Add(time.Duration(i) * 7 * time.Minute),
|
||||
})
|
||||
}
|
||||
r, err := Detect(events)
|
||||
if err != nil {
|
||||
t.Fatalf("Detect: %v", err)
|
||||
}
|
||||
if r != nil {
|
||||
t.Fatalf("four taps minutes apart proposed a routine every %.3f days", r.IntervalDays)
|
||||
}
|
||||
}
|
||||
|
||||
// The floor is two hours, not a day: a habit that runs several times a day is
|
||||
// still a habit.
|
||||
func TestDetectKeepsASeveralTimesADayHabit(t *testing.T) {
|
||||
base := time.Date(2026, 8, 4, 8, 0, 0, 0, time.UTC)
|
||||
var events []Event
|
||||
for i := 0; i < 5; i++ {
|
||||
events = append(events, Event{
|
||||
Action: "drink", Object: "water",
|
||||
Ts: base.Add(time.Duration(i) * 4 * time.Hour),
|
||||
})
|
||||
}
|
||||
r, err := Detect(events)
|
||||
if err != nil {
|
||||
t.Fatalf("Detect: %v", err)
|
||||
}
|
||||
if r == nil {
|
||||
t.Fatal("a four-hour rhythm over five events is a habit, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -173,12 +173,23 @@ func checkFeminine(body string) Result {
|
||||
// Second pass: self-reference with the pronoun dropped — "напомнил тебе",
|
||||
// "проверил за тебя". A masculine past-tense verb whose object is HIM can
|
||||
// only be her speaking about herself.
|
||||
//
|
||||
// Two guards, both from a false positive on the talk fixture: "ты заплатил
|
||||
// за домен до марта" scored as her drift and cost the run a point it had
|
||||
// earned (Vikunja #462). He is male, so a past-tense verb governed by "ты"
|
||||
// must be masculine. And a bare "за" is not evidence of anything — "за
|
||||
// домен" is a price, "за тебя" is her doing something on his behalf — so it
|
||||
// only counts when he is the one it points at.
|
||||
for i, w := range words {
|
||||
if !masculinePast(w) || i+1 >= len(words) {
|
||||
if !masculinePast(w) || i+1 >= len(words) || governedByYou(words, i) {
|
||||
continue
|
||||
}
|
||||
next := words[i+1]
|
||||
if next == "тебе" || next == "тебя" || next == "за" {
|
||||
aboutHim := next == "тебе" || next == "тебя"
|
||||
if next == "за" && i+2 < len(words) && (words[i+2] == "тебя" || words[i+2] == "тебе") {
|
||||
aboutHim = true
|
||||
}
|
||||
if aboutHim {
|
||||
return Result{CheckFeminine, false,
|
||||
fmt.Sprintf("masculine self-reference %q before %q", w, next)}
|
||||
}
|
||||
@@ -652,3 +663,19 @@ func checkEllipsis(body string) Result {
|
||||
}
|
||||
return Result{CheckEllipsis, true, ""}
|
||||
}
|
||||
|
||||
// governedByYou reports whether "ты" stands close enough in front of the verb
|
||||
// at index i to be its subject. Three words, the same window checkFeminine's
|
||||
// first pass uses after "я", and it stops at a first-person pronoun so "ты
|
||||
// просил, я напомнил" still trips.
|
||||
func governedByYou(words []string, i int) bool {
|
||||
for j := i - 1; j >= 0 && j >= i-3; j-- {
|
||||
switch words[j] {
|
||||
case "ты":
|
||||
return true
|
||||
case "я":
|
||||
return false
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -106,6 +106,12 @@ func TestChecksCatchWhatTheyClaim(t *testing.T) {
|
||||
{"masculine predicative", "я должен сказать: попей воды.", CheckFeminine},
|
||||
// The other direction: HE is male, so second-person masculine is right.
|
||||
{"second person masculine ok", "ты не пил воду четыре часа.", ""},
|
||||
// The recorded false positive: "заплатил" sits before "за", and the
|
||||
// second pass read that as her dropping the pronoun. The subject is
|
||||
// "ты" and he is male, so the reply is right (Vikunja #462).
|
||||
{"second person masculine before за", "ты заплатил за домен до марта, а воду пить всё равно надо.", ""},
|
||||
// The same shape she really does get wrong still trips.
|
||||
{"masculine on his behalf", "проверил за тебя — воды не было четыре часа.", CheckFeminine},
|
||||
// The real observed failure: she addressed him as a woman.
|
||||
{"feminine second person", "ты давно не отдыхала — попей воды.", CheckHisGender},
|
||||
{"feminine second person no dash", "ты пила воду четыре часа назад.", CheckHisGender},
|
||||
|
||||
@@ -26,20 +26,22 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/dialogue"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
//go:embed talk_v1.json
|
||||
var talkFixtureJSON []byte
|
||||
|
||||
// The three phrasing paths under test. Values match the fixture's "path" field.
|
||||
// The phrasing paths under test. Values match the fixture's "path" field.
|
||||
const (
|
||||
PathChat = "chat" // PhraseChat
|
||||
PathQuery = "query" // PhraseQuery with notes
|
||||
PathKnowledge = "knowledge" // PhraseQuery with no notes
|
||||
PathReply = "reply" // PhraseReply, the reactive confirmation
|
||||
)
|
||||
|
||||
// TalkPaths — report order.
|
||||
var TalkPaths = []string{PathChat, PathQuery, PathKnowledge}
|
||||
var TalkPaths = []string{PathChat, PathQuery, PathKnowledge, PathReply}
|
||||
|
||||
// TalkCheckNames — the checks that apply to a free-form reply, in report order.
|
||||
// Deliberately a subset of CheckNames: length, mood and "no questions" are nudge
|
||||
@@ -58,12 +60,19 @@ var TalkCheckNames = []string{
|
||||
// WantAny is the on-topic contract: at least one lowercased fragment must appear
|
||||
// in the reply. Fragments are stems ("пароль" → "парол") so declension does not
|
||||
// defeat them.
|
||||
//
|
||||
// Intent, Key and Value carry the reply path's decision: that path is phrased
|
||||
// from what the router already resolved, not from the raw utterance. Utterance
|
||||
// stays filled anyway, because it is what a human reads in the report.
|
||||
type TalkCase struct {
|
||||
ID string `json:"id"`
|
||||
Path string `json:"path"`
|
||||
Utterance string `json:"utterance"`
|
||||
History []string `json:"history,omitempty"`
|
||||
Notes []string `json:"notes,omitempty"`
|
||||
Intent string `json:"intent,omitempty"`
|
||||
Key string `json:"key,omitempty"`
|
||||
Value string `json:"value,omitempty"`
|
||||
WantAny []string `json:"want_any"`
|
||||
Tags []string `json:"tags,omitempty"`
|
||||
Note string `json:"note,omitempty"`
|
||||
@@ -92,13 +101,27 @@ func LoadTalk() (TalkFixture, error) {
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// Talker — the two methods a conversational path must have to be scorable.
|
||||
// *phraser.LLMPhraser satisfies it; same trick as Nudger.
|
||||
// Talker — the methods a conversational path must have to be scorable.
|
||||
// *phraser.LLMPhraser satisfies the first two; *phraser.Replier satisfies the
|
||||
// third, so a run that scores all four paths passes a Pair.
|
||||
type Talker interface {
|
||||
PhraseChat(ctx context.Context, utterance string, history []dialogue.Turn) (string, error)
|
||||
PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error)
|
||||
}
|
||||
|
||||
// Confirmer — the reply path. *phraser.Replier satisfies it.
|
||||
type Confirmer interface {
|
||||
PhraseReply(ctx context.Context, d router.Decision) (string, error)
|
||||
}
|
||||
|
||||
// Pair joins the two objects the daemon wires separately — the phraser and the
|
||||
// replier — so one ScoreTalk call covers every path Maven speaks through. A bare
|
||||
// Talker still works; its reply cases score as errors, which is honest.
|
||||
type Pair struct {
|
||||
Talker
|
||||
Confirmer
|
||||
}
|
||||
|
||||
// TalkOutcome — one scored case.
|
||||
type TalkOutcome struct {
|
||||
Case TalkCase
|
||||
@@ -194,10 +217,30 @@ func (c TalkCase) run(ctx context.Context, t Talker) (string, error) {
|
||||
return t.PhraseQuery(ctx, c.Utterance, c.Notes)
|
||||
case PathKnowledge:
|
||||
return t.PhraseQuery(ctx, c.Utterance, nil)
|
||||
case PathReply:
|
||||
conf, ok := t.(Confirmer)
|
||||
if !ok {
|
||||
return "", fmt.Errorf("target cannot phrase replies — pass a Pair")
|
||||
}
|
||||
return conf.PhraseReply(ctx, c.decision())
|
||||
}
|
||||
return "", fmt.Errorf("unknown path %q", c.Path)
|
||||
}
|
||||
|
||||
// decision rebuilds what the router would have handed the replier. Text is the
|
||||
// utterance for a note or a reminder, which is what the router puts there.
|
||||
func (c TalkCase) decision() router.Decision {
|
||||
return router.Decision{
|
||||
Intent: router.Intent(c.Intent),
|
||||
Slots: router.Slots{
|
||||
Key: c.Key,
|
||||
Value: c.Value,
|
||||
Text: c.Utterance,
|
||||
HasKey: c.Key != "",
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func (c TalkCase) turns() []dialogue.Turn {
|
||||
turns := make([]dialogue.Turn, 0, len(c.History))
|
||||
for _, h := range c.History {
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/persona"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// perPathMinimum — the resolution floor. A per-path score built on a handful of
|
||||
@@ -36,6 +37,10 @@ func TestTalkFixture(t *testing.T) {
|
||||
|
||||
switch c.Path {
|
||||
case PathChat, PathQuery, PathKnowledge:
|
||||
case PathReply:
|
||||
if c.Intent == "" {
|
||||
t.Errorf("%s: reply case has no intent — the replier is phrased from the decision", c.ID)
|
||||
}
|
||||
default:
|
||||
t.Errorf("%s: unknown path %q", c.ID, c.Path)
|
||||
}
|
||||
@@ -69,10 +74,15 @@ type fakeTalker struct{ reply string }
|
||||
func (f fakeTalker) PhraseChat(context.Context, string, []dialogue.Turn) (string, error) {
|
||||
return f.reply, nil
|
||||
}
|
||||
|
||||
func (f fakeTalker) PhraseQuery(context.Context, string, []string) (string, error) {
|
||||
return f.reply, nil
|
||||
}
|
||||
|
||||
func (f fakeTalker) PhraseReply(context.Context, router.Decision) (string, error) {
|
||||
return f.reply, nil
|
||||
}
|
||||
|
||||
// TestScoreTalkCounts — a reply that fails on purpose must be counted on every
|
||||
// path, so a real run cannot report a hidden zero.
|
||||
func TestScoreTalkCounts(t *testing.T) {
|
||||
@@ -104,7 +114,7 @@ func TestScoreTalkCounts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestLLMTalkBaseline — the resident model on the three conversational paths.
|
||||
// TestLLMTalkBaseline — the resident model on all four phrasing paths.
|
||||
// Opt-in exactly like TestLLMPhrasingBaseline: CI has no model and a run costs
|
||||
// minutes on the CPU target.
|
||||
//
|
||||
@@ -148,7 +158,12 @@ func TestLLMTalkBaseline(t *testing.T) {
|
||||
}
|
||||
t.Logf("scoring model %s at %s", model, base)
|
||||
|
||||
rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", p, f)
|
||||
// The reply path is a separate object in the daemon too: the phraser owns its
|
||||
// own llama-server, the replier is handed an llm.Client. Pair scores both.
|
||||
block := func() string { return persona.Facts{}.Block(time.Now()) }
|
||||
target := Pair{Talker: p, Confirmer: phraser.NewReplier(llm.New(base, cfg.Timeout), block)}
|
||||
|
||||
rep, err := ScoreTalk(ctx, "llm ("+model+", built-in persona)", target, f)
|
||||
if err != nil {
|
||||
t.Fatalf("ScoreTalk: %v", err)
|
||||
}
|
||||
|
||||
@@ -222,6 +222,90 @@
|
||||
"utterance": "почему гром слышно позже молнии?",
|
||||
"want_any": ["звук", "све", "быстр", "гром", "молни"],
|
||||
"tags": ["general"]
|
||||
},
|
||||
{
|
||||
"id": "reply-fact-coffee",
|
||||
"path": "reply",
|
||||
"intent": "fact",
|
||||
"key": "кофе",
|
||||
"value": "закончился",
|
||||
"utterance": "кофе закончился",
|
||||
"want_any": ["коф"],
|
||||
"tags": ["fact"],
|
||||
"note": "The plainest confirmation there is, and the sentence he hears most often."
|
||||
},
|
||||
{
|
||||
"id": "reply-fact-weight",
|
||||
"path": "reply",
|
||||
"intent": "fact",
|
||||
"key": "вес",
|
||||
"value": "82",
|
||||
"utterance": "мой вес 82",
|
||||
"want_any": ["вес", "82"],
|
||||
"tags": ["fact", "number"],
|
||||
"note": "A number must survive into the confirmation; a paraphrase that drops it is useless."
|
||||
},
|
||||
{
|
||||
"id": "reply-fact-pill",
|
||||
"path": "reply",
|
||||
"intent": "fact",
|
||||
"key": "таблетки",
|
||||
"value": "выпил",
|
||||
"utterance": "таблетки выпил",
|
||||
"want_any": ["таблетк"],
|
||||
"tags": ["fact", "feminine"],
|
||||
"note": "He says 'выпил', masculine and about himself. She must not copy the form onto herself."
|
||||
},
|
||||
{
|
||||
"id": "reply-note-router",
|
||||
"path": "reply",
|
||||
"intent": "note",
|
||||
"utterance": "роутер перезагружается сам по ночам",
|
||||
"want_any": ["роутер"],
|
||||
"tags": ["note"]
|
||||
},
|
||||
{
|
||||
"id": "reply-note-long",
|
||||
"path": "reply",
|
||||
"intent": "note",
|
||||
"utterance": "если диск снова отвалится, посмотреть кабель, а не контроллер, в прошлый раз был кабель",
|
||||
"want_any": ["диск", "кабел"],
|
||||
"tags": ["note", "length"],
|
||||
"note": "A long note baits a long confirmation. One sentence is the contract."
|
||||
},
|
||||
{
|
||||
"id": "reply-reminder-evening",
|
||||
"path": "reply",
|
||||
"intent": "reminder",
|
||||
"utterance": "напомни вечером полить цветы",
|
||||
"want_any": ["цвет", "полит", "вечер"],
|
||||
"tags": ["reminder"]
|
||||
},
|
||||
{
|
||||
"id": "reply-reminder-tomorrow",
|
||||
"path": "reply",
|
||||
"intent": "reminder",
|
||||
"utterance": "напомни завтра позвонить в поликлинику",
|
||||
"want_any": ["поликлиник", "позвон", "звон"],
|
||||
"tags": ["reminder"]
|
||||
},
|
||||
{
|
||||
"id": "reply-formality-bait",
|
||||
"path": "reply",
|
||||
"intent": "note",
|
||||
"utterance": "запишите пожалуйста что счётчики я сдал",
|
||||
"want_any": ["счётчик", "счетчик"],
|
||||
"tags": ["note", "persona-bait", "address"],
|
||||
"note": "Polite plural in the input. The confirmation must still be на ты."
|
||||
},
|
||||
{
|
||||
"id": "reply-question-bait",
|
||||
"path": "reply",
|
||||
"intent": "note",
|
||||
"utterance": "надо купить фильтр для воды, не помню какой",
|
||||
"want_any": ["фильтр"],
|
||||
"tags": ["note", "no-question"],
|
||||
"note": "An unresolved note invites her to ask which filter. A confirmation does not ask."
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
+148
-43
@@ -1,6 +1,7 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
@@ -8,6 +9,7 @@ import (
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strings"
|
||||
@@ -46,6 +48,12 @@ type LLMPhraser struct {
|
||||
launch func(ctx context.Context, cfg Config) (backend, error)
|
||||
probe func(ctx context.Context, base string) (string, error)
|
||||
|
||||
// remote — the workstation model, when one is configured. Set once at wiring
|
||||
// time by UseRemote and read on every phrasing call. nil ⇒ every call goes to
|
||||
// the resident llama-server this phraser owns, which is the whole deploy
|
||||
// before a `workstation` block exists. See world.go.
|
||||
remote Remote
|
||||
|
||||
// swapMu — single-flight around Swap. Held for the whole swap, including the
|
||||
// model load, so two concurrent swap requests can never both be loading.
|
||||
swapMu sync.Mutex
|
||||
@@ -77,6 +85,19 @@ type Config struct {
|
||||
NCtx int
|
||||
Timeout time.Duration
|
||||
|
||||
// CacheRAMMiB bounds llama-server's prompt cache, which is what actually ate
|
||||
// this box. Measured on homesrv 2026-08-03: the server's own default limit is
|
||||
// 8192 MiB, it stores the full KV state of every idle slot it evicts (112 kiB
|
||||
// per token, so 166 MiB for one 1521-token prompt), and RSS climbed by that
|
||||
// much per distinct prompt until it hit 7.9 GB and half a gigabyte went to
|
||||
// swap. Weights are only 1.1 GB and mmapped, and -ngl 99 costs almost no RSS
|
||||
// because RADV keeps device memory outside the process.
|
||||
//
|
||||
// 0 ⇒ the flag is not passed and the server's own 8 GiB default applies. That
|
||||
// is the escape hatch for a llama-server too old to know --cache-ram, not a
|
||||
// recommendation. See docs/evals/2026-08-03-llama-prompt-cache.md.
|
||||
CacheRAMMiB int
|
||||
|
||||
// ContextBlock renders the shared context block (who he is, how to
|
||||
// address him, the time) fresh for each turn. See internal/persona.
|
||||
// nil ⇒ no block, the prompts stand alone.
|
||||
@@ -110,7 +131,9 @@ func DefaultConfig(modelPath string) Config {
|
||||
Listen: "127.0.0.1:0",
|
||||
NGpuLayers: -1,
|
||||
NCtx: 2048,
|
||||
Timeout: 30 * time.Second,
|
||||
// 512 MiB caps total RSS near 1 GB and still holds several recent prompts.
|
||||
CacheRAMMiB: 512,
|
||||
Timeout: 30 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -219,8 +242,10 @@ func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
|
||||
return p, nil
|
||||
}
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
p := &llamaProc{}
|
||||
// llamaArgs is the command line for one resident server. It is a function and
|
||||
// not an inline literal because kill-maven.sh's orphan sweep matches against
|
||||
// this exact line, and a test pins the two together.
|
||||
func llamaArgs(cfg Config) []string {
|
||||
args := []string{
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
@@ -229,7 +254,15 @@ func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, args...)
|
||||
if cfg.CacheRAMMiB > 0 {
|
||||
args = append(args, "--cache-ram", fmt.Sprintf("%d", cfg.CacheRAMMiB))
|
||||
}
|
||||
return args
|
||||
}
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
p := &llamaProc{}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...)
|
||||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||||
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
|
||||
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
|
||||
@@ -240,63 +273,104 @@ func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true, Pdeathsig: syscall.SIGKILL}
|
||||
p.cmd = cmd
|
||||
|
||||
stderr, err := cmd.StderrPipe()
|
||||
// One pipe for both streams. llama.cpp writes its buffer sizes, KV-cache
|
||||
// layout and offload lines to stderr and its request log to stdout, and
|
||||
// stdout used to go nowhere at all — so nothing about the model's memory was
|
||||
// diagnosable from a running box. Both ends land in mavend's log now.
|
||||
pr, pw, err := os.Pipe()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("llm: stderr pipe: %w", err)
|
||||
return nil, fmt.Errorf("llm: output pipe: %w", err)
|
||||
}
|
||||
cmd.Stdout = pw
|
||||
cmd.Stderr = pw
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
stderr.Close()
|
||||
pr.Close()
|
||||
pw.Close()
|
||||
return nil, fmt.Errorf("llm: start: %w", err)
|
||||
}
|
||||
// The child holds the only other reference to the write end. Dropping ours
|
||||
// is what makes the reader see EOF when the child dies.
|
||||
pw.Close()
|
||||
|
||||
portCh := make(chan string, 1)
|
||||
errCh := make(chan error, 1)
|
||||
tail := &lineTail{}
|
||||
p.wg.Add(1)
|
||||
go func() {
|
||||
defer p.wg.Done()
|
||||
buf := make([]byte, 4096)
|
||||
var leftover []byte
|
||||
for {
|
||||
n, err := stderr.Read(buf)
|
||||
if n > 0 {
|
||||
data := append(leftover, buf[:n]...)
|
||||
lines := bytes.Split(data, []byte("\n"))
|
||||
for _, line := range lines[:len(lines)-1] {
|
||||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||||
addr := string(m[1])
|
||||
portCh <- addr
|
||||
close(portCh)
|
||||
}
|
||||
defer pr.Close()
|
||||
sc := bufio.NewScanner(pr)
|
||||
// llama.cpp prints one prompt per line and a prompt can be long.
|
||||
sc.Buffer(make([]byte, 0, 64*1024), 1024*1024)
|
||||
listening := false
|
||||
for sc.Scan() {
|
||||
line := sc.Bytes()
|
||||
log.Printf("llama: %s", line)
|
||||
if !listening {
|
||||
tail.add(string(line))
|
||||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||||
listening = true
|
||||
portCh <- string(m[1])
|
||||
close(portCh)
|
||||
}
|
||||
leftover = lines[len(lines)-1]
|
||||
}
|
||||
if err != nil {
|
||||
errCh <- err
|
||||
return
|
||||
}
|
||||
}
|
||||
err := sc.Err()
|
||||
if err == nil {
|
||||
err = io.EOF
|
||||
}
|
||||
errCh <- err
|
||||
}()
|
||||
|
||||
fail := func(err error) (*llamaProc, error) {
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, err
|
||||
}
|
||||
select {
|
||||
case addr := <-portCh:
|
||||
p.base = addr
|
||||
return p, nil
|
||||
case err := <-errCh:
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, fmt.Errorf("llm: server output: %w", err)
|
||||
// The tail is the whole diagnosis when the server dies during load: bare
|
||||
// "EOF" never said which layer or which allocation it choked on.
|
||||
return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String()))
|
||||
case <-ctx.Done():
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, ctx.Err()
|
||||
return fail(ctx.Err())
|
||||
case <-time.After(60 * time.Second):
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, fmt.Errorf("llm: server did not start within 60s")
|
||||
return fail(fmt.Errorf("llm: server did not start within 60s; last output: %s", tail.String()))
|
||||
}
|
||||
}
|
||||
|
||||
// lineTail keeps the last few startup lines so a server that dies before it
|
||||
// listens can say why in the error, not just "EOF". Written by the reader
|
||||
// goroutine and read by whoever gives up on startup, so it takes a lock.
|
||||
type lineTail struct {
|
||||
mu sync.Mutex
|
||||
lines []string
|
||||
}
|
||||
|
||||
const lineTailMax = 12
|
||||
|
||||
func (t *lineTail) add(line string) {
|
||||
t.mu.Lock()
|
||||
defer t.mu.Unlock()
|
||||
t.lines = append(t.lines, line)
|
||||
if len(t.lines) > lineTailMax {
|
||||
t.lines = t.lines[len(t.lines)-lineTailMax:]
|
||||
}
|
||||
}
|
||||
|
||||
func (t *lineTail) String() string {
|
||||
t.mu.Lock()
|
||||
defer t.mu.Unlock()
|
||||
if len(t.lines) == 0 {
|
||||
return "(no output)"
|
||||
}
|
||||
return strings.Join(t.lines, " | ")
|
||||
}
|
||||
|
||||
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
||||
// the model is swapped, so callers that cache it must register an observer
|
||||
// (OnSwap) rather than keeping the string forever.
|
||||
@@ -363,10 +437,7 @@ func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []
|
||||
// prompt guaranteed to make a small model fill the gap from memory.
|
||||
notes = nonEmpty(notes)
|
||||
if len(notes) == 0 {
|
||||
// General knowledge — no notes to ground the answer. The system
|
||||
// prompt is the single tested source in router.KnowledgePrompt.
|
||||
sys := persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt())
|
||||
prompt := fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance)
|
||||
sys, prompt := p.knowledgePrompt(utterance)
|
||||
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
|
||||
if err != nil || resp == "" {
|
||||
return "не знаю.", nil
|
||||
@@ -381,11 +452,7 @@ func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
sys := p.querySystemPrompt()
|
||||
prompt := fmt.Sprintf(
|
||||
"Он спрашивает: \"%s\"\n\nИсточники:\n%s\nОтветь ему коротко и своими словами, опираясь только на эти источники. Если ответа в них нет — так и скажи.",
|
||||
utterance, evidenceBlock(notes),
|
||||
)
|
||||
sys, prompt := p.evidencePrompt(utterance, notes)
|
||||
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
|
||||
text, _, perr := parseResponseMood(resp)
|
||||
if err != nil || perr != nil {
|
||||
@@ -481,6 +548,17 @@ func chatSystemPrompt(block func() string) string {
|
||||
// the LLM completion endpoint. Like chatWithSystem but for an arbitrary message
|
||||
// slice — the caller owns the system prompt placement.
|
||||
func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTokens int) (string, error) {
|
||||
// Same silent preference as chatWithSystem, when the array is the shape
|
||||
// llm.Req can carry: one system turn and one user turn. PhraseChat already
|
||||
// folds the history into a single user message (some chat templates reject
|
||||
// consecutive user turns), so today that is every call. A longer array goes
|
||||
// to the resident model rather than get flattened here, because flattening a
|
||||
// conversation is a decision its owner should make.
|
||||
if len(msgs) == 2 && msgs[0].Role == "system" && msgs[1].Role == "user" {
|
||||
if out, ok := p.remoteChat(ctx, msgs[0].Content, msgs[1].Content, maxTokens); ok {
|
||||
return out, nil
|
||||
}
|
||||
}
|
||||
base, release, err := p.acquire()
|
||||
if err != nil {
|
||||
return "", err
|
||||
@@ -640,6 +718,12 @@ func (p *LLMPhraser) chat(ctx context.Context, userPrompt string) (string, error
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, maxTokens int) (string, error) {
|
||||
// The workstation model first when it will take work, and silently: every
|
||||
// caller of this helper is on the silent half of the degradation rule. It
|
||||
// answering is not news, and it being asleep is not news either.
|
||||
if out, ok := p.remoteChat(ctx, system, user, maxTokens); ok {
|
||||
return out, nil
|
||||
}
|
||||
base, release, err := p.acquire()
|
||||
if err != nil {
|
||||
return "", err
|
||||
@@ -733,6 +817,27 @@ func (p *LLMPhraser) systemPrompt() string {
|
||||
return persona.Prepend(p.cfg.ContextBlock, nudgeSystem)
|
||||
}
|
||||
|
||||
// knowledgePrompt — the no-sources branch: a world question, answered from
|
||||
// weights alone. The system prompt is the single tested source in
|
||||
// router.KnowledgePrompt.
|
||||
//
|
||||
// Split out of PhraseQuery so PhraseWorld sends the workstation model the same
|
||||
// bytes the resident model gets. Prompt parity across two models is a stated
|
||||
// constraint (CLAUDE.md), and two copies of a prompt is how it stops holding.
|
||||
func (p *LLMPhraser) knowledgePrompt(utterance string) (sys, user string) {
|
||||
return persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt()),
|
||||
fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance)
|
||||
}
|
||||
|
||||
// evidencePrompt — the sources branch: read these, add nothing. Shared with
|
||||
// PhraseWorld for the same reason as knowledgePrompt.
|
||||
func (p *LLMPhraser) evidencePrompt(utterance string, notes []string) (sys, user string) {
|
||||
return p.querySystemPrompt(), fmt.Sprintf(
|
||||
"Он спрашивает: \"%s\"\n\nИсточники:\n%s\nОтветь ему коротко и своими словами, опираясь только на эти источники. Если ответа в них нет — так и скажи.",
|
||||
utterance, evidenceBlock(notes),
|
||||
)
|
||||
}
|
||||
|
||||
// querySystemPrompt returns the system prompt for the evidence branch of
|
||||
// PhraseQuery. Prepends the configured persona when set.
|
||||
//
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
// phraser/replier.go — reactive reply phrasing, the confirmation he hears
|
||||
// after every fact, note and reminder.
|
||||
//
|
||||
// It lived in cmd/mavend as package main until Vikunja #396, which meant the
|
||||
// most frequently heard sentence Maven says was the one path the phrasing eval
|
||||
// could not import, let alone score. Nothing here talks to the daemon: the
|
||||
// caller supplies the completer and the context block, and cmd/mavend keeps the
|
||||
// stub fallback so a model error still answers.
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/persona"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// Completer is the model seam for the replier, a subset of router.Completer.
|
||||
// *llm.Client satisfies it.
|
||||
type Completer interface {
|
||||
Complete(ctx context.Context, r llm.Req) (string, error)
|
||||
}
|
||||
|
||||
// replyTimeout bounds one reply. Generous because the resident model on the CPU
|
||||
// floor is slow and the caller has a deterministic fallback anyway.
|
||||
const replyTimeout = 60 * time.Second
|
||||
|
||||
// ReplySystemPrompt — the reactive confirmation contract: one short Russian
|
||||
// sentence, feminine self-reference, informal address, no question.
|
||||
const ReplySystemPrompt = `Ты — Maven, домашняя ассистентка (о себе — в женском роде). Владелец — мужчина, говоришь с ним на "ты", в единственном числе; никогда не "вы"/"ваш" и не "он"/"его". Подтверди действие РОВНО ОДНИМ коротким предложением (≤120 символов), по-русски, спокойно и без официальных формулировок. Не задавай вопросов, не повторяй слова, не добавляй ничего после точки. Отвечай ТОЛЬКО одним объектом JSON с полями "response" (текст) и "mood" (ровно одно из: neutral, happy, thinking, tired, confused).
|
||||
Пример: {"response": "Записала, что ты выпил стакан воды.", "mood": "neutral"}
|
||||
Никогда не пиши "..." в поле response.`
|
||||
|
||||
// Replier phrases reactive confirmations with the resident model. It has no
|
||||
// fallback of its own: an error is returned, and the daemon answers from the
|
||||
// deterministic stub. That is also what makes it scorable — a dead server shows
|
||||
// up as an error rather than as bad phrasing.
|
||||
type Replier struct {
|
||||
c Completer
|
||||
|
||||
// block renders the shared context block per turn (who he is, the time).
|
||||
// nil ⇒ the prompt stands alone.
|
||||
block func() string
|
||||
}
|
||||
|
||||
// NewReplier builds a replier over c. block may be nil.
|
||||
func NewReplier(c Completer, block func() string) *Replier {
|
||||
return &Replier{c: c, block: block}
|
||||
}
|
||||
|
||||
// PhraseReply returns the confirmation for one decision. An empty string with a
|
||||
// nil error means the model produced nothing usable, which the caller must
|
||||
// treat exactly like an error.
|
||||
func (r *Replier) PhraseReply(ctx context.Context, d router.Decision) (string, error) {
|
||||
ctx, cancel := context.WithTimeout(ctx, replyTimeout)
|
||||
defer cancel()
|
||||
out, err := r.c.Complete(ctx, llm.Req{
|
||||
System: persona.Prepend(r.block, ReplySystemPrompt),
|
||||
User: replyContext(d),
|
||||
Grammar: ResponseGrammar,
|
||||
MaxTokens: 512,
|
||||
RepeatPenalty: 1.3,
|
||||
})
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
out = stripThink(out)
|
||||
if response, _, perr := parseResponseMood(out); perr != nil {
|
||||
return "", perr
|
||||
} else if response != "" {
|
||||
return response, nil
|
||||
}
|
||||
// fallback: the model answered in bare prose, which is fine here.
|
||||
return firstSentence(out), nil
|
||||
}
|
||||
|
||||
// firstSentence trims the model's output to a single clean confirmation: first
|
||||
// line, first sentence, whitespace-normalized — the last-line defense against a
|
||||
// small model that rambles past the first period despite the prompt + stop.
|
||||
func firstSentence(s string) string {
|
||||
s = strings.TrimSpace(s)
|
||||
if i := strings.IndexByte(s, '\n'); i >= 0 {
|
||||
s = s[:i]
|
||||
}
|
||||
// keep up to and including the first sentence-ending punctuation.
|
||||
if i := strings.IndexAny(s, ".!?"); i >= 0 {
|
||||
s = s[:i+1]
|
||||
}
|
||||
return strings.TrimSpace(s)
|
||||
}
|
||||
|
||||
// replyContext renders the decision into a compact RU description for the model.
|
||||
func replyContext(d router.Decision) string {
|
||||
switch d.Intent {
|
||||
case router.IntentFact:
|
||||
return "записала факт: " + d.Slots.Key + " " + d.Slots.Value
|
||||
case router.IntentNote:
|
||||
return "сохранила заметку: " + d.Slots.Text
|
||||
case router.IntentReminder:
|
||||
return "поставила напоминание: " + d.Slots.Text
|
||||
default:
|
||||
return string(d.Intent) + ": " + d.Slots.Text
|
||||
}
|
||||
}
|
||||
|
||||
// StripThink removes the <think> block a Thinking-variant model emits before its
|
||||
// answer. Exported for the daemon's own model callers, which parse output that
|
||||
// never passes through a phraser method.
|
||||
func StripThink(s string) string { return stripThink(s) }
|
||||
@@ -0,0 +1,90 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
type mockCompleter struct {
|
||||
out string
|
||||
err error
|
||||
}
|
||||
|
||||
func (m mockCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return m.out, m.err }
|
||||
|
||||
func TestReplierReturnsLLMReply(t *testing.T) {
|
||||
r := NewReplier(mockCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
|
||||
got, err := r.PhraseReply(context.Background(), noteDecision())
|
||||
if err != nil || got != "записала, кофе закончился" {
|
||||
t.Errorf("got %q, %v, want %q, nil", got, err, "записала, кофе закончился")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReplierFallsBackToPlainText(t *testing.T) {
|
||||
r := NewReplier(mockCompleter{out: "записала, кофе закончился"}, nil)
|
||||
got, err := r.PhraseReply(context.Background(), noteDecision())
|
||||
if err != nil || got != "записала, кофе закончился" {
|
||||
t.Errorf("got %q, %v, want %q, nil", got, err, "записала, кофе закончился")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReplierReportsTheModelError(t *testing.T) {
|
||||
r := NewReplier(mockCompleter{err: errTestLLMDown}, nil)
|
||||
got, err := r.PhraseReply(context.Background(), noteDecision())
|
||||
if err == nil {
|
||||
t.Errorf("got %q, nil error — a dead model must be reported, not phrased around", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A fragment the grammar left half-open is a failed generation. It must come
|
||||
// back as an error so the daemon reaches its stub, not as a reply.
|
||||
func TestReplierRejectsBrokenJSON(t *testing.T) {
|
||||
r := NewReplier(mockCompleter{out: `{"response":"запис`}, nil)
|
||||
got, err := r.PhraseReply(context.Background(), noteDecision())
|
||||
if err == nil || got != "" {
|
||||
t.Errorf("got %q, %v, want empty and an error", got, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReplierEmptyOutputIsEmpty(t *testing.T) {
|
||||
r := NewReplier(mockCompleter{out: ""}, nil)
|
||||
got, err := r.PhraseReply(context.Background(), noteDecision())
|
||||
if err != nil || got != "" {
|
||||
t.Errorf("got %q, %v, want empty and no error", got, err)
|
||||
}
|
||||
}
|
||||
|
||||
// grammarRecorder captures the request so the grammar can be asserted on.
|
||||
type grammarRecorder struct{ req llm.Req }
|
||||
|
||||
func (g *grammarRecorder) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||
g.req = r
|
||||
return `{"response":"записала","mood":"neutral"}`, nil
|
||||
}
|
||||
|
||||
func TestReplierCarriesTheResponseGrammar(t *testing.T) {
|
||||
rec := &grammarRecorder{}
|
||||
r := NewReplier(rec, nil)
|
||||
if _, err := r.PhraseReply(context.Background(), noteDecision()); err != nil {
|
||||
t.Fatalf("PhraseReply: %v", err)
|
||||
}
|
||||
if rec.req.Grammar != ResponseGrammar {
|
||||
t.Errorf("grammar = %q, want ResponseGrammar", rec.req.Grammar)
|
||||
}
|
||||
if rec.req.System != ReplySystemPrompt {
|
||||
t.Errorf("system prompt = %q, want ReplySystemPrompt", rec.req.System)
|
||||
}
|
||||
}
|
||||
|
||||
func noteDecision() router.Decision {
|
||||
return router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}}
|
||||
}
|
||||
|
||||
var errTestLLMDown = errTest("llm down")
|
||||
|
||||
type errTest string
|
||||
|
||||
func (e errTest) Error() string { return string(e) }
|
||||
@@ -1,9 +1,11 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
@@ -59,6 +61,19 @@ func TestExtractPort(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The prompt cache is what ate 6.8GB of the deployed server's RSS, so the cap
|
||||
// has to reach the command line, and the opt-out has to leave it off.
|
||||
func TestLlamaArgsCapsPromptCache(t *testing.T) {
|
||||
cfg := DefaultConfig("/m.gguf")
|
||||
if got := strings.Join(llamaArgs(cfg), " "); !strings.Contains(got, "--cache-ram 512") {
|
||||
t.Errorf("default args = %q, want --cache-ram 512", got)
|
||||
}
|
||||
cfg.CacheRAMMiB = 0
|
||||
if got := strings.Join(llamaArgs(cfg), " "); strings.Contains(got, "--cache-ram") {
|
||||
t.Errorf("args with the cap off = %q, want no --cache-ram flag", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartLlamaProcScrapesPortAndReaps(t *testing.T) {
|
||||
bin := fakeLlama(t, listensThenSleeps)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
@@ -86,6 +101,48 @@ func TestStartLlamaProcScrapesPortAndReaps(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// captureLog redirects the standard logger for the duration of a test and
|
||||
// returns what was written to it.
|
||||
func captureLog(t *testing.T) *bytes.Buffer {
|
||||
t.Helper()
|
||||
var buf bytes.Buffer
|
||||
old := log.Writer()
|
||||
flags := log.Flags()
|
||||
log.SetOutput(&buf)
|
||||
log.SetFlags(0)
|
||||
t.Cleanup(func() { log.SetOutput(old); log.SetFlags(flags) })
|
||||
return &buf
|
||||
}
|
||||
|
||||
// The child's buffer-size, KV-cache and offload lines are the only way to
|
||||
// account for its memory on a running box, and they used to be dropped: stderr
|
||||
// was scraped for the listen line and thrown away, stdout was never piped.
|
||||
func TestStartLlamaProcForwardsChildOutput(t *testing.T) {
|
||||
buf := captureLog(t)
|
||||
bin := fakeLlama(t, `echo "load_tensors: Vulkan0 model buffer size = 1053.34 MiB" >&2
|
||||
echo "llama_context: KV self size = 448.00 MiB"
|
||||
`+listensThenSleeps)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
p, err := startLlamaProc(ctx, testCfg(bin))
|
||||
if err != nil {
|
||||
t.Fatalf("startLlamaProc: %v", err)
|
||||
}
|
||||
p.cancel = cancel
|
||||
defer p.Close()
|
||||
|
||||
got := buf.String()
|
||||
for _, want := range []string{
|
||||
"llama: load_tensors: Vulkan0 model buffer size = 1053.34 MiB", // stderr
|
||||
"llama: llama_context: KV self size = 448.00 MiB", // stdout, previously discarded
|
||||
} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("log missing %q\nlog was:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartLlamaProcFailureArms(t *testing.T) {
|
||||
t.Run("binary missing", func(t *testing.T) {
|
||||
cfg := testCfg(filepath.Join(t.TempDir(), "does-not-exist"))
|
||||
@@ -96,13 +153,18 @@ func TestStartLlamaProcFailureArms(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("server exits without listening", func(t *testing.T) {
|
||||
// stderr closes, so the reader goroutine reports EOF on errCh.
|
||||
// stderr closes, so the reader goroutine reports EOF on errCh. The error
|
||||
// must carry the child's last words: bare "EOF" named no cause.
|
||||
captureLog(t)
|
||||
bin := fakeLlama(t, `echo "ggml_vulkan: no device" >&2
|
||||
exit 1`)
|
||||
_, err := startLlamaProc(context.Background(), testCfg(bin))
|
||||
if err == nil || !strings.Contains(err.Error(), "llm: server output") {
|
||||
t.Fatalf("err = %v, want the server-output arm", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "ggml_vulkan: no device") {
|
||||
t.Errorf("err = %v, want the child's last output in it", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("context cancelled during startup", func(t *testing.T) {
|
||||
@@ -241,15 +303,7 @@ func TestKillMavenScriptMatchesRealCommandLine(t *testing.T) {
|
||||
// startLlamaProc that breaks the sweep fails here instead of on the box.
|
||||
cfg := DefaultConfig("/opt/maven/models/llm/Qwen3-1.7B-UD-Q4_K_XL.gguf")
|
||||
cfg.NCtx, cfg.NGpuLayers = 4096, 99
|
||||
cmdline := strings.Join([]string{
|
||||
cfg.BinPath,
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
"--port", extractPort(cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}, " ")
|
||||
cmdline := cfg.BinPath + " " + strings.Join(llamaArgs(cfg), " ")
|
||||
if !pat.MatchString(cmdline) {
|
||||
t.Fatalf("kill-maven.sh pattern %q does not match %q — orphans would leak", m[1], cmdline)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
)
|
||||
|
||||
// Remote — the workstation model, seen from the phraser. `*llm.Pair` satisfies
|
||||
// it, and a test fake satisfies it in three lines.
|
||||
//
|
||||
// Only the refusing half of Pair is here on purpose. Pair.Complete falls back to
|
||||
// its own floor client, and the phraser already owns a floor: the llama-server it
|
||||
// spawned. Two floors under one call is one too many, so the phraser asks whether
|
||||
// the remote will take work, uses it when it will, and otherwise does exactly
|
||||
// what it did before this file existed.
|
||||
type Remote interface {
|
||||
// Available is an atomic read of a cached probe, so it is free to call per
|
||||
// turn. See llm.Pair.
|
||||
Available() bool
|
||||
// CompleteRemote runs on the workstation or returns ErrRemoteUnavailable. It
|
||||
// never falls back.
|
||||
CompleteRemote(ctx context.Context, r llm.Req) (string, error)
|
||||
}
|
||||
|
||||
// ErrNoWorldModel — a world question was asked, a workstation model is
|
||||
// configured to answer it, and that machine is not answering. The caller turns
|
||||
// this into a gap he is told about ("не могу сейчас"), never into an answer from
|
||||
// the resident model.
|
||||
//
|
||||
// This is the naming half of the degradation rule in docs/offload.md. The
|
||||
// resident Qwen3-1.7B does not answer a world question worse than the 12B, it
|
||||
// invents: measured, the workstation model scores knowledge 9/9 on the talk
|
||||
// fixture against the resident model's confabulations
|
||||
// (docs/evals/2026-08-02-workstation-gemma4-12b.md).
|
||||
var ErrNoWorldModel = errors.New("phraser: no world model available")
|
||||
|
||||
// chatTemperature — what the phraser's own transport has always sampled at.
|
||||
// Named so the remote path cannot drift from it silently. Whether 0.7 is right
|
||||
// at all is Vikunja #402, and answering that here would hide a phrasing change
|
||||
// inside a routing change.
|
||||
const chatTemperature = 0.7
|
||||
|
||||
// UseRemote points the phraser at the workstation model. Wiring time only, once,
|
||||
// before anything phrases: the field is read without a lock on every call
|
||||
// because a per-turn lock to answer a question that changes at deploy time is
|
||||
// not worth paying for.
|
||||
//
|
||||
// A nil remote is the normal state of a box with no `workstation` block, and it
|
||||
// must behave exactly as the box behaved before this seam existed.
|
||||
func (p *LLMPhraser) UseRemote(r Remote) {
|
||||
p.remote = r
|
||||
}
|
||||
|
||||
// PhraseWorld answers a question about the world — either from the model's own
|
||||
// knowledge (no sources) or from a passage someone fetched (a live search, a ZIM
|
||||
// article, a page he named). Three outcomes, and the middle one is the point:
|
||||
//
|
||||
// - No workstation configured. The resident model answers, exactly as it does
|
||||
// today. Naming a gap needs a gap: on a box that never had a second model,
|
||||
// refusing every world question would remove a capability he has now.
|
||||
// - Workstation configured and taking work. It answers.
|
||||
// - Workstation configured and down. ErrNoWorldModel, and the caller says so.
|
||||
//
|
||||
// The prompts are the ones PhraseQuery uses, built by the same two functions, so
|
||||
// the two models are asked the same question in the same words.
|
||||
func (p *LLMPhraser) PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error) {
|
||||
sources = nonEmpty(sources)
|
||||
if p.remote == nil {
|
||||
return p.PhraseQuery(ctx, utterance, sources)
|
||||
}
|
||||
var sys, user string
|
||||
if len(sources) == 0 {
|
||||
sys, user = p.knowledgePrompt(utterance)
|
||||
} else {
|
||||
sys, user = p.evidencePrompt(utterance, sources)
|
||||
}
|
||||
if !p.remote.Available() {
|
||||
return "", ErrNoWorldModel
|
||||
}
|
||||
resp, err := p.remote.CompleteRemote(ctx, llm.Req{
|
||||
System: sys,
|
||||
User: user,
|
||||
Grammar: p.grammar(),
|
||||
MaxTokens: 768,
|
||||
Temperature: chatTemperature,
|
||||
})
|
||||
if err != nil {
|
||||
// The cached probe was one interval stale, or the card went away
|
||||
// mid-request. Either way this is the gap, not an error to log and
|
||||
// paper over with the smaller model.
|
||||
log.Printf("phraser: world model: %v", err)
|
||||
return "", errors.Join(ErrNoWorldModel, err)
|
||||
}
|
||||
resp = stripThink(resp)
|
||||
text, _, perr := parseResponseMood(resp)
|
||||
if perr != nil {
|
||||
log.Printf("phraser: PhraseWorld: %v", perr)
|
||||
return "", errors.Join(ErrNoWorldModel, perr)
|
||||
}
|
||||
if text != "" {
|
||||
return text, nil
|
||||
}
|
||||
if resp == "" {
|
||||
return "", ErrNoWorldModel
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// remoteChat is the silent half, for the phrasing paths where the workstation
|
||||
// model is only better: a nudge, a reminder, a reply, a question answered from
|
||||
// his own notes. It reports whether it answered; it never reports why not,
|
||||
// because the caller's next move is the resident model either way.
|
||||
//
|
||||
// He is not told which of the two models phrased his reply. That is the rule.
|
||||
func (p *LLMPhraser) remoteChat(ctx context.Context, system, user string, maxTokens int) (string, bool) {
|
||||
if p.remote == nil || !p.remote.Available() {
|
||||
return "", false
|
||||
}
|
||||
out, err := p.remote.CompleteRemote(ctx, llm.Req{
|
||||
System: system,
|
||||
User: user,
|
||||
Grammar: p.grammar(),
|
||||
MaxTokens: maxTokens,
|
||||
Temperature: chatTemperature,
|
||||
})
|
||||
if err != nil {
|
||||
log.Printf("phraser: workstation model declined, phrasing here instead: %v", err)
|
||||
return "", false
|
||||
}
|
||||
if out = stripThink(out); out == "" {
|
||||
return "", false
|
||||
}
|
||||
return out, true
|
||||
}
|
||||
@@ -0,0 +1,164 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/loop"
|
||||
)
|
||||
|
||||
// fakeRemote — a workstation model that is up or down on command, and records
|
||||
// what it was asked.
|
||||
type fakeRemote struct {
|
||||
up bool
|
||||
reply string
|
||||
err error
|
||||
got []llm.Req
|
||||
}
|
||||
|
||||
func (f *fakeRemote) Available() bool { return f.up }
|
||||
|
||||
func (f *fakeRemote) CompleteRemote(_ context.Context, r llm.Req) (string, error) {
|
||||
f.got = append(f.got, r)
|
||||
if f.err != nil {
|
||||
return "", f.err
|
||||
}
|
||||
return f.reply, nil
|
||||
}
|
||||
|
||||
// The three outcomes of the naming half, in one place. The middle one is the
|
||||
// whole task: a gap he is told about, not an answer from the smaller model.
|
||||
func TestPhraseWorldNamesTheGapOnlyWhenThereIsOne(t *testing.T) {
|
||||
answer := `{"response": "Небо голубое из-за рэлеевского рассеяния.", "mood": "neutral"}`
|
||||
|
||||
t.Run("no workstation configured: the resident model answers as today", func(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
p := NewLLMPhraserAt(spy.srv.URL, Config{})
|
||||
got, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil)
|
||||
if err != nil {
|
||||
t.Fatalf("PhraseWorld: %v", err)
|
||||
}
|
||||
if got == "" {
|
||||
t.Fatal("no reply from the resident model")
|
||||
}
|
||||
if len(spy.user) != 1 {
|
||||
t.Fatalf("resident model saw %d requests, want 1", len(spy.user))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("workstation up: it answers and the resident model is not asked", func(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
p := NewLLMPhraserAt(spy.srv.URL, Config{})
|
||||
remote := &fakeRemote{up: true, reply: answer}
|
||||
p.UseRemote(remote)
|
||||
got, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil)
|
||||
if err != nil {
|
||||
t.Fatalf("PhraseWorld: %v", err)
|
||||
}
|
||||
if !strings.Contains(got, "рассеяния") {
|
||||
t.Errorf("reply is not the workstation's: %q", got)
|
||||
}
|
||||
if len(spy.user) != 0 {
|
||||
t.Errorf("the resident model was asked %d times, want 0", len(spy.user))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("workstation down: the gap, and nothing invented", func(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
p := NewLLMPhraserAt(spy.srv.URL, Config{})
|
||||
p.UseRemote(&fakeRemote{up: false})
|
||||
got, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil)
|
||||
if !errors.Is(err, ErrNoWorldModel) {
|
||||
t.Fatalf("err = %v, want ErrNoWorldModel", err)
|
||||
}
|
||||
if got != "" {
|
||||
t.Errorf("got a reply %q with no world model", got)
|
||||
}
|
||||
if len(spy.user) != 0 {
|
||||
t.Errorf("the resident model answered a world question %d times, want 0", len(spy.user))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("workstation errors mid-request: still the gap", func(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
p := NewLLMPhraserAt(spy.srv.URL, Config{})
|
||||
p.UseRemote(&fakeRemote{up: true, err: errors.New("connection refused")})
|
||||
if _, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil); !errors.Is(err, ErrNoWorldModel) {
|
||||
t.Fatalf("err = %v, want ErrNoWorldModel", err)
|
||||
}
|
||||
if len(spy.user) != 0 {
|
||||
t.Errorf("the resident model answered a world question %d times, want 0", len(spy.user))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// Prompt parity: the workstation model is asked the same question in the same
|
||||
// words, or the fixtures measure one thing and the daemon ships another.
|
||||
func TestPhraseWorldSendsTheSamePromptsAsPhraseQuery(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
resident := NewLLMPhraserAt(spy.srv.URL, Config{})
|
||||
if _, err := resident.PhraseQuery(context.Background(), "кто написал войну и мир", []string{"Лев Толстой"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
remote := &fakeRemote{up: true, reply: `{"response": "Толстой.", "mood": "neutral"}`}
|
||||
offloaded := NewLLMPhraserAt(spy.srv.URL, Config{})
|
||||
offloaded.UseRemote(remote)
|
||||
if _, err := offloaded.PhraseWorld(context.Background(), "кто написал войну и мир", []string{"Лев Толстой"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if len(remote.got) != 1 {
|
||||
t.Fatalf("the workstation saw %d requests, want 1", len(remote.got))
|
||||
}
|
||||
if remote.got[0].System != spy.system[0] {
|
||||
t.Errorf("system prompts differ:\nremote: %q\nresident: %q", remote.got[0].System, spy.system[0])
|
||||
}
|
||||
if remote.got[0].User != spy.user[0] {
|
||||
t.Errorf("user prompts differ:\nremote: %q\nresident: %q", remote.got[0].User, spy.user[0])
|
||||
}
|
||||
}
|
||||
|
||||
// The silent half. A nudge phrased on the workstation is not news, and one
|
||||
// phrased here because the card is busy is not news either — but it must be
|
||||
// sampled the same way, or the workstation quietly changes how she sounds.
|
||||
func TestNudgePhrasingPrefersTheWorkstationSilently(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
p := NewLLMPhraserAt(spy.srv.URL, Config{LLMNudges: true})
|
||||
remote := &fakeRemote{up: true, reply: `{"response": "Выпей воды.", "mood": "neutral"}`}
|
||||
p.UseRemote(remote)
|
||||
|
||||
pn, err := p.PhraseNudge(context.Background(), loop.Candidate{Rule: loop.WaterRule(), Severity: loop.Sev1})
|
||||
if err != nil {
|
||||
t.Fatalf("PhraseNudge: %v", err)
|
||||
}
|
||||
if pn.Body != "Выпей воды." {
|
||||
t.Errorf("body = %q, want the workstation's wording", pn.Body)
|
||||
}
|
||||
if len(remote.got) != 1 {
|
||||
t.Fatalf("the workstation saw %d requests, want 1", len(remote.got))
|
||||
}
|
||||
if remote.got[0].Temperature != chatTemperature {
|
||||
t.Errorf("temperature = %v, want %v (what the resident transport samples at)",
|
||||
remote.got[0].Temperature, chatTemperature)
|
||||
}
|
||||
if len(spy.user) != 0 {
|
||||
t.Errorf("the resident model phrased %d nudges, want 0", len(spy.user))
|
||||
}
|
||||
}
|
||||
|
||||
func TestNudgePhrasingFallsBackWhenTheCardIsBusy(t *testing.T) {
|
||||
spy := newPromptSpy(t)
|
||||
p := NewLLMPhraserAt(spy.srv.URL, Config{LLMNudges: true})
|
||||
p.UseRemote(&fakeRemote{up: false})
|
||||
|
||||
if _, err := p.PhraseNudge(context.Background(), loop.Candidate{Rule: loop.WaterRule(), Severity: loop.Sev1}); err != nil {
|
||||
t.Fatalf("PhraseNudge: %v", err)
|
||||
}
|
||||
if len(spy.user) != 1 {
|
||||
t.Fatalf("the resident model phrased %d nudges, want 1", len(spy.user))
|
||||
}
|
||||
}
|
||||
@@ -75,3 +75,47 @@ func TestAgendaGrammarSparesStatements(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The tomorrow form and the bare event noun. Both were measured answering
|
||||
// "пока не умею" on the deployed daemon, 02-08-2026, while the same question
|
||||
// about today worked — the first rule set needed "у меня" or a calendar noun
|
||||
// and these phrasings carry neither (Vikunja #471).
|
||||
func TestAgendaCoversOtherDaysAndNamedEvents(t *testing.T) {
|
||||
r := agendaRouter(t)
|
||||
for _, u := range []string{
|
||||
"какие планы на завтра?",
|
||||
"какие планы на послезавтра",
|
||||
"что по делам в среду",
|
||||
"какие планы на выходные",
|
||||
"когда планёрка?",
|
||||
"во сколько созвон",
|
||||
"когда будет совещание",
|
||||
} {
|
||||
d, err := r.Route(context.Background(), u, refNow())
|
||||
if err != nil {
|
||||
t.Fatalf("route(%q): %v", u, err)
|
||||
}
|
||||
if d.Intent != IntentQuery {
|
||||
t.Errorf("route(%q) = %s, want query", u, d.Intent)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The two new rules are narrow on purpose. A world question that opens with
|
||||
// "когда" is not an agenda question, and telling her about a plan is not
|
||||
// asking about one.
|
||||
func TestAgendaGrammarsLeaveTheWorldAlone(t *testing.T) {
|
||||
r := agendaRouter(t)
|
||||
for _, u := range []string{
|
||||
"когда была битва при ватерлоо",
|
||||
"когда изобрели телефон",
|
||||
} {
|
||||
d, err := r.Route(context.Background(), u, refNow())
|
||||
if err != nil {
|
||||
t.Fatalf("route(%q): %v", u, err)
|
||||
}
|
||||
if d.Stage == 0 {
|
||||
t.Errorf("route(%q) was claimed at stage 0 as %s", u, d.Intent)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
package router
|
||||
|
||||
import "strings"
|
||||
|
||||
// transientStems — the states a thing is in for an afternoon. Compared as
|
||||
// prefixes because Russian inflects the ending: "медленн" covers "медленная",
|
||||
// "медленный" and "медленно" without listing them.
|
||||
var transientStems = []string{
|
||||
"медленн", "тормоз", "лаг", "завис", "виснет", "глюч", "барахл",
|
||||
"отвал", "падает", "упал", "сдох", "греется", "перегре",
|
||||
"slow", "laggy", "stuck", "frozen", "flaky", "broken", "down",
|
||||
}
|
||||
|
||||
// brokenVerbs — what "не ..." is denying when the sentence is a complaint.
|
||||
// "не работает", "не грузит", "не открывается". Prefixes again.
|
||||
var brokenVerbs = []string{
|
||||
"работ", "пашет", "груз", "открыва", "включа", "коннект", "подключ",
|
||||
"work", "load", "connect", "respond",
|
||||
}
|
||||
|
||||
// selfMarkers — the words that make a sentence about him rather than about a
|
||||
// thing. Their presence turns the test off, because losing a fact he meant to
|
||||
// store is worse than keeping a complaint: "я сломал руку" is durable, and
|
||||
// "интернет не работает" is not.
|
||||
var selfMarkers = []string{"я", "мне", "меня", "мной", "i", "me", "my"}
|
||||
|
||||
// IsTransientComplaint reports whether text observes a passing state of some
|
||||
// thing rather than recording a fact.
|
||||
//
|
||||
// It exists because "сеть какая-то медленная" and "интернет не работает" were
|
||||
// written to the fact store as `self` rows at confidence 1.00 (Vikunja #481),
|
||||
// where recall reads them back later as if they were still true. A complaint
|
||||
// describes a moment; the fact store describes him.
|
||||
//
|
||||
// Deterministic, offline, and shaped exactly like IsQuestionShaped: an
|
||||
// explicit capture verb wins over everything, because "запомни что интернет
|
||||
// не работает" is an instruction and not a passing remark. A first-person
|
||||
// marker also turns it off — the test is meant to catch a sentence about a
|
||||
// thing, and it errs toward storing.
|
||||
func IsTransientComplaint(text string) bool {
|
||||
t := strings.TrimSpace(text)
|
||||
if t == "" {
|
||||
return false
|
||||
}
|
||||
toks := planTokens(strings.ToLower(t))
|
||||
for _, v := range captureVerbs {
|
||||
if hasTok(toks, v) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
for _, m := range selfMarkers {
|
||||
if hasTok(toks, m) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
for _, tok := range toks {
|
||||
for _, stem := range transientStems {
|
||||
if strings.HasPrefix(tok, stem) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
// "не" plus a verb of working, in either order of the two tokens that
|
||||
// follow it — "не работает" and "не очень работает" both deny the same
|
||||
// thing.
|
||||
for i, tok := range toks {
|
||||
if tok != "не" && tok != "not" && tok != "isn" {
|
||||
continue
|
||||
}
|
||||
for j := i + 1; j < len(toks) && j <= i+2; j++ {
|
||||
for _, v := range brokenVerbs {
|
||||
if strings.HasPrefix(toks[j], v) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package router
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestIsTransientComplaint(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
text string
|
||||
want bool
|
||||
}{
|
||||
// The two rows from the QA run that named this bug.
|
||||
{"сеть какая-то медленная", true},
|
||||
{"интернет не работает", true},
|
||||
{"вайфай тормозит", true},
|
||||
{"сервер завис", true},
|
||||
{"the wifi is slow", true},
|
||||
|
||||
// An instruction wins: he asked for it to be written down.
|
||||
{"запомни что интернет не работает", false},
|
||||
{"запиши что сеть медленная", false},
|
||||
|
||||
// About him, so it stays a fact even when it sounds like a complaint.
|
||||
{"я сломал руку", false},
|
||||
{"мне медленно думается", false},
|
||||
|
||||
// Ordinary captures must not be touched.
|
||||
{"поужинал", false},
|
||||
{"выпил воды", false},
|
||||
{"машина на парковке", false},
|
||||
{"", false},
|
||||
} {
|
||||
if got := IsTransientComplaint(tc.text); got != tc.want {
|
||||
t.Errorf("IsTransientComplaint(%q) = %v, want %v", tc.text, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -81,6 +81,9 @@ func NewPythonDateParser() *PythonDateParser {
|
||||
// or dateparser is unavailable, falls back to the stub parser. Returns
|
||||
// (time, true, nil) on success; (zero, false, nil) when no date is found.
|
||||
func (p *PythonDateParser) Parse(ctx context.Context, text string, now time.Time) (time.Time, bool, error) {
|
||||
// Speech says the hour in words, and neither this parser nor the stub
|
||||
// reads "в семь вечера" (Vikunja #469). Both see the digits instead.
|
||||
text = SpellOutDigits(text)
|
||||
t, ok, err := p.parseWithPython(ctx, text, now)
|
||||
if err != nil {
|
||||
// python3 missing, dateparser not installed, or process failure —
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user