Merge pull request 'Name the gap: world questions through the workstation model, and the four remaining callers' (#98) from task/490-name-the-gap-world-questions-through-the into master

Reviewed-on: #98
This commit was merged in pull request #98.
This commit is contained in:
2026-08-03 11:19:56 +02:00
12 changed files with 584 additions and 52 deletions
+7
View File
@@ -36,6 +36,13 @@ stays on homesrv permanently, because it backs that floor. Read `docs/offload.md
touching a daemon seam or adding a model caller. Vikunja #483 is the umbrella, #484 to #487
are the work.
Both halves are wired as of 2026-08-03. Routing and replies prefer the workstation silently
through `modelSeam`; nudge and reminder phrasing prefer it silently inside the phraser. A
world question goes through `LLMPhraser.PhraseWorld` and names the gap when the card is not
free — `worldGap` in `cmd/mavend/worldmodel.go`, which he hears instead of an invented
answer. A box with no `workstation` block behaves exactly as it did before the seam: naming
a gap requires a gap. The offload table in `docs/offload.md` says which caller is which.
## Build & test
CGO daemons (`mavend`, `mavsttd`, `mavttsd`, `mavenclient`) need the vendored toolchain
+25 -24
View File
@@ -13,6 +13,7 @@ import (
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/memory"
"github.com/kami/maven/internal/morning"
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
"github.com/kami/maven/internal/rss"
"github.com/kami/maven/internal/store"
@@ -523,10 +524,7 @@ func (h *reactiveHandler) queryWeb(ctx context.Context, t *queryTurn) (string, b
// the question he actually asked. She answers the question, she does not
// recite the page.
snippet := page.Title + "\n" + crawl.TrimRunes(page.Text, webPageContextRunes)
reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
if perr != nil {
log.Printf("voice: web: phrase: %v", perr)
}
reply := h.phraseSource(ctx, "web", t.dec.Utterance, []string{snippet})
if reply == "" {
// No phraser (or it failed): read back the top of the page rather than
// pretend the fetch did not happen.
@@ -592,14 +590,7 @@ func (h *reactiveHandler) querySearch(ctx context.Context, t *queryTurn) (string
// question he asked, not something to recite. The trim is one budget over the
// joined block, so a long first snippet cannot crowd out the rest.
evidence := crawl.TrimRunes(strings.Join(resp.Snippets(), "\n"), h.search.runes)
var reply string
if h.phraser != nil {
var perr error
reply, perr = h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{evidence})
if perr != nil {
log.Printf("voice: search: phrase: %v", perr)
}
}
reply := h.phraseSource(ctx, "search", t.dec.Utterance, []string{evidence})
if reply == "" {
// No phraser, or it failed. Read back the best evidence rather than
// pretend the search did not happen.
@@ -680,14 +671,7 @@ func (h *reactiveHandler) queryKiwix(ctx context.Context, t *queryTurn) (string,
// Handed over the same way a note or a page is: context for the question he
// asked, not something to recite.
snippet := top.Title + "\n" + crawl.TrimRunes(page.Text, h.kiwix.runes)
var reply string
if h.phraser != nil {
var perr error
reply, perr = h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{snippet})
if perr != nil {
log.Printf("voice: kiwix: phrase: %v", perr)
}
}
reply := h.phraseSource(ctx, "kiwix", t.dec.Utterance, []string{snippet})
if reply == "" {
// No phraser, or it failed. Read back the best hit rather than pretend
// the search did not happen.
@@ -755,11 +739,28 @@ func isPersonalQuery(utterance string) bool {
return false
}
// queryGeneral — general knowledge from the phraser, the last source before
// giving up. It always claims: either the model answers or Maven says she
// doesn't know.
// queryGeneral — general knowledge, the last source before giving up. It always
// claims: either a model answers, or Maven names the gap, or she says she does
// not know.
//
// This is the sharpest case for the naming half. Nothing has been fetched, so
// there is no passage to fall back on and no floor under the answer except the
// model's weights — and a 1.7B's weights are where the invented answers come
// from. With a workstation configured and asleep he is told that, rather than
// told something false in a confident voice. With no workstation configured at
// all the resident model answers exactly as it does today: naming a gap requires
// a gap, and on that box the 1.7B is the whole product.
func (h *reactiveHandler) queryGeneral(ctx context.Context, t *queryTurn) (string, bool) {
reply, err := h.phraser.PhraseQuery(ctx, t.dec.Utterance, nil)
if h.phraser == nil {
// No model of any size. That is not the workstation being asleep, so it
// is not that gap: it is simply not knowing.
return "не знаю.", true
}
reply, err := h.phraseWorld(ctx, t.dec.Utterance, nil)
if errors.Is(err, phraser.ErrNoWorldModel) {
log.Printf("voice: %q needs the world model and it is not available", t.dec.Utterance)
return worldGap, true
}
if err != nil || reply == "" {
return "не знаю.", true
}
+7
View File
@@ -200,6 +200,13 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
// either the pair, or the resident client alone, or nothing at all.
hot, pair := modelSeam(cfg, llmClient)
w.pair = pair
// The phraser gets the same pair, which is what carries the workstation model
// into the paths that do not go through `hot`: world questions (the naming
// half), and the digestion worker's nudge and reminder phrasing (the silent
// half). Wiring, so it happens once and before the voice server listens.
if lp, ok := phr.(*phraser.LLMPhraser); ok && pair != nil {
lp.UseRemote(pair)
}
// ----- router (the cascade; floor examples seed the classifier) -----
// The act matcher's allowlist is exactly the enabled tool names — the
// router only matches acts the executor can run (one source of truth).
+60
View File
@@ -0,0 +1,60 @@
package main
import (
"context"
"errors"
"log"
"github.com/kami/maven/internal/phraser"
)
// worldPhraser — the naming half of the degradation rule (docs/offload.md), as
// the query sources see it. Only *phraser.LLMPhraser implements it, so the
// Stub and every test double stay exactly as they are.
type worldPhraser interface {
PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error)
}
// worldGap — what he hears when the question is about the world, the workstation
// model is the one configured to answer it, and that machine is not answering.
//
// It says the true thing. The resident 1.7B is not a worse answer here, it is an
// invented one: "Война и мир" came back with Левитан as its author, and a
// question about his meeting came back as a swimming competition in Nottingham.
// Naming the gap is the rule CLAUDE.md already applies to a sibling service
// being down.
const worldGap = "сейчас не могу ответить — большая модель недоступна, а придумывать не хочу."
// phraseWorld asks the world model, or reports the gap.
//
// The three outcomes come straight from LLMPhraser.PhraseWorld: no workstation
// configured means the resident model answers as it always has, a workstation
// that is up answers, and a workstation that is down returns
// phraser.ErrNoWorldModel. A phraser that has no world seam at all — the Stub,
// and the doubles in the tests — is the first of those three.
func (h *reactiveHandler) phraseWorld(ctx context.Context, utterance string, sources []string) (string, error) {
if h.phraser == nil {
return "", phraser.ErrNoWorldModel
}
if w, ok := h.phraser.(worldPhraser); ok {
return w.PhraseWorld(ctx, utterance, sources)
}
return h.phraser.PhraseQuery(ctx, utterance, sources)
}
// phraseSource asks the world model to answer from a passage someone already
// fetched — a live search result, a ZIM article, a page he named. It returns ""
// rather than the gap phrase, because these callers hold something better than a
// gap: the passage itself, which their own floor reads back to him. Nothing is
// invented either way, and a real quote beats "не могу сейчас".
func (h *reactiveHandler) phraseSource(ctx context.Context, name, utterance string, sources []string) string {
reply, err := h.phraseWorld(ctx, utterance, sources)
switch {
case errors.Is(err, phraser.ErrNoWorldModel):
log.Printf("voice: %s: no world model, reading the source back instead", name)
return ""
case err != nil:
log.Printf("voice: %s: phrase: %v", name, err)
}
return reply
}
+85
View File
@@ -0,0 +1,85 @@
package main
import (
"context"
"strings"
"testing"
"github.com/kami/maven/internal/phraser"
"github.com/kami/maven/internal/router"
)
// gapPhraser — a phraser whose world model is configured and asleep, which is
// the state the naming half exists for.
type gapPhraser struct {
*phraser.Stub
worldCalls int
}
func (g *gapPhraser) PhraseWorld(context.Context, string, []string) (string, error) {
g.worldCalls++
return "", phraser.ErrNoWorldModel
}
func worldTurn(utterance string) *queryTurn {
return &queryTurn{dec: router.Decision{Intent: router.IntentQuery, Utterance: utterance}}
}
// A world question with the workstation asleep says so. The resident model is
// not asked, because what it produces here is an invention with no signal that
// it is one.
func TestQueryGeneralNamesTheGap(t *testing.T) {
g := &gapPhraser{Stub: phraser.NewStub()}
h := &reactiveHandler{phraser: g}
reply, ok := h.queryGeneral(context.Background(), worldTurn("почему небо голубое"))
if !ok {
t.Fatal("queryGeneral passed on the last source in the chain")
}
if reply != worldGap {
t.Fatalf("reply = %q, want the named gap", reply)
}
if g.worldCalls != 1 {
t.Fatalf("PhraseWorld called %d times, want 1", g.worldCalls)
}
}
// A phraser with no world seam at all — the Stub, and every box with no
// `workstation` block — answers exactly as it did before this seam existed.
func TestQueryGeneralWithoutAWorldModelIsUnchanged(t *testing.T) {
h := &reactiveHandler{phraser: phraser.NewStub()}
reply, ok := h.queryGeneral(context.Background(), worldTurn("почему небо голубое"))
if !ok {
t.Fatal("queryGeneral passed on the last source in the chain")
}
if reply != "не знаю." {
t.Fatalf("reply = %q, want the Stub's answer", reply)
}
}
// The gap is spoken aloud by a Russian voice, so it is Russian, feminine and
// informal. "не хочу" and "не могу" are her own verbs; there is no "вы" and no
// English in it.
func TestWorldGapIsInPersona(t *testing.T) {
for _, bad := range []string{"вы", "ваш", "рад ", "дорогой", "милый"} {
if strings.Contains(worldGap, bad) {
t.Errorf("the gap phrase contains %q: %s", bad, worldGap)
}
}
if strings.ContainsAny(worldGap, "abcdefghijklmnopqrstuvwxyz") {
t.Errorf("the gap phrase has Latin letters in it: %s", worldGap)
}
}
// The sources that hold a passage read it back rather than name a gap. He gets a
// real quote instead of "не могу сейчас", and nothing is invented either way.
func TestASourceWithAPassageReadsItBackInsteadOfNamingTheGap(t *testing.T) {
g := &gapPhraser{Stub: phraser.NewStub()}
h := &reactiveHandler{phraser: g}
if got := h.phraseSource(context.Background(), "search", "почему небо голубое",
[]string{"Рэлеевское рассеяние."}); got != "" {
t.Fatalf("phraseSource = %q, want \"\" so the caller's own floor reads the passage back", got)
}
if g.worldCalls != 1 {
t.Fatalf("PhraseWorld called %d times, want 1", g.worldCalls)
}
}
+1 -1
View File
@@ -40,7 +40,7 @@
},
"//workstation": [
"The big model on the desk PC (bugmachine, 7900 GRE 16GB), fronted by",
"The big model on the desk PC (workpc, 7900 GRE 16GB), fronted by",
"mavgpud on port 8080. It runs gemma-4-12b and it is preferred over the",
"resident Qwen3-1.7B for routing and replies whenever the card is free.",
"The machine is never assumed up: it sleeps, and the card is often held by",
+45 -16
View File
@@ -1,6 +1,6 @@
# Offloading model work to the workstation
*Last verified: 2026-08-02 @ 5c05163. Living doc: correct it in place, do not append.*
*Last verified: 2026-08-03 @ 12530c8. Living doc: correct it in place, do not append.*
Owner's call, 2026-08-02. Vikunja #483 is the umbrella. Tasks #484 to #487 are the
work, and this file holds the shape and the rules all four must obey.
@@ -47,6 +47,24 @@ service being down.
Nothing in between. A turn never breaks on the workstation being asleep.
Both halves are wired, 03-08-2026. `LLMPhraser.PhraseWorld`
(`internal/phraser/world.go`) is the naming half and has three outcomes, not two:
| State | What he hears |
|---|---|
| no `workstation` block | the resident model answers, exactly as before the seam existed |
| configured, card free | the workstation answers |
| configured, asleep or busy | the gap, `worldGap` in `cmd/mavend/worldmodel.go` |
The first row is the one worth stating. Naming a gap requires a gap. On a box with
no second model the 1.7B is the whole product. Refusing every world question there
would remove a capability the owner has today.
A source holding a passage is on the naming half too: a live search, a ZIM
article, a page he named. None of them says "не могу сейчас". They read the
passage back, which is what `phraseSource` returning `""` selects. A real quote
beats a gap, and neither path invents.
## Admission control, not a scheduler
There is no GPU arbiter. That is a service with its own failure modes, and nothing
@@ -104,17 +122,29 @@ it buys nothing. Four callers:
## Inventory: what runs a model on homesrv today
The **resident model** is one llama-server with seven callers:
The **resident model** is one llama-server with seven callers, and 03-08-2026 is
the date each of them stopped or did not stop being resident-only:
| Caller | What for |
|---|---|
| `cmd/mavend/voicewire.go` | routing |
| `cmd/mavend/replier_llm.go` | replies |
| `cmd/mavend/tick.go` | digestion worker: `PhraseNudge`, `PhraseReminder` |
| `cmd/mavend/capture.go` | capture summarisation (unreachable, see #480) |
| `cmd/mavend/mail.go` | mail extraction (off, no IMAP) |
| `cmd/mavend/kiwixwire.go` | answering from a Kiwix, search or crawl passage |
| `memoryeval.go`, `modelswap.go` | admin and evals |
| Caller | What for | Offloaded |
|---|---|---|
| `cmd/mavend/voicewire.go` | routing | silently, through `hot` |
| `cmd/mavend/replier_llm.go` | replies | silently, through `hot` |
| `cmd/mavend/tick.go` | digestion worker: `PhraseNudge`, `PhraseReminder` | silently, inside the phraser |
| `cmd/mavend/actions_query.go` | world questions, and any fetched passage | names the gap |
| `cmd/mavend/capture.go` | capture summarisation (unreachable, see #480) | no, holds its own client |
| `cmd/mavend/mail.go` | mail extraction (off, no IMAP) | no, holds its own client |
| `memoryeval.go`, `modelswap.go` | admin and evals | no, and deliberately |
The last three rows are resident-only on purpose. `memoryeval.go` and
`modelswap.go` measure and swap the resident model, so sending their work
elsewhere would measure the wrong thing. `capture.go` and `mail.go` are
background jobs that hold a gated background client (`llmBackgroundClientFor`),
and that priority has no equivalent on the remote yet. Both are also unreachable
on this deploy, so wiring them would ship an untestable path.
The `tick.go` row needs one caveat. `phraser.llm_nudges` is `false` in deploy, so
nudges come from templates and the seam under them changes nothing until that
flips. It is wired anyway: `PhraseReminder` is on the same transport and is on.
Then the embedder above, **whisper.cpp** in `mavsttd`, and **piper** in `mavttsd`.
`mavwaked` uses no model at all: an energy-threshold VAD over 30ms frames.
@@ -125,11 +155,10 @@ Then the embedder above, **whisper.cpp** in `mavsttd`, and **piper** in `mavttsd
`internal/netaddr` landed in PR #92. A seam address now carries its own scheme,
and a scheme-less one is still unix. A tcp seam requires a shared token, because
the filesystem permission that authenticated the unix socket is gone.
2. **The resident model** (#485). Half wired, 02-08-2026. A `workstation` block
builds an `llm.Pair` in `modelSeam` (`cmd/mavend/voicewire.go`), and routing
and replies complete through it. Both are the silent half of the rule. The
naming half is not wired. A world question still goes to the resident model
through `PhraseQuery`. That, and the four callers 485 did not reach, are #490.
2. **The resident model** (#485, #490). Wired. A `workstation` block builds an
`llm.Pair` in `modelSeam` (`cmd/mavend/voicewire.go`), routing and replies
complete through it, and the phraser holds the same pair (`UseRemote`). Both
halves of the rule are live: see the table above for which caller gets which.
Measured, `docs/evals/2026-08-02-workstation-gemma4-12b.md`: gemma-4-12b
through the cascade scores 84.4% full accuracy at p50 329ms. The resident
model scores 72.7% at p50 0.80-1.04s. On the talk fixture it is 25/27
+1 -1
View File
@@ -1110,7 +1110,7 @@ const (
DefaultKiwixSnippetRunes = 1500
)
// WorkstationConfig — the big model on the owner's desktop (bugmachine, a
// WorkstationConfig — the big model on the owner's desktop (workpc, a
// 7900 GRE with 16GB), fronted by mavgpud.
//
// homesrv cannot grow a GPU, so the resident Qwen3-1.7B is the floor and this
+6 -1
View File
@@ -129,6 +129,11 @@ type Req struct {
RepeatPenalty float64
// Stop — sequences that end generation early (e.g. newline for a one-liner).
Stop []string
// Temperature — 0 (the zero value) is greedy decoding, and greedy is what
// every caller here wanted before this field existed. It is set only by the
// phraser, whose own transport has always sampled at 0.7: routing a phrasing
// call through this client must not quietly change how it decodes.
Temperature float64
}
type msg struct {
@@ -176,7 +181,7 @@ func (c *Client) Complete(ctx context.Context, r Req) (string, error) {
Messages: []msg{{Role: "system", Content: r.System}, {Role: "user", Content: r.User}},
MaxTokens: r.MaxTokens,
Grammar: r.Grammar,
Temp: 0,
Temp: r.Temperature,
RepeatPenalty: r.RepeatPenalty,
Stop: r.Stop,
})
+46 -9
View File
@@ -46,6 +46,12 @@ type LLMPhraser struct {
launch func(ctx context.Context, cfg Config) (backend, error)
probe func(ctx context.Context, base string) (string, error)
// remote — the workstation model, when one is configured. Set once at wiring
// time by UseRemote and read on every phrasing call. nil ⇒ every call goes to
// the resident llama-server this phraser owns, which is the whole deploy
// before a `workstation` block exists. See world.go.
remote Remote
// swapMu — single-flight around Swap. Held for the whole swap, including the
// model load, so two concurrent swap requests can never both be loading.
swapMu sync.Mutex
@@ -363,10 +369,7 @@ func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []
// prompt guaranteed to make a small model fill the gap from memory.
notes = nonEmpty(notes)
if len(notes) == 0 {
// General knowledge — no notes to ground the answer. The system
// prompt is the single tested source in router.KnowledgePrompt.
sys := persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt())
prompt := fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance)
sys, prompt := p.knowledgePrompt(utterance)
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
if err != nil || resp == "" {
return "не знаю.", nil
@@ -381,11 +384,7 @@ func (p *LLMPhraser) PhraseQuery(ctx context.Context, utterance string, notes []
}
return resp, nil
}
sys := p.querySystemPrompt()
prompt := fmt.Sprintf(
"Он спрашивает: \"%s\"\n\nИсточники:\n%s\nОтветь ему коротко и своими словами, опираясь только на эти источники. Если ответа в них нет — так и скажи.",
utterance, evidenceBlock(notes),
)
sys, prompt := p.evidencePrompt(utterance, notes)
resp, err := p.chatWithSystem(ctx, sys, prompt, 768)
text, _, perr := parseResponseMood(resp)
if err != nil || perr != nil {
@@ -481,6 +480,17 @@ func chatSystemPrompt(block func() string) string {
// the LLM completion endpoint. Like chatWithSystem but for an arbitrary message
// slice — the caller owns the system prompt placement.
func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTokens int) (string, error) {
// Same silent preference as chatWithSystem, when the array is the shape
// llm.Req can carry: one system turn and one user turn. PhraseChat already
// folds the history into a single user message (some chat templates reject
// consecutive user turns), so today that is every call. A longer array goes
// to the resident model rather than get flattened here, because flattening a
// conversation is a decision its owner should make.
if len(msgs) == 2 && msgs[0].Role == "system" && msgs[1].Role == "user" {
if out, ok := p.remoteChat(ctx, msgs[0].Content, msgs[1].Content, maxTokens); ok {
return out, nil
}
}
base, release, err := p.acquire()
if err != nil {
return "", err
@@ -640,6 +650,12 @@ func (p *LLMPhraser) chat(ctx context.Context, userPrompt string) (string, error
}
func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, maxTokens int) (string, error) {
// The workstation model first when it will take work, and silently: every
// caller of this helper is on the silent half of the degradation rule. It
// answering is not news, and it being asleep is not news either.
if out, ok := p.remoteChat(ctx, system, user, maxTokens); ok {
return out, nil
}
base, release, err := p.acquire()
if err != nil {
return "", err
@@ -733,6 +749,27 @@ func (p *LLMPhraser) systemPrompt() string {
return persona.Prepend(p.cfg.ContextBlock, nudgeSystem)
}
// knowledgePrompt — the no-sources branch: a world question, answered from
// weights alone. The system prompt is the single tested source in
// router.KnowledgePrompt.
//
// Split out of PhraseQuery so PhraseWorld sends the workstation model the same
// bytes the resident model gets. Prompt parity across two models is a stated
// constraint (CLAUDE.md), and two copies of a prompt is how it stops holding.
func (p *LLMPhraser) knowledgePrompt(utterance string) (sys, user string) {
return persona.Prepend(p.cfg.ContextBlock, router.KnowledgePrompt()),
fmt.Sprintf("Пользователь спрашивает: \"%s\".", utterance)
}
// evidencePrompt — the sources branch: read these, add nothing. Shared with
// PhraseWorld for the same reason as knowledgePrompt.
func (p *LLMPhraser) evidencePrompt(utterance string, notes []string) (sys, user string) {
return p.querySystemPrompt(), fmt.Sprintf(
"Он спрашивает: \"%s\"\n\nИсточники:\n%s\nОтветь ему коротко и своими словами, опираясь только на эти источники. Если ответа в них нет — так и скажи.",
utterance, evidenceBlock(notes),
)
}
// querySystemPrompt returns the system prompt for the evidence branch of
// PhraseQuery. Prepends the configured persona when set.
//
+137
View File
@@ -0,0 +1,137 @@
package phraser
import (
"context"
"errors"
"log"
"github.com/kami/maven/internal/llm"
)
// Remote — the workstation model, seen from the phraser. `*llm.Pair` satisfies
// it, and a test fake satisfies it in three lines.
//
// Only the refusing half of Pair is here on purpose. Pair.Complete falls back to
// its own floor client, and the phraser already owns a floor: the llama-server it
// spawned. Two floors under one call is one too many, so the phraser asks whether
// the remote will take work, uses it when it will, and otherwise does exactly
// what it did before this file existed.
type Remote interface {
// Available is an atomic read of a cached probe, so it is free to call per
// turn. See llm.Pair.
Available() bool
// CompleteRemote runs on the workstation or returns ErrRemoteUnavailable. It
// never falls back.
CompleteRemote(ctx context.Context, r llm.Req) (string, error)
}
// ErrNoWorldModel — a world question was asked, a workstation model is
// configured to answer it, and that machine is not answering. The caller turns
// this into a gap he is told about ("не могу сейчас"), never into an answer from
// the resident model.
//
// This is the naming half of the degradation rule in docs/offload.md. The
// resident Qwen3-1.7B does not answer a world question worse than the 12B, it
// invents: measured, the workstation model scores knowledge 9/9 on the talk
// fixture against the resident model's confabulations
// (docs/evals/2026-08-02-workstation-gemma4-12b.md).
var ErrNoWorldModel = errors.New("phraser: no world model available")
// chatTemperature — what the phraser's own transport has always sampled at.
// Named so the remote path cannot drift from it silently. Whether 0.7 is right
// at all is Vikunja #402, and answering that here would hide a phrasing change
// inside a routing change.
const chatTemperature = 0.7
// UseRemote points the phraser at the workstation model. Wiring time only, once,
// before anything phrases: the field is read without a lock on every call
// because a per-turn lock to answer a question that changes at deploy time is
// not worth paying for.
//
// A nil remote is the normal state of a box with no `workstation` block, and it
// must behave exactly as the box behaved before this seam existed.
func (p *LLMPhraser) UseRemote(r Remote) {
p.remote = r
}
// PhraseWorld answers a question about the world — either from the model's own
// knowledge (no sources) or from a passage someone fetched (a live search, a ZIM
// article, a page he named). Three outcomes, and the middle one is the point:
//
// - No workstation configured. The resident model answers, exactly as it does
// today. Naming a gap needs a gap: on a box that never had a second model,
// refusing every world question would remove a capability he has now.
// - Workstation configured and taking work. It answers.
// - Workstation configured and down. ErrNoWorldModel, and the caller says so.
//
// The prompts are the ones PhraseQuery uses, built by the same two functions, so
// the two models are asked the same question in the same words.
func (p *LLMPhraser) PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error) {
sources = nonEmpty(sources)
if p.remote == nil {
return p.PhraseQuery(ctx, utterance, sources)
}
var sys, user string
if len(sources) == 0 {
sys, user = p.knowledgePrompt(utterance)
} else {
sys, user = p.evidencePrompt(utterance, sources)
}
if !p.remote.Available() {
return "", ErrNoWorldModel
}
resp, err := p.remote.CompleteRemote(ctx, llm.Req{
System: sys,
User: user,
Grammar: p.grammar(),
MaxTokens: 768,
Temperature: chatTemperature,
})
if err != nil {
// The cached probe was one interval stale, or the card went away
// mid-request. Either way this is the gap, not an error to log and
// paper over with the smaller model.
log.Printf("phraser: world model: %v", err)
return "", errors.Join(ErrNoWorldModel, err)
}
resp = stripThink(resp)
text, _, perr := parseResponseMood(resp)
if perr != nil {
log.Printf("phraser: PhraseWorld: %v", perr)
return "", errors.Join(ErrNoWorldModel, perr)
}
if text != "" {
return text, nil
}
if resp == "" {
return "", ErrNoWorldModel
}
return resp, nil
}
// remoteChat is the silent half, for the phrasing paths where the workstation
// model is only better: a nudge, a reminder, a reply, a question answered from
// his own notes. It reports whether it answered; it never reports why not,
// because the caller's next move is the resident model either way.
//
// He is not told which of the two models phrased his reply. That is the rule.
func (p *LLMPhraser) remoteChat(ctx context.Context, system, user string, maxTokens int) (string, bool) {
if p.remote == nil || !p.remote.Available() {
return "", false
}
out, err := p.remote.CompleteRemote(ctx, llm.Req{
System: system,
User: user,
Grammar: p.grammar(),
MaxTokens: maxTokens,
Temperature: chatTemperature,
})
if err != nil {
log.Printf("phraser: workstation model declined, phrasing here instead: %v", err)
return "", false
}
if out = stripThink(out); out == "" {
return "", false
}
return out, true
}
+164
View File
@@ -0,0 +1,164 @@
package phraser
import (
"context"
"errors"
"strings"
"testing"
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/loop"
)
// fakeRemote — a workstation model that is up or down on command, and records
// what it was asked.
type fakeRemote struct {
up bool
reply string
err error
got []llm.Req
}
func (f *fakeRemote) Available() bool { return f.up }
func (f *fakeRemote) CompleteRemote(_ context.Context, r llm.Req) (string, error) {
f.got = append(f.got, r)
if f.err != nil {
return "", f.err
}
return f.reply, nil
}
// The three outcomes of the naming half, in one place. The middle one is the
// whole task: a gap he is told about, not an answer from the smaller model.
func TestPhraseWorldNamesTheGapOnlyWhenThereIsOne(t *testing.T) {
answer := `{"response": "Небо голубое из-за рэлеевского рассеяния.", "mood": "neutral"}`
t.Run("no workstation configured: the resident model answers as today", func(t *testing.T) {
spy := newPromptSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{})
got, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil)
if err != nil {
t.Fatalf("PhraseWorld: %v", err)
}
if got == "" {
t.Fatal("no reply from the resident model")
}
if len(spy.user) != 1 {
t.Fatalf("resident model saw %d requests, want 1", len(spy.user))
}
})
t.Run("workstation up: it answers and the resident model is not asked", func(t *testing.T) {
spy := newPromptSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{})
remote := &fakeRemote{up: true, reply: answer}
p.UseRemote(remote)
got, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil)
if err != nil {
t.Fatalf("PhraseWorld: %v", err)
}
if !strings.Contains(got, "рассеяния") {
t.Errorf("reply is not the workstation's: %q", got)
}
if len(spy.user) != 0 {
t.Errorf("the resident model was asked %d times, want 0", len(spy.user))
}
})
t.Run("workstation down: the gap, and nothing invented", func(t *testing.T) {
spy := newPromptSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{})
p.UseRemote(&fakeRemote{up: false})
got, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil)
if !errors.Is(err, ErrNoWorldModel) {
t.Fatalf("err = %v, want ErrNoWorldModel", err)
}
if got != "" {
t.Errorf("got a reply %q with no world model", got)
}
if len(spy.user) != 0 {
t.Errorf("the resident model answered a world question %d times, want 0", len(spy.user))
}
})
t.Run("workstation errors mid-request: still the gap", func(t *testing.T) {
spy := newPromptSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{})
p.UseRemote(&fakeRemote{up: true, err: errors.New("connection refused")})
if _, err := p.PhraseWorld(context.Background(), "почему небо голубое", nil); !errors.Is(err, ErrNoWorldModel) {
t.Fatalf("err = %v, want ErrNoWorldModel", err)
}
if len(spy.user) != 0 {
t.Errorf("the resident model answered a world question %d times, want 0", len(spy.user))
}
})
}
// Prompt parity: the workstation model is asked the same question in the same
// words, or the fixtures measure one thing and the daemon ships another.
func TestPhraseWorldSendsTheSamePromptsAsPhraseQuery(t *testing.T) {
spy := newPromptSpy(t)
resident := NewLLMPhraserAt(spy.srv.URL, Config{})
if _, err := resident.PhraseQuery(context.Background(), "кто написал войну и мир", []string{"Лев Толстой"}); err != nil {
t.Fatal(err)
}
remote := &fakeRemote{up: true, reply: `{"response": "Толстой.", "mood": "neutral"}`}
offloaded := NewLLMPhraserAt(spy.srv.URL, Config{})
offloaded.UseRemote(remote)
if _, err := offloaded.PhraseWorld(context.Background(), "кто написал войну и мир", []string{"Лев Толстой"}); err != nil {
t.Fatal(err)
}
if len(remote.got) != 1 {
t.Fatalf("the workstation saw %d requests, want 1", len(remote.got))
}
if remote.got[0].System != spy.system[0] {
t.Errorf("system prompts differ:\nremote: %q\nresident: %q", remote.got[0].System, spy.system[0])
}
if remote.got[0].User != spy.user[0] {
t.Errorf("user prompts differ:\nremote: %q\nresident: %q", remote.got[0].User, spy.user[0])
}
}
// The silent half. A nudge phrased on the workstation is not news, and one
// phrased here because the card is busy is not news either — but it must be
// sampled the same way, or the workstation quietly changes how she sounds.
func TestNudgePhrasingPrefersTheWorkstationSilently(t *testing.T) {
spy := newPromptSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{LLMNudges: true})
remote := &fakeRemote{up: true, reply: `{"response": "Выпей воды.", "mood": "neutral"}`}
p.UseRemote(remote)
pn, err := p.PhraseNudge(context.Background(), loop.Candidate{Rule: loop.WaterRule(), Severity: loop.Sev1})
if err != nil {
t.Fatalf("PhraseNudge: %v", err)
}
if pn.Body != "Выпей воды." {
t.Errorf("body = %q, want the workstation's wording", pn.Body)
}
if len(remote.got) != 1 {
t.Fatalf("the workstation saw %d requests, want 1", len(remote.got))
}
if remote.got[0].Temperature != chatTemperature {
t.Errorf("temperature = %v, want %v (what the resident transport samples at)",
remote.got[0].Temperature, chatTemperature)
}
if len(spy.user) != 0 {
t.Errorf("the resident model phrased %d nudges, want 0", len(spy.user))
}
}
func TestNudgePhrasingFallsBackWhenTheCardIsBusy(t *testing.T) {
spy := newPromptSpy(t)
p := NewLLMPhraserAt(spy.srv.URL, Config{LLMNudges: true})
p.UseRemote(&fakeRemote{up: false})
if _, err := p.PhraseNudge(context.Background(), loop.Candidate{Rule: loop.WaterRule(), Severity: loop.Sev1}); err != nil {
t.Fatalf("PhraseNudge: %v", err)
}
if len(spy.user) != 1 {
t.Fatalf("the resident model phrased %d nudges, want 1", len(spy.user))
}
}