Merge branch 'fix/g06' into fix/integrated

# Conflicts:
#	cmd/mavend/memoryeval.go
This commit is contained in:
kami
2026-08-01 14:20:04 +04:00
24 changed files with 1180 additions and 170 deletions
+67 -7
View File
@@ -22,7 +22,9 @@ import (
"context"
"fmt"
"log"
"strings"
"time"
"unicode"
"github.com/kami/maven/internal/config"
"github.com/kami/maven/internal/email"
@@ -37,6 +39,35 @@ import (
// list into a copy of his mailbox.
const evidenceMaxChars = 160
// captureTimeout — how long the capture writes get, separately from the
// extraction budget. A candidate the model already produced must not be lost
// because the model was slow.
const captureTimeout = 30 * time.Second
// maxMailboxChars — a mailbox name is an IMAP folder, not free text. It ends up
// in the provenance string, which is a small controlled vocabulary.
const maxMailboxChars = 64
// validMailbox checks the name this method is willing to write provenance for.
// Empty is refused: "email:" is not a source. So is anything with a control
// character or a space-only value, so the source string stays greppable and
// stays one token.
func validMailbox(s string) (string, error) {
s = strings.TrimSpace(s)
if s == "" {
return "", fmt.Errorf("mail intake: mailbox is required")
}
if len([]rune(s)) > maxMailboxChars {
return "", fmt.Errorf("mail intake: mailbox name too long")
}
for _, r := range s {
if r < 0x20 || r == 0x7f || unicode.IsSpace(r) {
return "", fmt.Errorf("mail intake: mailbox name has whitespace or a control character")
}
}
return s, nil
}
// mailIntake — extraction + capture for one message at a time.
type mailIntake struct {
st *store.Store
@@ -64,15 +95,25 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config, bus
}
lp, ok := phr.(*phraser.LLMPhraser)
if !ok {
log.Printf("mail intake: configured but no llama-server phraser — mail ingestion disabled")
// The phraser is not an *LLMPhraser. Today that means there is no
// llama-server; if anything ever WRAPS the phraser it will mean that
// instead, so the line names the assertion rather than guessing why.
log.Printf("mail intake: configured but the phraser is not an *phraser.LLMPhraser (%T) — mail ingestion disabled", phr)
return nil
}
timeout := time.Duration(cfg.Email.Timeout)
if timeout <= 0 {
timeout = config.DefaultEmailTimeout
}
ex := email.NewExtractor(llmClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", cfg.Email.MaxTasks, timeout)
// Background client: extraction is a job nobody is waiting on, and it shares
// one llama-server slot with the voice turn. Through the gate it yields to
// anything he is waiting for and only one extraction runs at a time, so a
// first poll of 25 unseen messages cannot queue 25 model calls in front of
// him. See llm.Gate.
ex := email.NewExtractor(llmBackgroundClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
// The NORMALISED bound, not the configured one: with "email": {} in
// mavend.json the configured value is 0 and the daemon allows three.
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", ex.Max(), timeout)
return &mailIntake{st: st, ex: ex, timeout: timeout, now: time.Now, bus: bus}
}
@@ -88,6 +129,13 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config, bus
// to the point, a task he already marked done is not re-proposed the next time
// the same unread message is read again.
func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.IngestMailResp, error) {
// The mailbox name becomes provenance ("email:INBOX"), and the source
// vocabulary is what the loop's rules trust. An empty name gave "email:" and
// an arbitrary string gave an arbitrary source under that namespace.
mailbox, err := validMailbox(req.Mailbox)
if err != nil {
return ipc.IngestMailResp{}, err
}
msg := email.Message{
UID: req.UID,
From: req.From,
@@ -100,9 +148,15 @@ func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.Ing
return ipc.IngestMailResp{Skipped: true}, nil
}
ctx, cancel := context.WithTimeout(ctx, m.timeout)
defer cancel()
cands, err := m.ex.Extract(ctx, msg)
// The timeout scopes the EXTRACTION and nothing else. It used to wrap the
// capture writes too, so a model that answered at 119 seconds of a 120
// second budget left the first CaptureTask one second and the third none:
// the work was done, the answer was good, and it was dropped with a
// deadline error. Config calls this a per-message extraction budget, and now
// it is one.
exCtx, cancel := context.WithTimeout(ctx, m.timeout)
cands, err := m.ex.Extract(exCtx, msg)
cancel()
if err != nil {
// The error from internal/email never carries mail text; keep it that way
// by not adding the subject here.
@@ -112,7 +166,13 @@ func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.Ing
return ipc.IngestMailResp{}, nil
}
source := email.SourcePrefix + req.Mailbox
// A fresh budget for the writes, derived from the caller's context rather
// than from the extraction's. Encrypted-store writes are fast; what this
// bounds is a stuck store, not the model.
ctx, cancel = context.WithTimeout(ctx, captureTimeout)
defer cancel()
source := email.SourcePrefix + mailbox
evidence := truncateRunes(req.Subject, evidenceMaxChars)
now := m.now()
var resp ipc.IngestMailResp
+59
View File
@@ -180,3 +180,62 @@ func TestNewMailIntakeOffWithoutConfig(t *testing.T) {
t.Error("without a llama-server phraser there is nothing to extract with")
}
}
// The mailbox name becomes the provenance string, which is the vocabulary the
// loop's rules trust. "email:" is not a source and neither is "email:anything
// he could post at the socket".
func TestIngestRejectsBadMailbox(t *testing.T) {
for _, name := range []string{"", " ", "IN BOX", "IN\nBOX", "IN\x00BOX", strings.Repeat("щ", maxMailboxChars+1)} {
mi, st, fake := newTestIntake(t, `[{"text":"дело","due":""}]`)
req := ingestReq()
req.Mailbox = name
if _, err := mi.ingest(context.Background(), req); err == nil {
t.Errorf("mailbox %q was accepted", name)
}
if fake.calls != 0 {
t.Errorf("mailbox %q reached the model", name)
}
if tasks, _ := st.ListTasks(context.Background(), ""); len(tasks) != 0 {
t.Errorf("mailbox %q wrote %d tasks", name, len(tasks))
}
}
}
// slowLLM burns most of the extraction budget before answering, the way a
// Thinking 1.7B does on a long mail.
type slowLLM struct {
reply string
delay time.Duration
}
func (s *slowLLM) Complete(ctx context.Context, _ llm.Req) (string, error) {
select {
case <-time.After(s.delay):
return s.reply, nil
case <-ctx.Done():
return "", ctx.Err()
}
}
// The extraction budget must not also bound the writes. It used to be one
// context, so a model answering near the deadline lost the candidates it had
// just produced.
func TestIngestCapturesAfterASlowExtraction(t *testing.T) {
st := newTestStore(t)
mi := &mailIntake{
st: st,
ex: email.NewExtractor(&slowLLM{reply: `[{"text":"оплатить счёт","due":""}]`, delay: 90 * time.Millisecond}, 0, nil),
timeout: 100 * time.Millisecond,
now: func() time.Time { return time.Date(2026, 8, 1, 10, 0, 0, 0, time.UTC) },
}
resp, err := mi.ingest(context.Background(), ingestReq())
if err != nil {
t.Fatalf("ingest: %v", err)
}
if resp.Created != 1 {
t.Fatalf("resp = %+v, want the candidate captured", resp)
}
if tasks, _ := st.ListTasks(context.Background(), ""); len(tasks) != 1 {
t.Errorf("got %d tasks, want 1", len(tasks))
}
}
+11 -6
View File
@@ -25,11 +25,14 @@ import (
//
// It used to be five minutes, on the grounds that nobody waits for the answer.
// Nobody waits for the evaluation, but there is ONE resident model behind one
// llama-server, so a voice turn that arrives mid-evaluation waits behind it:
// five minutes of evaluation is five minutes of a mute assistant. Sixty seconds
// is long enough for a Thinking model on this prompt and short enough that the
// worst collision is one turn answered late rather than a turn abandoned. An
// evaluation cut off here costs nothing: it is retried at the next interval.
// llama-server, so a voice turn arriving mid-evaluation waited behind it: five
// minutes of evaluation was five minutes of a mute assistant.
//
// The background client now yields the slot while a turn is in flight, so the
// collision is handled where it belongs and this is a prompt budget again.
// Sixty seconds is long enough for a Thinking model here, and an evaluation cut
// off costs nothing, because it is retried at the next interval. Raise it if
// observations start truncating.
const memoryEvalTimeout = 60 * time.Second
// memoryEvalWorker — ticker + evaluator.
@@ -58,7 +61,9 @@ func newMemoryEvalWorker(st *store.Store, phr phraser.Phraser, cfg *config.Confi
if interval <= 0 {
interval = config.DefaultMemoryEvalInterval
}
client := llmClientFor(lp, memoryEvalTimeout)
// Background: nobody is waiting on an observation, and it must not sit in
// front of a voice turn on the single llama-server slot.
client := llmBackgroundClientFor(lp, memoryEvalTimeout)
ev := memeval.NewEvaluator(st, st, client, memeval.Config{
MaxItems: cfg.MemoryEval.MaxItems,
MinConfidence: cfg.MemoryEval.MinConfidence,
+28
View File
@@ -108,6 +108,34 @@ func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
// rebuilt, so nothing that holds it has to know a swap happened.
func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
c := llm.New(lp.BaseURL(), timeout)
c.SetGate(residentGate, false)
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
return c
}
// backgroundQuiet — how long background work stays off the resident model after
// a foreground request. Long enough to cover the gap between the router call and
// the phraser call of one turn (router p50 is ~2.7s on this box), short enough
// that a quiet mailbox is still read promptly.
const backgroundQuiet = 10 * time.Second
// residentGate — the priority gate on the one llama-server slot, shared by every
// client llmClientFor builds. Package level because the daemon owns exactly one
// llama-server: two gates would be two opinions about one queue.
//
// The problem it solves: llama-server runs a single slot, so requests queue. Mail
// extraction is allowed two minutes, and a first poll can hand core 25 messages
// back to back. Without a gate a voice turn arriving mid-extraction waits for
// whatever is left of that budget, the router times out into the classifier
// cascade at its 36.8% floor, and the phraser just waits.
var residentGate = llm.NewGate(backgroundQuiet)
// llmBackgroundClientFor is llmClientFor for work nobody is waiting on: mail
// extraction and memory evaluation. Same swap-following client, but it yields
// to voice turns and only one such request runs at a time.
func llmBackgroundClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
c := llm.New(lp.BaseURL(), timeout)
c.SetGate(residentGate, true)
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
return c
}