Merge branch 'fix/g06' into fix/integrated
# Conflicts: # cmd/mavend/memoryeval.go
This commit is contained in:
+67
-7
@@ -22,7 +22,9 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/email"
|
||||
@@ -37,6 +39,35 @@ import (
|
||||
// list into a copy of his mailbox.
|
||||
const evidenceMaxChars = 160
|
||||
|
||||
// captureTimeout — how long the capture writes get, separately from the
|
||||
// extraction budget. A candidate the model already produced must not be lost
|
||||
// because the model was slow.
|
||||
const captureTimeout = 30 * time.Second
|
||||
|
||||
// maxMailboxChars — a mailbox name is an IMAP folder, not free text. It ends up
|
||||
// in the provenance string, which is a small controlled vocabulary.
|
||||
const maxMailboxChars = 64
|
||||
|
||||
// validMailbox checks the name this method is willing to write provenance for.
|
||||
// Empty is refused: "email:" is not a source. So is anything with a control
|
||||
// character or a space-only value, so the source string stays greppable and
|
||||
// stays one token.
|
||||
func validMailbox(s string) (string, error) {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return "", fmt.Errorf("mail intake: mailbox is required")
|
||||
}
|
||||
if len([]rune(s)) > maxMailboxChars {
|
||||
return "", fmt.Errorf("mail intake: mailbox name too long")
|
||||
}
|
||||
for _, r := range s {
|
||||
if r < 0x20 || r == 0x7f || unicode.IsSpace(r) {
|
||||
return "", fmt.Errorf("mail intake: mailbox name has whitespace or a control character")
|
||||
}
|
||||
}
|
||||
return s, nil
|
||||
}
|
||||
|
||||
// mailIntake — extraction + capture for one message at a time.
|
||||
type mailIntake struct {
|
||||
st *store.Store
|
||||
@@ -64,15 +95,25 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config, bus
|
||||
}
|
||||
lp, ok := phr.(*phraser.LLMPhraser)
|
||||
if !ok {
|
||||
log.Printf("mail intake: configured but no llama-server phraser — mail ingestion disabled")
|
||||
// The phraser is not an *LLMPhraser. Today that means there is no
|
||||
// llama-server; if anything ever WRAPS the phraser it will mean that
|
||||
// instead, so the line names the assertion rather than guessing why.
|
||||
log.Printf("mail intake: configured but the phraser is not an *phraser.LLMPhraser (%T) — mail ingestion disabled", phr)
|
||||
return nil
|
||||
}
|
||||
timeout := time.Duration(cfg.Email.Timeout)
|
||||
if timeout <= 0 {
|
||||
timeout = config.DefaultEmailTimeout
|
||||
}
|
||||
ex := email.NewExtractor(llmClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
|
||||
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", cfg.Email.MaxTasks, timeout)
|
||||
// Background client: extraction is a job nobody is waiting on, and it shares
|
||||
// one llama-server slot with the voice turn. Through the gate it yields to
|
||||
// anything he is waiting for and only one extraction runs at a time, so a
|
||||
// first poll of 25 unseen messages cannot queue 25 model calls in front of
|
||||
// him. See llm.Gate.
|
||||
ex := email.NewExtractor(llmBackgroundClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
|
||||
// The NORMALISED bound, not the configured one: with "email": {} in
|
||||
// mavend.json the configured value is 0 and the daemon allows three.
|
||||
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", ex.Max(), timeout)
|
||||
return &mailIntake{st: st, ex: ex, timeout: timeout, now: time.Now, bus: bus}
|
||||
}
|
||||
|
||||
@@ -88,6 +129,13 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config, bus
|
||||
// to the point, a task he already marked done is not re-proposed the next time
|
||||
// the same unread message is read again.
|
||||
func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.IngestMailResp, error) {
|
||||
// The mailbox name becomes provenance ("email:INBOX"), and the source
|
||||
// vocabulary is what the loop's rules trust. An empty name gave "email:" and
|
||||
// an arbitrary string gave an arbitrary source under that namespace.
|
||||
mailbox, err := validMailbox(req.Mailbox)
|
||||
if err != nil {
|
||||
return ipc.IngestMailResp{}, err
|
||||
}
|
||||
msg := email.Message{
|
||||
UID: req.UID,
|
||||
From: req.From,
|
||||
@@ -100,9 +148,15 @@ func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.Ing
|
||||
return ipc.IngestMailResp{Skipped: true}, nil
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(ctx, m.timeout)
|
||||
defer cancel()
|
||||
cands, err := m.ex.Extract(ctx, msg)
|
||||
// The timeout scopes the EXTRACTION and nothing else. It used to wrap the
|
||||
// capture writes too, so a model that answered at 119 seconds of a 120
|
||||
// second budget left the first CaptureTask one second and the third none:
|
||||
// the work was done, the answer was good, and it was dropped with a
|
||||
// deadline error. Config calls this a per-message extraction budget, and now
|
||||
// it is one.
|
||||
exCtx, cancel := context.WithTimeout(ctx, m.timeout)
|
||||
cands, err := m.ex.Extract(exCtx, msg)
|
||||
cancel()
|
||||
if err != nil {
|
||||
// The error from internal/email never carries mail text; keep it that way
|
||||
// by not adding the subject here.
|
||||
@@ -112,7 +166,13 @@ func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.Ing
|
||||
return ipc.IngestMailResp{}, nil
|
||||
}
|
||||
|
||||
source := email.SourcePrefix + req.Mailbox
|
||||
// A fresh budget for the writes, derived from the caller's context rather
|
||||
// than from the extraction's. Encrypted-store writes are fast; what this
|
||||
// bounds is a stuck store, not the model.
|
||||
ctx, cancel = context.WithTimeout(ctx, captureTimeout)
|
||||
defer cancel()
|
||||
|
||||
source := email.SourcePrefix + mailbox
|
||||
evidence := truncateRunes(req.Subject, evidenceMaxChars)
|
||||
now := m.now()
|
||||
var resp ipc.IngestMailResp
|
||||
|
||||
@@ -180,3 +180,62 @@ func TestNewMailIntakeOffWithoutConfig(t *testing.T) {
|
||||
t.Error("without a llama-server phraser there is nothing to extract with")
|
||||
}
|
||||
}
|
||||
|
||||
// The mailbox name becomes the provenance string, which is the vocabulary the
|
||||
// loop's rules trust. "email:" is not a source and neither is "email:anything
|
||||
// he could post at the socket".
|
||||
func TestIngestRejectsBadMailbox(t *testing.T) {
|
||||
for _, name := range []string{"", " ", "IN BOX", "IN\nBOX", "IN\x00BOX", strings.Repeat("щ", maxMailboxChars+1)} {
|
||||
mi, st, fake := newTestIntake(t, `[{"text":"дело","due":""}]`)
|
||||
req := ingestReq()
|
||||
req.Mailbox = name
|
||||
if _, err := mi.ingest(context.Background(), req); err == nil {
|
||||
t.Errorf("mailbox %q was accepted", name)
|
||||
}
|
||||
if fake.calls != 0 {
|
||||
t.Errorf("mailbox %q reached the model", name)
|
||||
}
|
||||
if tasks, _ := st.ListTasks(context.Background(), ""); len(tasks) != 0 {
|
||||
t.Errorf("mailbox %q wrote %d tasks", name, len(tasks))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// slowLLM burns most of the extraction budget before answering, the way a
|
||||
// Thinking 1.7B does on a long mail.
|
||||
type slowLLM struct {
|
||||
reply string
|
||||
delay time.Duration
|
||||
}
|
||||
|
||||
func (s *slowLLM) Complete(ctx context.Context, _ llm.Req) (string, error) {
|
||||
select {
|
||||
case <-time.After(s.delay):
|
||||
return s.reply, nil
|
||||
case <-ctx.Done():
|
||||
return "", ctx.Err()
|
||||
}
|
||||
}
|
||||
|
||||
// The extraction budget must not also bound the writes. It used to be one
|
||||
// context, so a model answering near the deadline lost the candidates it had
|
||||
// just produced.
|
||||
func TestIngestCapturesAfterASlowExtraction(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
mi := &mailIntake{
|
||||
st: st,
|
||||
ex: email.NewExtractor(&slowLLM{reply: `[{"text":"оплатить счёт","due":""}]`, delay: 90 * time.Millisecond}, 0, nil),
|
||||
timeout: 100 * time.Millisecond,
|
||||
now: func() time.Time { return time.Date(2026, 8, 1, 10, 0, 0, 0, time.UTC) },
|
||||
}
|
||||
resp, err := mi.ingest(context.Background(), ingestReq())
|
||||
if err != nil {
|
||||
t.Fatalf("ingest: %v", err)
|
||||
}
|
||||
if resp.Created != 1 {
|
||||
t.Fatalf("resp = %+v, want the candidate captured", resp)
|
||||
}
|
||||
if tasks, _ := st.ListTasks(context.Background(), ""); len(tasks) != 1 {
|
||||
t.Errorf("got %d tasks, want 1", len(tasks))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -25,11 +25,14 @@ import (
|
||||
//
|
||||
// It used to be five minutes, on the grounds that nobody waits for the answer.
|
||||
// Nobody waits for the evaluation, but there is ONE resident model behind one
|
||||
// llama-server, so a voice turn that arrives mid-evaluation waits behind it:
|
||||
// five minutes of evaluation is five minutes of a mute assistant. Sixty seconds
|
||||
// is long enough for a Thinking model on this prompt and short enough that the
|
||||
// worst collision is one turn answered late rather than a turn abandoned. An
|
||||
// evaluation cut off here costs nothing: it is retried at the next interval.
|
||||
// llama-server, so a voice turn arriving mid-evaluation waited behind it: five
|
||||
// minutes of evaluation was five minutes of a mute assistant.
|
||||
//
|
||||
// The background client now yields the slot while a turn is in flight, so the
|
||||
// collision is handled where it belongs and this is a prompt budget again.
|
||||
// Sixty seconds is long enough for a Thinking model here, and an evaluation cut
|
||||
// off costs nothing, because it is retried at the next interval. Raise it if
|
||||
// observations start truncating.
|
||||
const memoryEvalTimeout = 60 * time.Second
|
||||
|
||||
// memoryEvalWorker — ticker + evaluator.
|
||||
@@ -58,7 +61,9 @@ func newMemoryEvalWorker(st *store.Store, phr phraser.Phraser, cfg *config.Confi
|
||||
if interval <= 0 {
|
||||
interval = config.DefaultMemoryEvalInterval
|
||||
}
|
||||
client := llmClientFor(lp, memoryEvalTimeout)
|
||||
// Background: nobody is waiting on an observation, and it must not sit in
|
||||
// front of a voice turn on the single llama-server slot.
|
||||
client := llmBackgroundClientFor(lp, memoryEvalTimeout)
|
||||
ev := memeval.NewEvaluator(st, st, client, memeval.Config{
|
||||
MaxItems: cfg.MemoryEval.MaxItems,
|
||||
MinConfidence: cfg.MemoryEval.MinConfidence,
|
||||
|
||||
@@ -108,6 +108,34 @@ func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
|
||||
// rebuilt, so nothing that holds it has to know a swap happened.
|
||||
func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
||||
c := llm.New(lp.BaseURL(), timeout)
|
||||
c.SetGate(residentGate, false)
|
||||
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
||||
return c
|
||||
}
|
||||
|
||||
// backgroundQuiet — how long background work stays off the resident model after
|
||||
// a foreground request. Long enough to cover the gap between the router call and
|
||||
// the phraser call of one turn (router p50 is ~2.7s on this box), short enough
|
||||
// that a quiet mailbox is still read promptly.
|
||||
const backgroundQuiet = 10 * time.Second
|
||||
|
||||
// residentGate — the priority gate on the one llama-server slot, shared by every
|
||||
// client llmClientFor builds. Package level because the daemon owns exactly one
|
||||
// llama-server: two gates would be two opinions about one queue.
|
||||
//
|
||||
// The problem it solves: llama-server runs a single slot, so requests queue. Mail
|
||||
// extraction is allowed two minutes, and a first poll can hand core 25 messages
|
||||
// back to back. Without a gate a voice turn arriving mid-extraction waits for
|
||||
// whatever is left of that budget, the router times out into the classifier
|
||||
// cascade at its 36.8% floor, and the phraser just waits.
|
||||
var residentGate = llm.NewGate(backgroundQuiet)
|
||||
|
||||
// llmBackgroundClientFor is llmClientFor for work nobody is waiting on: mail
|
||||
// extraction and memory evaluation. Same swap-following client, but it yields
|
||||
// to voice turns and only one such request runs at a time.
|
||||
func llmBackgroundClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
||||
c := llm.New(lp.BaseURL(), timeout)
|
||||
c.SetGate(residentGate, true)
|
||||
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
||||
return c
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user