llm: give voice turns priority on the single llama-server slot
llama-server is started without -np, so it serves one request at a time and everything else queues. Mail extraction is allowed two minutes on a Thinking 1.7B, and the reader hands core up to 25 messages back to back. A turn arriving mid-extraction therefore waited for whatever was left of that budget: the router timed out into the classifier cascade and its 36.8% floor, and the phraser, which has no floor, simply waited. Memory evaluation had the same shape with a five minute budget. llm.Gate is the bound. Foreground requests never wait. Background requests run one at a time and yield while a foreground request is in flight, plus a quiet window after it that covers the gap between the router call and the phraser call of one turn. Clients get their priority from llmClientFor or llmBackgroundClientFor, so which side a caller is on is decided at wiring time. It gates only what goes through those clients, which the comment on Gate says. mail intake: the extraction timeout no longer wraps the capture writes. A model answering at 119 seconds of a 120 second budget left the first CaptureTask one second and the third none, so candidates the model had already produced were dropped with a deadline error. The mailbox name is validated before it becomes provenance, since "email:" is not a source and neither is an arbitrary string posted at the socket. The enable log prints the normalised candidate bound rather than the configured one, which said "max 0" and then wrote three. Found in review of #64. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TrVSBKe3RFDF4fGYKWYQnX
This commit is contained in:
+67
-7
@@ -22,7 +22,9 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/email"
|
||||
@@ -37,6 +39,35 @@ import (
|
||||
// list into a copy of his mailbox.
|
||||
const evidenceMaxChars = 160
|
||||
|
||||
// captureTimeout — how long the capture writes get, separately from the
|
||||
// extraction budget. A candidate the model already produced must not be lost
|
||||
// because the model was slow.
|
||||
const captureTimeout = 30 * time.Second
|
||||
|
||||
// maxMailboxChars — a mailbox name is an IMAP folder, not free text. It ends up
|
||||
// in the provenance string, which is a small controlled vocabulary.
|
||||
const maxMailboxChars = 64
|
||||
|
||||
// validMailbox checks the name this method is willing to write provenance for.
|
||||
// Empty is refused: "email:" is not a source. So is anything with a control
|
||||
// character or a space-only value, so the source string stays greppable and
|
||||
// stays one token.
|
||||
func validMailbox(s string) (string, error) {
|
||||
s = strings.TrimSpace(s)
|
||||
if s == "" {
|
||||
return "", fmt.Errorf("mail intake: mailbox is required")
|
||||
}
|
||||
if len([]rune(s)) > maxMailboxChars {
|
||||
return "", fmt.Errorf("mail intake: mailbox name too long")
|
||||
}
|
||||
for _, r := range s {
|
||||
if r < 0x20 || r == 0x7f || unicode.IsSpace(r) {
|
||||
return "", fmt.Errorf("mail intake: mailbox name has whitespace or a control character")
|
||||
}
|
||||
}
|
||||
return s, nil
|
||||
}
|
||||
|
||||
// mailIntake — extraction + capture for one message at a time.
|
||||
type mailIntake struct {
|
||||
st *store.Store
|
||||
@@ -64,15 +95,25 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config, bus
|
||||
}
|
||||
lp, ok := phr.(*phraser.LLMPhraser)
|
||||
if !ok {
|
||||
log.Printf("mail intake: configured but no llama-server phraser — mail ingestion disabled")
|
||||
// The phraser is not an *LLMPhraser. Today that means there is no
|
||||
// llama-server; if anything ever WRAPS the phraser it will mean that
|
||||
// instead, so the line names the assertion rather than guessing why.
|
||||
log.Printf("mail intake: configured but the phraser is not an *phraser.LLMPhraser (%T) — mail ingestion disabled", phr)
|
||||
return nil
|
||||
}
|
||||
timeout := time.Duration(cfg.Email.Timeout)
|
||||
if timeout <= 0 {
|
||||
timeout = config.DefaultEmailTimeout
|
||||
}
|
||||
ex := email.NewExtractor(llmClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
|
||||
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", cfg.Email.MaxTasks, timeout)
|
||||
// Background client: extraction is a job nobody is waiting on, and it shares
|
||||
// one llama-server slot with the voice turn. Through the gate it yields to
|
||||
// anything he is waiting for and only one extraction runs at a time, so a
|
||||
// first poll of 25 unseen messages cannot queue 25 model calls in front of
|
||||
// him. See llm.Gate.
|
||||
ex := email.NewExtractor(llmBackgroundClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
|
||||
// The NORMALISED bound, not the configured one: with "email": {} in
|
||||
// mavend.json the configured value is 0 and the daemon allows three.
|
||||
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", ex.Max(), timeout)
|
||||
return &mailIntake{st: st, ex: ex, timeout: timeout, now: time.Now, bus: bus}
|
||||
}
|
||||
|
||||
@@ -86,6 +127,13 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config, bus
|
||||
// live rows, so a mailbox re-read after a restart produces Created=0 rather
|
||||
// than a second copy of every task.
|
||||
func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.IngestMailResp, error) {
|
||||
// The mailbox name becomes provenance ("email:INBOX"), and the source
|
||||
// vocabulary is what the loop's rules trust. An empty name gave "email:" and
|
||||
// an arbitrary string gave an arbitrary source under that namespace.
|
||||
mailbox, err := validMailbox(req.Mailbox)
|
||||
if err != nil {
|
||||
return ipc.IngestMailResp{}, err
|
||||
}
|
||||
msg := email.Message{
|
||||
UID: req.UID,
|
||||
From: req.From,
|
||||
@@ -98,9 +146,15 @@ func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.Ing
|
||||
return ipc.IngestMailResp{Skipped: true}, nil
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(ctx, m.timeout)
|
||||
defer cancel()
|
||||
cands, err := m.ex.Extract(ctx, msg)
|
||||
// The timeout scopes the EXTRACTION and nothing else. It used to wrap the
|
||||
// capture writes too, so a model that answered at 119 seconds of a 120
|
||||
// second budget left the first CaptureTask one second and the third none:
|
||||
// the work was done, the answer was good, and it was dropped with a
|
||||
// deadline error. Config calls this a per-message extraction budget, and now
|
||||
// it is one.
|
||||
exCtx, cancel := context.WithTimeout(ctx, m.timeout)
|
||||
cands, err := m.ex.Extract(exCtx, msg)
|
||||
cancel()
|
||||
if err != nil {
|
||||
// The error from internal/email never carries mail text; keep it that way
|
||||
// by not adding the subject here.
|
||||
@@ -110,7 +164,13 @@ func (m *mailIntake) ingest(ctx context.Context, req ipc.IngestMailReq) (ipc.Ing
|
||||
return ipc.IngestMailResp{}, nil
|
||||
}
|
||||
|
||||
source := email.SourcePrefix + req.Mailbox
|
||||
// A fresh budget for the writes, derived from the caller's context rather
|
||||
// than from the extraction's. Encrypted-store writes are fast; what this
|
||||
// bounds is a stuck store, not the model.
|
||||
ctx, cancel = context.WithTimeout(ctx, captureTimeout)
|
||||
defer cancel()
|
||||
|
||||
source := email.SourcePrefix + mailbox
|
||||
evidence := truncateRunes(req.Subject, evidenceMaxChars)
|
||||
now := m.now()
|
||||
var resp ipc.IngestMailResp
|
||||
|
||||
Reference in New Issue
Block a user