aee20a6abc
llama-server is started without -np, so it serves one request at a time and everything else queues. Mail extraction is allowed two minutes on a Thinking 1.7B, and the reader hands core up to 25 messages back to back. A turn arriving mid-extraction therefore waited for whatever was left of that budget: the router timed out into the classifier cascade and its 36.8% floor, and the phraser, which has no floor, simply waited. Memory evaluation had the same shape with a five minute budget. llm.Gate is the bound. Foreground requests never wait. Background requests run one at a time and yield while a foreground request is in flight, plus a quiet window after it that covers the gap between the router call and the phraser call of one turn. Clients get their priority from llmClientFor or llmBackgroundClientFor, so which side a caller is on is decided at wiring time. It gates only what goes through those clients, which the comment on Gate says. mail intake: the extraction timeout no longer wraps the capture writes. A model answering at 119 seconds of a 120 second budget left the first CaptureTask one second and the third none, so candidates the model had already produced were dropped with a deadline error. The mailbox name is validated before it becomes provenance, since "email:" is not a source and neither is an arbitrary string posted at the socket. The enable log prints the normalised candidate bound rather than the configured one, which said "max 0" and then wrote three. Found in review of #64. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TrVSBKe3RFDF4fGYKWYQnX
142 lines
5.4 KiB
Go
142 lines
5.4 KiB
Go
package main
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"log"
|
|
"path/filepath"
|
|
"time"
|
|
|
|
"github.com/kami/maven/internal/config"
|
|
"github.com/kami/maven/internal/ipc"
|
|
"github.com/kami/maven/internal/llm"
|
|
"github.com/kami/maven/internal/phraser"
|
|
)
|
|
|
|
// Swapping the resident model while the daemon runs (Vikunja #250).
|
|
//
|
|
// Off unless configured: with no phraser.swap_models allowlist the two IPC
|
|
// methods are never wired, so they answer ErrUnknownMethod. When it is wired the
|
|
// swap method is AuthStepUp (internal/auth), which means an authed human surface
|
|
// only — there is no act, no intent and no timer that reaches it. The daemon
|
|
// never decides to change its own brain.
|
|
//
|
|
// The allowlist is exact-match against paths a human wrote in mavend.json. The
|
|
// request carries a path and llama-server is started with it as `-m`, so
|
|
// anything looser would turn "swap the model" into "load any file on my disk".
|
|
func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
|
|
if cfg.Phraser == nil || len(cfg.Phraser.SwapModels) == 0 {
|
|
return
|
|
}
|
|
lp, ok := phr.(*phraser.LLMPhraser)
|
|
if !ok {
|
|
log.Printf("model swap: phraser.swap_models is set but there is no llama-server phraser — swap disabled")
|
|
return
|
|
}
|
|
allowed := map[string]bool{}
|
|
for _, m := range cfg.Phraser.SwapModels {
|
|
allowed[filepath.Clean(m)] = true
|
|
}
|
|
// The configured model is always swappable back to, listed or not: the way
|
|
// out of a bad swap must not depend on remembering to allowlist the model
|
|
// you are already running.
|
|
allowed[filepath.Clean(cfg.Phraser.ModelPath)] = true
|
|
|
|
srv.SwapModelFn = func(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error) {
|
|
path := filepath.Clean(req.ModelPath)
|
|
if !allowed[path] {
|
|
log.Printf("model swap: REFUSED %q — not in phraser.swap_models", req.ModelPath)
|
|
return ipc.SwapModelResp{}, fmt.Errorf("%w: %q is not in phraser.swap_models", ipc.ErrForbidden, req.ModelPath)
|
|
}
|
|
res, err := lp.Swap(ctx, phraser.SwapSpec{
|
|
ModelPath: path,
|
|
NGpuLayers: req.NGpuLayers,
|
|
NCtx: req.NCtx,
|
|
})
|
|
resp := ipc.SwapModelResp{
|
|
Model: res.Model,
|
|
ModelPath: res.ModelPath,
|
|
BaseURL: res.BaseURL,
|
|
RolledBack: res.RolledBack,
|
|
TookMs: res.Took.Milliseconds(),
|
|
}
|
|
if err != nil {
|
|
// A rolled-back swap is a failure that left a working daemon behind.
|
|
// Both halves matter to the caller, so the response is filled in even
|
|
// though the error is returned.
|
|
log.Printf("model swap: %v", err)
|
|
return resp, err
|
|
}
|
|
return resp, nil
|
|
}
|
|
|
|
srv.ModelStatusFn = func(ctx context.Context) (ipc.ModelStatusResp, error) {
|
|
path, ngl, nctx := lp.LiveModel()
|
|
base := lp.BaseURL()
|
|
resp := ipc.ModelStatusResp{
|
|
ModelPath: path,
|
|
BaseURL: base,
|
|
NGpuLayers: ngl,
|
|
NCtx: nctx,
|
|
Swappable: cfg.Phraser.SwapModels,
|
|
}
|
|
if base == "" {
|
|
resp.Model = llm.UnknownModel
|
|
return resp, nil
|
|
}
|
|
id, err := llm.ModelID(ctx, base)
|
|
if err != nil {
|
|
// Report the honest "I could not confirm it" rather than echoing the
|
|
// configured filename as if the server had said it.
|
|
resp.Model = llm.UnknownModel
|
|
return resp, nil
|
|
}
|
|
resp.Model = id
|
|
return resp, nil
|
|
}
|
|
|
|
log.Printf("model swap: enabled, %d allowlisted model(s) — step-up required", len(cfg.Phraser.SwapModels))
|
|
}
|
|
|
|
// llmClientFor builds a completion client on the phraser's llama-server and
|
|
// keeps it pointed at the right one across a model swap.
|
|
//
|
|
// Without the OnSwap registration every holder of a base URL — the LLM router,
|
|
// the replier, the mail extractor, the memory evaluator — would keep talking to
|
|
// the port of a server that no longer exists, and the daemon would degrade to
|
|
// the classifier permanently after the first swap. The client is re-pointed, not
|
|
// rebuilt, so nothing that holds it has to know a swap happened.
|
|
func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
|
c := llm.New(lp.BaseURL(), timeout)
|
|
c.SetGate(residentGate, false)
|
|
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
|
return c
|
|
}
|
|
|
|
// backgroundQuiet — how long background work stays off the resident model after
|
|
// a foreground request. Long enough to cover the gap between the router call and
|
|
// the phraser call of one turn (router p50 is ~2.7s on this box), short enough
|
|
// that a quiet mailbox is still read promptly.
|
|
const backgroundQuiet = 10 * time.Second
|
|
|
|
// residentGate — the priority gate on the one llama-server slot, shared by every
|
|
// client llmClientFor builds. Package level because the daemon owns exactly one
|
|
// llama-server: two gates would be two opinions about one queue.
|
|
//
|
|
// The problem it solves: llama-server runs a single slot, so requests queue. Mail
|
|
// extraction is allowed two minutes, and a first poll can hand core 25 messages
|
|
// back to back. Without a gate a voice turn arriving mid-extraction waits for
|
|
// whatever is left of that budget, the router times out into the classifier
|
|
// cascade at its 36.8% floor, and the phraser just waits.
|
|
var residentGate = llm.NewGate(backgroundQuiet)
|
|
|
|
// llmBackgroundClientFor is llmClientFor for work nobody is waiting on: mail
|
|
// extraction and memory evaluation. Same swap-following client, but it yields
|
|
// to voice turns and only one such request runs at a time.
|
|
func llmBackgroundClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
|
c := llm.New(lp.BaseURL(), timeout)
|
|
c.SetGate(residentGate, true)
|
|
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
|
return c
|
|
}
|