package main import ( "context" "fmt" "log" "path/filepath" "time" "github.com/kami/maven/internal/config" "github.com/kami/maven/internal/ipc" "github.com/kami/maven/internal/llm" "github.com/kami/maven/internal/phraser" ) // Swapping the resident model while the daemon runs (Vikunja #250). // // Off unless configured: with no phraser.swap_models allowlist the two IPC // methods are never wired, so they answer ErrUnknownMethod. When it is wired the // swap method is AuthStepUp (internal/auth), which means an authed human surface // only — there is no act, no intent and no timer that reaches it. The daemon // never decides to change its own brain. // // The allowlist is exact-match against paths a human wrote in mavend.json. The // request carries a path and llama-server is started with it as `-m`, so // anything looser would turn "swap the model" into "load any file on my disk". func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) { if cfg.Phraser == nil || len(cfg.Phraser.SwapModels) == 0 { return } lp, ok := phr.(*phraser.LLMPhraser) if !ok { log.Printf("model swap: phraser.swap_models is set but there is no llama-server phraser — swap disabled") return } allowed := map[string]bool{} for _, m := range cfg.Phraser.SwapModels { allowed[filepath.Clean(m)] = true } // The configured model is always swappable back to, listed or not: the way // out of a bad swap must not depend on remembering to allowlist the model // you are already running. allowed[filepath.Clean(cfg.Phraser.ModelPath)] = true srv.SwapModelFn = func(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error) { path := filepath.Clean(req.ModelPath) if !allowed[path] { log.Printf("model swap: REFUSED %q — not in phraser.swap_models", req.ModelPath) return ipc.SwapModelResp{}, fmt.Errorf("%w: %q is not in phraser.swap_models", ipc.ErrForbidden, req.ModelPath) } res, err := lp.Swap(ctx, phraser.SwapSpec{ ModelPath: path, NGpuLayers: req.NGpuLayers, NCtx: req.NCtx, }) resp := ipc.SwapModelResp{ Model: res.Model, ModelPath: res.ModelPath, BaseURL: res.BaseURL, RolledBack: res.RolledBack, NoBackend: res.NoBackend, TookMs: res.Took.Milliseconds(), } if err != nil { // A rolled-back swap is a failure that left a working daemon behind. // Both halves matter to the caller, so the response is filled in even // though the error is returned. log.Printf("model swap: %v", err) return resp, err } return resp, nil } srv.ModelStatusFn = func(ctx context.Context) (ipc.ModelStatusResp, error) { path, ngl, nctx := lp.LiveModel() base := lp.BaseURL() resp := ipc.ModelStatusResp{ ModelPath: path, BaseURL: base, NGpuLayers: ngl, NCtx: nctx, Swappable: cfg.Phraser.SwapModels, } if base == "" { resp.Model = llm.UnknownModel return resp, nil } id, err := llm.ModelID(ctx, base) if err != nil { // Report the honest "I could not confirm it" rather than echoing the // configured filename as if the server had said it. resp.Model = llm.UnknownModel return resp, nil } resp.Model = id return resp, nil } log.Printf("model swap: enabled, %d allowlisted model(s) — step-up required", len(cfg.Phraser.SwapModels)) } // llmClientFor builds a completion client on the phraser's llama-server and // keeps it pointed at the right one across a model swap. // // Without the OnSwap registration every holder of a base URL — the LLM router, // the replier, the mail extractor, the memory evaluator — would keep talking to // the port of a server that no longer exists, and the daemon would degrade to // the classifier permanently after the first swap. The client is re-pointed, not // rebuilt, so nothing that holds it has to know a swap happened. // SetSwapGate is the other half, and on the deploy shape it is the load-bearing // one: // llama-server is relaunched on the same fixed port, so SetBaseURL is usually a // no-op, while the gate is what makes the swap's drain count these callers at // all. Without it a swap can kill the server mid-routing-decision. func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client { c := llm.New(lp.BaseURL(), timeout) c.SetGate(residentGate, false) c.SetSwapGate(lp) lp.OnSwap(func(base string) { c.SetBaseURL(base) }) return c } // backgroundQuiet — how long background work stays off the resident model after // a foreground request. Long enough to cover the gap between the router call and // the phraser call of one turn (router p50 is ~2.7s on this box), short enough // that a quiet mailbox is still read promptly. const backgroundQuiet = 10 * time.Second // residentGate — the priority gate on the one llama-server slot, shared by every // client llmClientFor builds. Package level because the daemon owns exactly one // llama-server: two gates would be two opinions about one queue. // // The problem it solves: llama-server runs a single slot, so requests queue. Mail // extraction is allowed two minutes, and a first poll can hand core 25 messages // back to back. Without a gate a voice turn arriving mid-extraction waits for // whatever is left of that budget, the router times out into the classifier // cascade at its 36.8% floor, and the phraser just waits. var residentGate = llm.NewGate(backgroundQuiet) // llmBackgroundClientFor is llmClientFor for work nobody is waiting on: mail // extraction and memory evaluation. Same swap-following client, but it yields // to voice turns and only one such request runs at a time. func llmBackgroundClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client { c := llm.New(lp.BaseURL(), timeout) c.SetGate(residentGate, true) c.SetSwapGate(lp) lp.OnSwap(func(base string) { c.SetBaseURL(base) }) return c }