Files
Maven/cmd/mavend/modelswap.go
T
kami 2bf11f052d Merge branch 'fix/g07' into fix/integrated
# Conflicts:
#	internal/ipc/api.go
#	internal/ipc/client.go
#	internal/llm/client.go
2026-08-01 14:36:48 +04:00

150 lines
5.7 KiB
Go

package main
import (
"context"
"fmt"
"log"
"path/filepath"
"time"
"github.com/kami/maven/internal/config"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/phraser"
)
// Swapping the resident model while the daemon runs (Vikunja #250).
//
// Off unless configured: with no phraser.swap_models allowlist the two IPC
// methods are never wired, so they answer ErrUnknownMethod. When it is wired the
// swap method is AuthStepUp (internal/auth), which means an authed human surface
// only — there is no act, no intent and no timer that reaches it. The daemon
// never decides to change its own brain.
//
// The allowlist is exact-match against paths a human wrote in mavend.json. The
// request carries a path and llama-server is started with it as `-m`, so
// anything looser would turn "swap the model" into "load any file on my disk".
func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
if cfg.Phraser == nil || len(cfg.Phraser.SwapModels) == 0 {
return
}
lp, ok := phr.(*phraser.LLMPhraser)
if !ok {
log.Printf("model swap: phraser.swap_models is set but there is no llama-server phraser — swap disabled")
return
}
allowed := map[string]bool{}
for _, m := range cfg.Phraser.SwapModels {
allowed[filepath.Clean(m)] = true
}
// The configured model is always swappable back to, listed or not: the way
// out of a bad swap must not depend on remembering to allowlist the model
// you are already running.
allowed[filepath.Clean(cfg.Phraser.ModelPath)] = true
srv.SwapModelFn = func(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error) {
path := filepath.Clean(req.ModelPath)
if !allowed[path] {
log.Printf("model swap: REFUSED %q — not in phraser.swap_models", req.ModelPath)
return ipc.SwapModelResp{}, fmt.Errorf("%w: %q is not in phraser.swap_models", ipc.ErrForbidden, req.ModelPath)
}
res, err := lp.Swap(ctx, phraser.SwapSpec{
ModelPath: path,
NGpuLayers: req.NGpuLayers,
NCtx: req.NCtx,
})
resp := ipc.SwapModelResp{
Model: res.Model,
ModelPath: res.ModelPath,
BaseURL: res.BaseURL,
RolledBack: res.RolledBack,
NoBackend: res.NoBackend,
TookMs: res.Took.Milliseconds(),
}
if err != nil {
// A rolled-back swap is a failure that left a working daemon behind.
// Both halves matter to the caller, so the response is filled in even
// though the error is returned.
log.Printf("model swap: %v", err)
return resp, err
}
return resp, nil
}
srv.ModelStatusFn = func(ctx context.Context) (ipc.ModelStatusResp, error) {
path, ngl, nctx := lp.LiveModel()
base := lp.BaseURL()
resp := ipc.ModelStatusResp{
ModelPath: path,
BaseURL: base,
NGpuLayers: ngl,
NCtx: nctx,
Swappable: cfg.Phraser.SwapModels,
}
if base == "" {
resp.Model = llm.UnknownModel
return resp, nil
}
id, err := llm.ModelID(ctx, base)
if err != nil {
// Report the honest "I could not confirm it" rather than echoing the
// configured filename as if the server had said it.
resp.Model = llm.UnknownModel
return resp, nil
}
resp.Model = id
return resp, nil
}
log.Printf("model swap: enabled, %d allowlisted model(s) — step-up required", len(cfg.Phraser.SwapModels))
}
// llmClientFor builds a completion client on the phraser's llama-server and
// keeps it pointed at the right one across a model swap.
//
// Without the OnSwap registration every holder of a base URL — the LLM router,
// the replier, the mail extractor, the memory evaluator — would keep talking to
// the port of a server that no longer exists, and the daemon would degrade to
// the classifier permanently after the first swap. The client is re-pointed, not
// rebuilt, so nothing that holds it has to know a swap happened.
// SetSwapGate is the other half, and on the deploy shape it is the load-bearing
// one:
// llama-server is relaunched on the same fixed port, so SetBaseURL is usually a
// no-op, while the gate is what makes the swap's drain count these callers at
// all. Without it a swap can kill the server mid-routing-decision.
func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
c := llm.New(lp.BaseURL(), timeout)
c.SetGate(residentGate, false)
c.SetSwapGate(lp)
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
return c
}
// backgroundQuiet — how long background work stays off the resident model after
// a foreground request. Long enough to cover the gap between the router call and
// the phraser call of one turn (router p50 is ~2.7s on this box), short enough
// that a quiet mailbox is still read promptly.
const backgroundQuiet = 10 * time.Second
// residentGate — the priority gate on the one llama-server slot, shared by every
// client llmClientFor builds. Package level because the daemon owns exactly one
// llama-server: two gates would be two opinions about one queue.
//
// The problem it solves: llama-server runs a single slot, so requests queue. Mail
// extraction is allowed two minutes, and a first poll can hand core 25 messages
// back to back. Without a gate a voice turn arriving mid-extraction waits for
// whatever is left of that budget, the router times out into the classifier
// cascade at its 36.8% floor, and the phraser just waits.
var residentGate = llm.NewGate(backgroundQuiet)
// llmBackgroundClientFor is llmClientFor for work nobody is waiting on: mail
// extraction and memory evaluation. Same swap-following client, but it yields
// to voice turns and only one such request runs at a time.
func llmBackgroundClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
c := llm.New(lp.BaseURL(), timeout)
c.SetGate(residentGate, true)
c.SetSwapGate(lp)
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
return c
}