Swap the resident model without restarting mavend (#250)

Loading a different gguf was a one-line edit to phraser.model_path plus a
restart. It is now an owner-triggered IPC call, off unless configured.

internal/phraser/swap.go holds the safety properties as code:

  - Never two models resident. The old llama-server is killed and reaped
    before the new one is launched. One 1.7B fits the Vega iGPU; a
    blue/green overlap would OOM the box, so it is not offered.
  - Atomic from a turn's point of view. Swap drains the in-flight turns
    (they finish on the old model), then refuses arrivals with ErrSwapping
    until the new server has answered /v1/models. No turn ever sees half a
    swap; refused turns fall back to the classifier cascade.
  - A failed load rolls back. If the new model does not start or does not
    probe, the previous one is reloaded and the call returns RolledBack
    with the error. If the rollback also fails the daemon says so and
    degrades to the classifier rather than pretending to serve.

Holders of the completion client are re-pointed, not rebuilt: llm.Client
guards its base URL and LLMPhraser.OnSwap re-points it, so the router, the
replier, the mail extractor and the memory evaluator follow the new port
without knowing a swap happened.

Reach is deliberately narrow. phraser.swap_models is an exact-match
allowlist of absolute paths a human wrote, rejected at startup otherwise,
so "swap the model" can never mean "load any file on my disk"; the running
model is always swappable back to. MethodSwapModel is AuthStepUp, the same
rung as mutating the tool allowlist, and /models gates POST through the
same stepUpOK the tools page uses. Nothing calls Swap on a timer and no
act, intent or utterance reaches it.

Vikunja #250
This commit is contained in:
kami
2026-08-01 03:59:08 +04:00
parent 2c1b0eede0
commit ad074cea31
22 changed files with 1523 additions and 43 deletions
+150 -34
View File
@@ -27,14 +27,46 @@ var listenRE = regexp.MustCompile(`listening on (https?://\S+)`)
type LLMPhraser struct {
cfg Config
client *http.Client
port string
cmd *exec.Cmd
cancel context.CancelFunc
wg sync.WaitGroup
// tmpl — the hand-written Russian nudges. Default path for nudges; see
// Config.LLMNudges. nil only if the template file failed to load.
tmpl *NudgeTemplates
// spawnCtx — the parent of every llama-server this phraser starts, i.e. the
// daemon's own context. Deliberately NOT the per-request context of the call
// that asked for a model swap: that one is cancelled the moment the request
// returns, which would kill the model it had just loaded.
spawnCtx context.Context
cancel context.CancelFunc
// launch / probe — the two side effects of a swap, injectable so the swap
// logic is testable without a real llama-server and a real model file.
// launch is nil when this phraser does not own its server (NewLLMPhraserAt),
// which is also what makes Swap refuse there.
launch func(ctx context.Context, cfg Config) (backend, error)
probe func(ctx context.Context, base string) (string, error)
// swapMu — single-flight around Swap. Held for the whole swap, including the
// model load, so two concurrent swap requests can never both be loading.
swapMu sync.Mutex
// mu guards everything below: the live backend, the swap gate and the
// in-flight request count. See acquire/quiesce in swap.go.
mu sync.Mutex
be backend
live liveModel
swapping bool
inflight int
observers []func(baseURL string)
}
// liveModel — what is actually loaded right now. Distinct from Config, which
// stays immutable after construction: a swap changes these three fields and
// nothing else, so no reader of cfg (prompts, grammar, timeouts) races a swap.
type liveModel struct {
ModelPath string
NGpuLayers int
NCtx int
}
type Config struct {
@@ -85,15 +117,21 @@ func DefaultConfig(modelPath string) Config {
func NewLLMPhraser(ctx context.Context, cfg Config) (*LLMPhraser, error) {
ctx, cancel := context.WithCancel(ctx)
p := &LLMPhraser{
cfg: cfg,
client: &http.Client{Timeout: cfg.Timeout},
cancel: cancel,
tmpl: loadNudgeTemplates(),
cfg: cfg,
client: &http.Client{Timeout: cfg.Timeout},
tmpl: loadNudgeTemplates(),
spawnCtx: ctx,
cancel: cancel,
launch: spawnLlamaServer,
probe: defaultProbe,
live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
}
if err := p.start(ctx); err != nil {
be, err := p.launch(ctx, cfg)
if err != nil {
cancel()
return nil, err
}
p.be = be
return p, nil
}
@@ -106,11 +144,17 @@ func NewLLMPhraser(ctx context.Context, cfg Config) (*LLMPhraser, error) {
// still uses NewLLMPhraser and still owns its own child process.
func NewLLMPhraserAt(baseURL string, cfg Config) *LLMPhraser {
return &LLMPhraser{
cfg: cfg,
client: &http.Client{Timeout: cfg.Timeout},
port: strings.TrimSuffix(baseURL, "/"),
cancel: func() {},
tmpl: loadNudgeTemplates(),
cfg: cfg,
client: &http.Client{Timeout: cfg.Timeout},
tmpl: loadNudgeTemplates(),
spawnCtx: context.Background(),
cancel: func() {},
probe: defaultProbe,
// launch stays nil: we did not start this server, so we must not stop it.
// Swap therefore refuses here (ErrSwapNotOwned) instead of killing a
// server another process depends on.
be: borrowedBackend(strings.TrimSuffix(baseURL, "/")),
live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
}
}
@@ -126,16 +170,66 @@ func loadNudgeTemplates() *NudgeTemplates {
return nt
}
func (p *LLMPhraser) start(ctx context.Context) error {
// backend — one llama-server this phraser talks to. Two implementations: a
// llamaProc we spawned and must reap, and a borrowedBackend someone else owns.
type backend interface {
BaseURL() string
Close() error
}
// borrowedBackend — a server started and owned by someone else (the phrasing
// scorer's shared llama-server). Closing it is a no-op by construction.
type borrowedBackend string
func (b borrowedBackend) BaseURL() string { return string(b) }
func (b borrowedBackend) Close() error { return nil }
// llamaProc — a llama-server child process plus the goroutine reading its
// stderr. Close kills and reaps it; see the Pdeathsig note in spawnLlamaServer.
type llamaProc struct {
base string
cmd *exec.Cmd
cancel context.CancelFunc
wg sync.WaitGroup
}
func (l *llamaProc) BaseURL() string { return l.base }
func (l *llamaProc) Close() error {
l.cancel()
if l.cmd != nil && l.cmd.Process != nil {
_ = l.cmd.Process.Kill()
_ = l.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
}
l.wg.Wait()
return nil
}
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
// address it is listening on. ctx owns the process lifetime, so it must be the
// daemon's context, not a request's.
func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
ctx, cancel := context.WithCancel(ctx)
p, err := startLlamaProc(ctx, cfg)
if err != nil {
cancel()
return nil, err
}
p.cancel = cancel
return p, nil
}
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
p := &llamaProc{}
args := []string{
"-m", p.cfg.ModelPath,
"-m", cfg.ModelPath,
"--host", "127.0.0.1",
"--port", extractPort(p.cfg.Listen),
"-c", fmt.Sprintf("%d", p.cfg.NCtx),
"-ngl", fmt.Sprintf("%d", p.cfg.NGpuLayers),
"--port", extractPort(cfg.Listen),
"-c", fmt.Sprintf("%d", cfg.NCtx),
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
"--no-webui",
}
cmd := exec.CommandContext(ctx, p.cfg.BinPath, args...)
cmd := exec.CommandContext(ctx, cfg.BinPath, args...)
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
@@ -148,12 +242,12 @@ func (p *LLMPhraser) start(ctx context.Context) error {
stderr, err := cmd.StderrPipe()
if err != nil {
return fmt.Errorf("llm: stderr pipe: %w", err)
return nil, fmt.Errorf("llm: stderr pipe: %w", err)
}
if err := cmd.Start(); err != nil {
stderr.Close()
return fmt.Errorf("llm: start: %w", err)
return nil, fmt.Errorf("llm: start: %w", err)
}
portCh := make(chan string, 1)
@@ -186,32 +280,44 @@ func (p *LLMPhraser) start(ctx context.Context) error {
select {
case addr := <-portCh:
p.port = addr
return nil
p.base = addr
return p, nil
case err := <-errCh:
_ = cmd.Process.Kill()
_ = cmd.Wait()
return fmt.Errorf("llm: server output: %w", err)
return nil, fmt.Errorf("llm: server output: %w", err)
case <-ctx.Done():
_ = cmd.Process.Kill()
_ = cmd.Wait()
return ctx.Err()
return nil, ctx.Err()
case <-time.After(60 * time.Second):
_ = cmd.Process.Kill()
_ = cmd.Wait()
return fmt.Errorf("llm: server did not start within 60s")
return nil, fmt.Errorf("llm: server did not start within 60s")
}
}
func (p *LLMPhraser) BaseURL() string { return p.port }
// BaseURL is the llama-server this phraser talks to right now. It changes when
// the model is swapped, so callers that cache it must register an observer
// (OnSwap) rather than keeping the string forever.
func (p *LLMPhraser) BaseURL() string {
p.mu.Lock()
defer p.mu.Unlock()
if p.be == nil {
return ""
}
return p.be.BaseURL()
}
func (p *LLMPhraser) Close() error {
p.cancel()
if p.cmd != nil && p.cmd.Process != nil {
_ = p.cmd.Process.Kill()
_ = p.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
p.mu.Lock()
be := p.be
p.be = nil
p.mu.Unlock()
if be != nil {
return be.Close()
}
p.wg.Wait()
return nil
}
@@ -350,6 +456,11 @@ func chatSystemPrompt(block func() string) string {
// the LLM completion endpoint. Like chatWithSystem but for an arbitrary message
// slice — the caller owns the system prompt placement.
func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTokens int) (string, error) {
base, release, err := p.acquire()
if err != nil {
return "", err
}
defer release()
req := chatReq{
Messages: msgs,
Temperature: 0.7,
@@ -360,7 +471,7 @@ func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTo
if err != nil {
return "", fmt.Errorf("llm: marshal: %w", err)
}
httpReq, err := http.NewRequestWithContext(ctx, "POST", p.port+"/v1/chat/completions", bytes.NewReader(body))
httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body))
if err != nil {
return "", fmt.Errorf("llm: request: %w", err)
}
@@ -493,6 +604,11 @@ func (p *LLMPhraser) chat(ctx context.Context, userPrompt string) (string, error
}
func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, maxTokens int) (string, error) {
base, release, err := p.acquire()
if err != nil {
return "", err
}
defer release()
req := chatReq{
Messages: []chatMsg{
{Role: "system", Content: system},
@@ -507,7 +623,7 @@ func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, ma
return "", fmt.Errorf("llm: marshal: %w", err)
}
httpReq, err := http.NewRequestWithContext(ctx, "POST", p.port+"/v1/chat/completions", bytes.NewReader(body))
httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body))
if err != nil {
return "", fmt.Errorf("llm: request: %w", err)
}