Swap the resident model without restarting mavend (#250)
Loading a different gguf was a one-line edit to phraser.model_path plus a
restart. It is now an owner-triggered IPC call, off unless configured.
internal/phraser/swap.go holds the safety properties as code:
- Never two models resident. The old llama-server is killed and reaped
before the new one is launched. One 1.7B fits the Vega iGPU; a
blue/green overlap would OOM the box, so it is not offered.
- Atomic from a turn's point of view. Swap drains the in-flight turns
(they finish on the old model), then refuses arrivals with ErrSwapping
until the new server has answered /v1/models. No turn ever sees half a
swap; refused turns fall back to the classifier cascade.
- A failed load rolls back. If the new model does not start or does not
probe, the previous one is reloaded and the call returns RolledBack
with the error. If the rollback also fails the daemon says so and
degrades to the classifier rather than pretending to serve.
Holders of the completion client are re-pointed, not rebuilt: llm.Client
guards its base URL and LLMPhraser.OnSwap re-points it, so the router, the
replier, the mail extractor and the memory evaluator follow the new port
without knowing a swap happened.
Reach is deliberately narrow. phraser.swap_models is an exact-match
allowlist of absolute paths a human wrote, rejected at startup otherwise,
so "swap the model" can never mean "load any file on my disk"; the running
model is always swappable back to. MethodSwapModel is AuthStepUp, the same
rung as mutating the tool allowlist, and /models gates POST through the
same stepUpOK the tools page uses. Nothing calls Swap on a timer and no
act, intent or utterance reaches it.
Vikunja #250
This commit is contained in:
+150
-34
@@ -27,14 +27,46 @@ var listenRE = regexp.MustCompile(`listening on (https?://\S+)`)
|
||||
type LLMPhraser struct {
|
||||
cfg Config
|
||||
client *http.Client
|
||||
port string
|
||||
cmd *exec.Cmd
|
||||
cancel context.CancelFunc
|
||||
wg sync.WaitGroup
|
||||
|
||||
// tmpl — the hand-written Russian nudges. Default path for nudges; see
|
||||
// Config.LLMNudges. nil only if the template file failed to load.
|
||||
tmpl *NudgeTemplates
|
||||
|
||||
// spawnCtx — the parent of every llama-server this phraser starts, i.e. the
|
||||
// daemon's own context. Deliberately NOT the per-request context of the call
|
||||
// that asked for a model swap: that one is cancelled the moment the request
|
||||
// returns, which would kill the model it had just loaded.
|
||||
spawnCtx context.Context
|
||||
cancel context.CancelFunc
|
||||
|
||||
// launch / probe — the two side effects of a swap, injectable so the swap
|
||||
// logic is testable without a real llama-server and a real model file.
|
||||
// launch is nil when this phraser does not own its server (NewLLMPhraserAt),
|
||||
// which is also what makes Swap refuse there.
|
||||
launch func(ctx context.Context, cfg Config) (backend, error)
|
||||
probe func(ctx context.Context, base string) (string, error)
|
||||
|
||||
// swapMu — single-flight around Swap. Held for the whole swap, including the
|
||||
// model load, so two concurrent swap requests can never both be loading.
|
||||
swapMu sync.Mutex
|
||||
|
||||
// mu guards everything below: the live backend, the swap gate and the
|
||||
// in-flight request count. See acquire/quiesce in swap.go.
|
||||
mu sync.Mutex
|
||||
be backend
|
||||
live liveModel
|
||||
swapping bool
|
||||
inflight int
|
||||
observers []func(baseURL string)
|
||||
}
|
||||
|
||||
// liveModel — what is actually loaded right now. Distinct from Config, which
|
||||
// stays immutable after construction: a swap changes these three fields and
|
||||
// nothing else, so no reader of cfg (prompts, grammar, timeouts) races a swap.
|
||||
type liveModel struct {
|
||||
ModelPath string
|
||||
NGpuLayers int
|
||||
NCtx int
|
||||
}
|
||||
|
||||
type Config struct {
|
||||
@@ -85,15 +117,21 @@ func DefaultConfig(modelPath string) Config {
|
||||
func NewLLMPhraser(ctx context.Context, cfg Config) (*LLMPhraser, error) {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
p := &LLMPhraser{
|
||||
cfg: cfg,
|
||||
client: &http.Client{Timeout: cfg.Timeout},
|
||||
cancel: cancel,
|
||||
tmpl: loadNudgeTemplates(),
|
||||
cfg: cfg,
|
||||
client: &http.Client{Timeout: cfg.Timeout},
|
||||
tmpl: loadNudgeTemplates(),
|
||||
spawnCtx: ctx,
|
||||
cancel: cancel,
|
||||
launch: spawnLlamaServer,
|
||||
probe: defaultProbe,
|
||||
live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
|
||||
}
|
||||
if err := p.start(ctx); err != nil {
|
||||
be, err := p.launch(ctx, cfg)
|
||||
if err != nil {
|
||||
cancel()
|
||||
return nil, err
|
||||
}
|
||||
p.be = be
|
||||
return p, nil
|
||||
}
|
||||
|
||||
@@ -106,11 +144,17 @@ func NewLLMPhraser(ctx context.Context, cfg Config) (*LLMPhraser, error) {
|
||||
// still uses NewLLMPhraser and still owns its own child process.
|
||||
func NewLLMPhraserAt(baseURL string, cfg Config) *LLMPhraser {
|
||||
return &LLMPhraser{
|
||||
cfg: cfg,
|
||||
client: &http.Client{Timeout: cfg.Timeout},
|
||||
port: strings.TrimSuffix(baseURL, "/"),
|
||||
cancel: func() {},
|
||||
tmpl: loadNudgeTemplates(),
|
||||
cfg: cfg,
|
||||
client: &http.Client{Timeout: cfg.Timeout},
|
||||
tmpl: loadNudgeTemplates(),
|
||||
spawnCtx: context.Background(),
|
||||
cancel: func() {},
|
||||
probe: defaultProbe,
|
||||
// launch stays nil: we did not start this server, so we must not stop it.
|
||||
// Swap therefore refuses here (ErrSwapNotOwned) instead of killing a
|
||||
// server another process depends on.
|
||||
be: borrowedBackend(strings.TrimSuffix(baseURL, "/")),
|
||||
live: liveModel{ModelPath: cfg.ModelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -126,16 +170,66 @@ func loadNudgeTemplates() *NudgeTemplates {
|
||||
return nt
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) start(ctx context.Context) error {
|
||||
// backend — one llama-server this phraser talks to. Two implementations: a
|
||||
// llamaProc we spawned and must reap, and a borrowedBackend someone else owns.
|
||||
type backend interface {
|
||||
BaseURL() string
|
||||
Close() error
|
||||
}
|
||||
|
||||
// borrowedBackend — a server started and owned by someone else (the phrasing
|
||||
// scorer's shared llama-server). Closing it is a no-op by construction.
|
||||
type borrowedBackend string
|
||||
|
||||
func (b borrowedBackend) BaseURL() string { return string(b) }
|
||||
func (b borrowedBackend) Close() error { return nil }
|
||||
|
||||
// llamaProc — a llama-server child process plus the goroutine reading its
|
||||
// stderr. Close kills and reaps it; see the Pdeathsig note in spawnLlamaServer.
|
||||
type llamaProc struct {
|
||||
base string
|
||||
cmd *exec.Cmd
|
||||
cancel context.CancelFunc
|
||||
wg sync.WaitGroup
|
||||
}
|
||||
|
||||
func (l *llamaProc) BaseURL() string { return l.base }
|
||||
|
||||
func (l *llamaProc) Close() error {
|
||||
l.cancel()
|
||||
if l.cmd != nil && l.cmd.Process != nil {
|
||||
_ = l.cmd.Process.Kill()
|
||||
_ = l.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
|
||||
}
|
||||
l.wg.Wait()
|
||||
return nil
|
||||
}
|
||||
|
||||
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
|
||||
// address it is listening on. ctx owns the process lifetime, so it must be the
|
||||
// daemon's context, not a request's.
|
||||
func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
p, err := startLlamaProc(ctx, cfg)
|
||||
if err != nil {
|
||||
cancel()
|
||||
return nil, err
|
||||
}
|
||||
p.cancel = cancel
|
||||
return p, nil
|
||||
}
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
p := &llamaProc{}
|
||||
args := []string{
|
||||
"-m", p.cfg.ModelPath,
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
"--port", extractPort(p.cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", p.cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", p.cfg.NGpuLayers),
|
||||
"--port", extractPort(cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}
|
||||
cmd := exec.CommandContext(ctx, p.cfg.BinPath, args...)
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, args...)
|
||||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||||
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
|
||||
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
|
||||
@@ -148,12 +242,12 @@ func (p *LLMPhraser) start(ctx context.Context) error {
|
||||
|
||||
stderr, err := cmd.StderrPipe()
|
||||
if err != nil {
|
||||
return fmt.Errorf("llm: stderr pipe: %w", err)
|
||||
return nil, fmt.Errorf("llm: stderr pipe: %w", err)
|
||||
}
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
stderr.Close()
|
||||
return fmt.Errorf("llm: start: %w", err)
|
||||
return nil, fmt.Errorf("llm: start: %w", err)
|
||||
}
|
||||
|
||||
portCh := make(chan string, 1)
|
||||
@@ -186,32 +280,44 @@ func (p *LLMPhraser) start(ctx context.Context) error {
|
||||
|
||||
select {
|
||||
case addr := <-portCh:
|
||||
p.port = addr
|
||||
return nil
|
||||
p.base = addr
|
||||
return p, nil
|
||||
case err := <-errCh:
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return fmt.Errorf("llm: server output: %w", err)
|
||||
return nil, fmt.Errorf("llm: server output: %w", err)
|
||||
case <-ctx.Done():
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return ctx.Err()
|
||||
return nil, ctx.Err()
|
||||
case <-time.After(60 * time.Second):
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return fmt.Errorf("llm: server did not start within 60s")
|
||||
return nil, fmt.Errorf("llm: server did not start within 60s")
|
||||
}
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) BaseURL() string { return p.port }
|
||||
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
||||
// the model is swapped, so callers that cache it must register an observer
|
||||
// (OnSwap) rather than keeping the string forever.
|
||||
func (p *LLMPhraser) BaseURL() string {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
if p.be == nil {
|
||||
return ""
|
||||
}
|
||||
return p.be.BaseURL()
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) Close() error {
|
||||
p.cancel()
|
||||
if p.cmd != nil && p.cmd.Process != nil {
|
||||
_ = p.cmd.Process.Kill()
|
||||
_ = p.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
|
||||
p.mu.Lock()
|
||||
be := p.be
|
||||
p.be = nil
|
||||
p.mu.Unlock()
|
||||
if be != nil {
|
||||
return be.Close()
|
||||
}
|
||||
p.wg.Wait()
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -350,6 +456,11 @@ func chatSystemPrompt(block func() string) string {
|
||||
// the LLM completion endpoint. Like chatWithSystem but for an arbitrary message
|
||||
// slice — the caller owns the system prompt placement.
|
||||
func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTokens int) (string, error) {
|
||||
base, release, err := p.acquire()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer release()
|
||||
req := chatReq{
|
||||
Messages: msgs,
|
||||
Temperature: 0.7,
|
||||
@@ -360,7 +471,7 @@ func (p *LLMPhraser) chatWithMessages(ctx context.Context, msgs []chatMsg, maxTo
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("llm: marshal: %w", err)
|
||||
}
|
||||
httpReq, err := http.NewRequestWithContext(ctx, "POST", p.port+"/v1/chat/completions", bytes.NewReader(body))
|
||||
httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body))
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("llm: request: %w", err)
|
||||
}
|
||||
@@ -493,6 +604,11 @@ func (p *LLMPhraser) chat(ctx context.Context, userPrompt string) (string, error
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, maxTokens int) (string, error) {
|
||||
base, release, err := p.acquire()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer release()
|
||||
req := chatReq{
|
||||
Messages: []chatMsg{
|
||||
{Role: "system", Content: system},
|
||||
@@ -507,7 +623,7 @@ func (p *LLMPhraser) chatWithSystem(ctx context.Context, system, user string, ma
|
||||
return "", fmt.Errorf("llm: marshal: %w", err)
|
||||
}
|
||||
|
||||
httpReq, err := http.NewRequestWithContext(ctx, "POST", p.port+"/v1/chat/completions", bytes.NewReader(body))
|
||||
httpReq, err := http.NewRequestWithContext(ctx, "POST", base+"/v1/chat/completions", bytes.NewReader(body))
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("llm: request: %w", err)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,330 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
)
|
||||
|
||||
// Swapping the resident model without restarting the daemon (Vikunja #250).
|
||||
//
|
||||
// Three properties this file exists to hold, in order of importance:
|
||||
//
|
||||
// 1. NEVER two models resident at once. The deploy target is a laptop iGPU
|
||||
// with the whole 1.7B offloaded to it (`n_gpu_layers: 99`); loading a second
|
||||
// model beside the first is how you OOM the box, and a blue/green swap that
|
||||
// "keeps the old one warm until the new one answers" does exactly that. So
|
||||
// the old server is killed FIRST and the new one loaded after. The cost of
|
||||
// that ordering is a window with no model at all, which is why:
|
||||
//
|
||||
// 2. A swap is atomic from a turn's point of view. An in-flight turn finishes
|
||||
// on the old model — Swap waits for the last one to return before killing
|
||||
// anything. A turn that arrives during the swap is REFUSED immediately with
|
||||
// ErrSwapping rather than blocked: every phrasing path already has a
|
||||
// fallback (templates, "вот что я нашла", the classifier for routing), so a
|
||||
// fast refusal degrades one turn instead of hanging it for the length of a
|
||||
// model load. No turn ever gets half of one model and half of another.
|
||||
//
|
||||
// 3. A failed load rolls back to the model that was working. The new server is
|
||||
// probed (it must say which model it loaded) before it is published; if the
|
||||
// launch or the probe fails, the previous config is relaunched and the
|
||||
// phraser goes back to serving. Only if the rollback ALSO fails is the
|
||||
// phraser left without a backend, and then it says so loudly and every turn
|
||||
// degrades rather than breaks.
|
||||
//
|
||||
// Not here, deliberately: nothing calls Swap on a timer, and no act or intent can
|
||||
// reach it. It is an IPC method behind the step-up gate, i.e. owner-triggered.
|
||||
|
||||
var (
|
||||
// ErrSwapping — a turn arrived while the model was being swapped. Callers
|
||||
// treat it like any other LLM error and use their fallback.
|
||||
ErrSwapping = errors.New("phraser: model swap in progress")
|
||||
|
||||
// ErrSwapNotOwned — this phraser did not start its llama-server, so it must
|
||||
// not stop one (NewLLMPhraserAt: the eval harness shares a server).
|
||||
ErrSwapNotOwned = errors.New("phraser: llama-server is not ours to swap")
|
||||
|
||||
// ErrNoBackend — no model is loaded at all. Only reachable after a failed
|
||||
// swap whose rollback also failed.
|
||||
ErrNoBackend = errors.New("phraser: no llama-server loaded")
|
||||
|
||||
// ErrSwapBusy — a turn was still running when the drain deadline expired, so
|
||||
// the swap was abandoned. Nothing was killed; ask again.
|
||||
ErrSwapBusy = errors.New("phraser: turns still in flight, swap abandoned")
|
||||
)
|
||||
|
||||
// SwapSpec — what to load. Zero NGpuLayers/NCtx keep whatever is live, so the
|
||||
// common case ("same settings, different gguf") is one field.
|
||||
type SwapSpec struct {
|
||||
ModelPath string
|
||||
NGpuLayers int
|
||||
NCtx int
|
||||
}
|
||||
|
||||
// SwapResult — what happened. Model is the identity the NEW server reported, so
|
||||
// it is evidence rather than an echo of the request: if the file at ModelPath is
|
||||
// not what the operator thought it was, this is where that shows up.
|
||||
type SwapResult struct {
|
||||
Model string
|
||||
BaseURL string
|
||||
ModelPath string
|
||||
RolledBack bool
|
||||
Took time.Duration
|
||||
}
|
||||
|
||||
// drainTimeout — how long Swap waits for in-flight turns before giving up. A
|
||||
// turn is at most Config.Timeout (30s in deploy) plus the model's own latency;
|
||||
// 90s covers a slow Thinking generation without wedging the caller forever.
|
||||
const drainTimeout = 90 * time.Second
|
||||
|
||||
// probeTimeout — how long the new server gets to answer "which model do you
|
||||
// have". The load itself is bounded by spawnLlamaServer's own 60s wait.
|
||||
const probeTimeout = 30 * time.Second
|
||||
|
||||
// defaultProbe asks the server which model it has loaded. This is the health
|
||||
// check: a server that answers /v1/models has finished loading weights and is
|
||||
// serving, and its answer is the identity we report back.
|
||||
func defaultProbe(ctx context.Context, base string) (string, error) {
|
||||
return llm.ModelID(ctx, base)
|
||||
}
|
||||
|
||||
// OnSwap registers a callback fired with the new base URL every time the live
|
||||
// backend changes, including after a rollback. Holders of an *llm.Client (the
|
||||
// LLM router, the replier, the mail extractor) register SetBaseURL here so a
|
||||
// swap re-points them without rebuilding the router or the handler.
|
||||
//
|
||||
// Callbacks run with no lock held, in registration order.
|
||||
func (p *LLMPhraser) OnSwap(fn func(baseURL string)) {
|
||||
if fn == nil {
|
||||
return
|
||||
}
|
||||
p.mu.Lock()
|
||||
p.observers = append(p.observers, fn)
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// LiveModel is the model file currently loaded (and its load settings). Empty
|
||||
// ModelPath means no model is loaded.
|
||||
func (p *LLMPhraser) LiveModel() (path string, nGpuLayers, nCtx int) {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
return p.live.ModelPath, p.live.NGpuLayers, p.live.NCtx
|
||||
}
|
||||
|
||||
// acquire reserves a slot for one request and returns the base URL to use.
|
||||
// Every request path must call it and must call the returned release exactly
|
||||
// once — that count is what Swap drains.
|
||||
func (p *LLMPhraser) acquire() (string, func(), error) {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
if p.swapping {
|
||||
return "", nil, ErrSwapping
|
||||
}
|
||||
if p.be == nil {
|
||||
return "", nil, ErrNoBackend
|
||||
}
|
||||
p.inflight++
|
||||
base := p.be.BaseURL()
|
||||
var once bool
|
||||
return base, func() {
|
||||
p.mu.Lock()
|
||||
if !once {
|
||||
once = true
|
||||
p.inflight--
|
||||
}
|
||||
p.mu.Unlock()
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Swap loads another model in place of the live one. See the file comment for
|
||||
// the properties it guarantees. Returns the new model's reported identity, or
|
||||
// an error plus RolledBack=true when the old model was put back.
|
||||
//
|
||||
// ctx bounds the drain and the probe. It does NOT own the new server's lifetime
|
||||
// — that is the daemon's context, captured at construction — so a swap survives
|
||||
// the request that asked for it.
|
||||
func (p *LLMPhraser) Swap(ctx context.Context, spec SwapSpec) (SwapResult, error) {
|
||||
if spec.ModelPath == "" {
|
||||
return SwapResult{}, fmt.Errorf("phraser: swap needs a model path")
|
||||
}
|
||||
p.swapMu.Lock()
|
||||
defer p.swapMu.Unlock()
|
||||
|
||||
if p.launch == nil {
|
||||
return SwapResult{}, ErrSwapNotOwned
|
||||
}
|
||||
|
||||
started := time.Now()
|
||||
oldLive := p.liveSnapshot()
|
||||
newLive := liveModel{
|
||||
ModelPath: spec.ModelPath,
|
||||
NGpuLayers: pickInt(spec.NGpuLayers, oldLive.NGpuLayers),
|
||||
NCtx: pickInt(spec.NCtx, oldLive.NCtx),
|
||||
}
|
||||
if newLive == oldLive && p.BaseURL() != "" {
|
||||
// Already serving exactly this. Report the live identity rather than
|
||||
// pointlessly unloading and reloading the same weights.
|
||||
base := p.BaseURL()
|
||||
id, err := p.probeWith(ctx, base)
|
||||
if err != nil {
|
||||
return SwapResult{}, err
|
||||
}
|
||||
return SwapResult{Model: id, BaseURL: base, ModelPath: oldLive.ModelPath, Took: time.Since(started)}, nil
|
||||
}
|
||||
|
||||
if err := p.quiesce(ctx); err != nil {
|
||||
return SwapResult{}, err
|
||||
}
|
||||
defer p.resume()
|
||||
|
||||
// Property 1: the old model leaves the GPU before the new one arrives.
|
||||
p.mu.Lock()
|
||||
old := p.be
|
||||
p.be = nil
|
||||
p.mu.Unlock()
|
||||
if old != nil {
|
||||
_ = old.Close()
|
||||
}
|
||||
|
||||
be, err := p.loadAndProbe(ctx, newLive)
|
||||
if err != nil {
|
||||
log.Printf("phraser: swap to %s FAILED (%v) — rolling back to %s", newLive.ModelPath, err, oldLive.ModelPath)
|
||||
rb, rbErr := p.loadAndProbe(ctx, oldLive)
|
||||
if rbErr != nil {
|
||||
log.Printf("phraser: ROLLBACK to %s ALSO FAILED (%v) — no model is loaded, every phrasing path is on its fallback and routing is on the classifier until the daemon is restarted", oldLive.ModelPath, rbErr)
|
||||
return SwapResult{RolledBack: true, Took: time.Since(started)},
|
||||
fmt.Errorf("phraser: swap failed (%w) and rollback failed too: %v", err, rbErr)
|
||||
}
|
||||
p.publish(rb, oldLive)
|
||||
return SwapResult{
|
||||
Model: rb.id, BaseURL: rb.be.BaseURL(), ModelPath: oldLive.ModelPath,
|
||||
RolledBack: true, Took: time.Since(started),
|
||||
},
|
||||
fmt.Errorf("phraser: swap to %s failed, rolled back to %s: %w", newLive.ModelPath, oldLive.ModelPath, err)
|
||||
}
|
||||
p.publish(be, newLive)
|
||||
log.Printf("phraser: model swapped to %s (%s) at %s in %s", newLive.ModelPath, be.id, be.be.BaseURL(), time.Since(started).Round(time.Millisecond))
|
||||
return SwapResult{
|
||||
Model: be.id, BaseURL: be.be.BaseURL(), ModelPath: newLive.ModelPath,
|
||||
Took: time.Since(started),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// loaded — a started server plus the identity it reported.
|
||||
type loaded struct {
|
||||
be backend
|
||||
id string
|
||||
}
|
||||
|
||||
// loadAndProbe starts a server for lm and verifies it answers. A server that
|
||||
// starts but will not say what it loaded is treated as a failed load and is
|
||||
// killed here — publishing it would hand every turn to a backend we could not
|
||||
// confirm.
|
||||
func (p *LLMPhraser) loadAndProbe(ctx context.Context, lm liveModel) (loaded, error) {
|
||||
cfg := p.cfg
|
||||
cfg.ModelPath = lm.ModelPath
|
||||
cfg.NGpuLayers = lm.NGpuLayers
|
||||
cfg.NCtx = lm.NCtx
|
||||
// p.spawnCtx, not ctx: the process must outlive the request asking for it.
|
||||
be, err := p.launch(p.spawnCtx, cfg)
|
||||
if err != nil {
|
||||
return loaded{}, err
|
||||
}
|
||||
id, err := p.probeWith(ctx, be.BaseURL())
|
||||
if err != nil {
|
||||
_ = be.Close()
|
||||
return loaded{}, fmt.Errorf("phraser: %s started but would not answer: %w", lm.ModelPath, err)
|
||||
}
|
||||
return loaded{be: be, id: id}, nil
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) probeWith(ctx context.Context, base string) (string, error) {
|
||||
probe := p.probe
|
||||
if probe == nil {
|
||||
probe = defaultProbe
|
||||
}
|
||||
pctx, cancel := context.WithTimeout(ctx, probeTimeout)
|
||||
defer cancel()
|
||||
return probe(pctx, base)
|
||||
}
|
||||
|
||||
// quiesce closes the door on new turns and waits for the ones already running.
|
||||
// Polling rather than a sync.Cond: the wait happens once per swap, a 25ms poll
|
||||
// is invisible next to a model load, and a poll cannot deadlock on a release
|
||||
// path that panicked.
|
||||
func (p *LLMPhraser) quiesce(ctx context.Context) error {
|
||||
p.mu.Lock()
|
||||
if p.swapping {
|
||||
p.mu.Unlock()
|
||||
return ErrSwapping
|
||||
}
|
||||
p.swapping = true
|
||||
inflight := p.inflight
|
||||
p.mu.Unlock()
|
||||
if inflight == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
deadline := time.Now().Add(drainTimeout)
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
p.resume()
|
||||
return ctx.Err()
|
||||
case <-time.After(25 * time.Millisecond):
|
||||
}
|
||||
p.mu.Lock()
|
||||
inflight = p.inflight
|
||||
p.mu.Unlock()
|
||||
if inflight == 0 {
|
||||
return nil
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
// Nothing has been killed yet, so abandoning is free: reopen the door
|
||||
// and let the operator try again rather than cutting a live turn off
|
||||
// mid-generation.
|
||||
p.resume()
|
||||
return fmt.Errorf("%w (%d still running after %s)", ErrSwapBusy, inflight, drainTimeout)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) resume() {
|
||||
p.mu.Lock()
|
||||
p.swapping = false
|
||||
p.mu.Unlock()
|
||||
}
|
||||
|
||||
// publish makes l the live backend and tells everyone holding a base URL.
|
||||
func (p *LLMPhraser) publish(l loaded, lm liveModel) {
|
||||
p.mu.Lock()
|
||||
p.be = l.be
|
||||
p.live = lm
|
||||
obs := make([]func(string), len(p.observers))
|
||||
copy(obs, p.observers)
|
||||
p.mu.Unlock()
|
||||
base := l.be.BaseURL()
|
||||
for _, fn := range obs {
|
||||
fn(base)
|
||||
}
|
||||
}
|
||||
|
||||
func (p *LLMPhraser) liveSnapshot() liveModel {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
return p.live
|
||||
}
|
||||
|
||||
// pickInt returns v when the caller set it, and fallback otherwise. 0 is the
|
||||
// "unset" value: -1 already means "offload every layer" and deploy uses 99, so
|
||||
// nothing legitimate asks for exactly zero GPU layers through this path.
|
||||
func pickInt(v, fallback int) int {
|
||||
if v == 0 {
|
||||
return fallback
|
||||
}
|
||||
return v
|
||||
}
|
||||
@@ -0,0 +1,311 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// fakeModel — a stand-in llama-server. It answers /v1/models with its own name
|
||||
// and /v1/chat/completions with a phrasing-contract reply that names itself, so
|
||||
// a test can tell WHICH model answered a turn — the property the swap is about.
|
||||
type fakeModel struct {
|
||||
srv *httptest.Server
|
||||
name string
|
||||
closed atomic.Bool
|
||||
}
|
||||
|
||||
func newFakeModel(t *testing.T, name string) *fakeModel {
|
||||
t.Helper()
|
||||
f := &fakeModel{name: name}
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/v1/models", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Write([]byte(`{"data":[{"id":"/models/` + name + `.gguf"}]}`))
|
||||
})
|
||||
mux.HandleFunc("/v1/chat/completions", func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Write([]byte(`{"choices":[{"message":{"content":"{\"response\":\"` + name + `\",\"mood\":\"neutral\"}"}}]}`))
|
||||
})
|
||||
f.srv = httptest.NewServer(mux)
|
||||
t.Cleanup(f.srv.Close)
|
||||
return f
|
||||
}
|
||||
|
||||
func (f *fakeModel) BaseURL() string { return f.srv.URL }
|
||||
func (f *fakeModel) Close() error { f.closed.Store(true); return nil }
|
||||
|
||||
// fakeFleet is the injected launcher: it hands out a prepared fakeModel per
|
||||
// model path, and refuses paths the test did not prepare (that is what a bad
|
||||
// gguf looks like from here). It also asserts the invariant that matters on a
|
||||
// laptop iGPU: never two servers alive at the same time.
|
||||
type fakeFleet struct {
|
||||
mu sync.Mutex
|
||||
models map[string]string // model path → fake name
|
||||
live int
|
||||
maxLive int
|
||||
launch int
|
||||
}
|
||||
|
||||
func (fl *fakeFleet) launcher(t *testing.T) func(context.Context, Config) (backend, error) {
|
||||
return func(ctx context.Context, cfg Config) (backend, error) {
|
||||
fl.mu.Lock()
|
||||
name, ok := fl.models[cfg.ModelPath]
|
||||
fl.launch++
|
||||
if !ok {
|
||||
fl.mu.Unlock()
|
||||
return nil, errors.New("no such model file: " + cfg.ModelPath)
|
||||
}
|
||||
fl.live++
|
||||
if fl.live > fl.maxLive {
|
||||
fl.maxLive = fl.live
|
||||
}
|
||||
fl.mu.Unlock()
|
||||
f := newFakeModel(t, name)
|
||||
return &fleetBackend{fleet: fl, model: f}, nil
|
||||
}
|
||||
}
|
||||
|
||||
type fleetBackend struct {
|
||||
fleet *fakeFleet
|
||||
model *fakeModel
|
||||
once sync.Once
|
||||
}
|
||||
|
||||
func (b *fleetBackend) BaseURL() string { return b.model.BaseURL() }
|
||||
func (b *fleetBackend) Close() error {
|
||||
b.once.Do(func() {
|
||||
b.fleet.mu.Lock()
|
||||
b.fleet.live--
|
||||
b.fleet.mu.Unlock()
|
||||
})
|
||||
return b.model.Close()
|
||||
}
|
||||
|
||||
// newSwapPhraser builds an LLMPhraser with an injected launcher, so the swap
|
||||
// path is exercised without a gguf or a GPU.
|
||||
func newSwapPhraser(t *testing.T, fl *fakeFleet, modelPath string) *LLMPhraser {
|
||||
t.Helper()
|
||||
cfg := DefaultConfig(modelPath)
|
||||
cfg.Timeout = 5 * time.Second
|
||||
p := &LLMPhraser{
|
||||
cfg: cfg,
|
||||
client: &http.Client{Timeout: cfg.Timeout},
|
||||
spawnCtx: context.Background(),
|
||||
cancel: func() {},
|
||||
launch: fl.launcher(t),
|
||||
probe: defaultProbe,
|
||||
live: liveModel{ModelPath: modelPath, NGpuLayers: cfg.NGpuLayers, NCtx: cfg.NCtx},
|
||||
}
|
||||
be, err := p.launch(p.spawnCtx, cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("initial launch: %v", err)
|
||||
}
|
||||
p.be = be
|
||||
t.Cleanup(func() { p.Close() })
|
||||
return p
|
||||
}
|
||||
|
||||
func TestSwap_LoadsNewModelAndRepointsHolders(t *testing.T) {
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old", "/m/new.gguf": "new"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
|
||||
// A holder of the base URL (the LLM router's client, in the daemon).
|
||||
var seen []string
|
||||
p.OnSwap(func(base string) { seen = append(seen, base) })
|
||||
|
||||
before, err := p.PhraseChat(context.Background(), "привет", nil)
|
||||
if err != nil || before != "old" {
|
||||
t.Fatalf("before swap: %q, %v; want the old model to answer", before, err)
|
||||
}
|
||||
|
||||
res, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/new.gguf"})
|
||||
if err != nil {
|
||||
t.Fatalf("Swap: %v", err)
|
||||
}
|
||||
if res.Model != "new" {
|
||||
t.Errorf("res.Model = %q; want the identity the NEW server reported (%q)", res.Model, "new")
|
||||
}
|
||||
if res.RolledBack {
|
||||
t.Errorf("res.RolledBack = true on a successful swap")
|
||||
}
|
||||
after, err := p.PhraseChat(context.Background(), "привет", nil)
|
||||
if err != nil || after != "new" {
|
||||
t.Fatalf("after swap: %q, %v; want the new model to answer", after, err)
|
||||
}
|
||||
if path, _, _ := p.LiveModel(); path != "/m/new.gguf" {
|
||||
t.Errorf("LiveModel = %q; want /m/new.gguf", path)
|
||||
}
|
||||
if len(seen) != 1 || seen[0] != p.BaseURL() {
|
||||
t.Errorf("observers saw %v; want exactly one call with the new base %q", seen, p.BaseURL())
|
||||
}
|
||||
if fl.maxLive > 1 {
|
||||
t.Errorf("%d servers were alive at once; the iGPU only fits one model", fl.maxLive)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_FailedLoadRollsBackToTheWorkingModel(t *testing.T) {
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
|
||||
res, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/broken.gguf"})
|
||||
if err == nil {
|
||||
t.Fatal("Swap to a model that will not load returned nil error")
|
||||
}
|
||||
if !res.RolledBack {
|
||||
t.Errorf("res.RolledBack = false; a failed swap must say it rolled back")
|
||||
}
|
||||
if res.Model != "old" {
|
||||
t.Errorf("res.Model = %q; want the old model back", res.Model)
|
||||
}
|
||||
// The point of the rollback: turns keep working.
|
||||
got, err := p.PhraseChat(context.Background(), "привет", nil)
|
||||
if err != nil || got != "old" {
|
||||
t.Fatalf("after rollback: %q, %v; want the old model serving again", got, err)
|
||||
}
|
||||
if path, _, _ := p.LiveModel(); path != "/m/old.gguf" {
|
||||
t.Errorf("LiveModel = %q; want the old model", path)
|
||||
}
|
||||
if fl.maxLive > 1 {
|
||||
t.Errorf("%d servers alive at once during a rollback", fl.maxLive)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_ProbeFailureIsTreatedAsAFailedLoad(t *testing.T) {
|
||||
// A server that starts but will not say what it loaded must never be
|
||||
// published — we would be serving turns from a backend we cannot confirm.
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old", "/m/mute.gguf": "mute"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
// Fail the probe once — for the newly launched server — and let the
|
||||
// rollback's probe through.
|
||||
calls := 0
|
||||
p.probe = func(ctx context.Context, base string) (string, error) {
|
||||
calls++
|
||||
if calls == 1 {
|
||||
return "", errors.New("no answer from the new server")
|
||||
}
|
||||
return defaultProbe(ctx, base)
|
||||
}
|
||||
|
||||
_, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/mute.gguf"})
|
||||
if err == nil {
|
||||
t.Fatal("Swap published a server that failed its probe")
|
||||
}
|
||||
if path, _, _ := p.LiveModel(); path != "/m/old.gguf" {
|
||||
t.Errorf("LiveModel = %q; want the old model after a failed probe", path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_RollbackFailureLeavesNoBackendAndDegrades(t *testing.T) {
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
// Make the rollback fail too: the old file "disappears" mid-swap.
|
||||
fl.mu.Lock()
|
||||
delete(fl.models, "/m/old.gguf")
|
||||
fl.mu.Unlock()
|
||||
|
||||
_, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/broken.gguf"})
|
||||
if err == nil {
|
||||
t.Fatal("Swap returned nil when both the load and the rollback failed")
|
||||
}
|
||||
// Nothing is loaded, and the request path says so rather than panicking.
|
||||
if _, _, aerr := p.acquire(); !errors.Is(aerr, ErrNoBackend) {
|
||||
t.Errorf("acquire error = %v; want ErrNoBackend", aerr)
|
||||
}
|
||||
// Phrasing degrades to its fallback instead of failing the turn.
|
||||
got, err := p.PhraseChat(context.Background(), "привет", nil)
|
||||
if err != nil {
|
||||
t.Fatalf("PhraseChat after a total failure returned an error: %v", err)
|
||||
}
|
||||
if got == "" {
|
||||
t.Error("PhraseChat returned empty; the fallback must still say something")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_WaitsForInFlightTurnAndRefusesNewOnes(t *testing.T) {
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old", "/m/new.gguf": "new"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
|
||||
// Hold one turn open by taking a slot directly — the same slot every
|
||||
// request path takes.
|
||||
base, release, err := p.acquire()
|
||||
if err != nil {
|
||||
t.Fatalf("acquire: %v", err)
|
||||
}
|
||||
if base == "" {
|
||||
t.Fatal("acquire returned an empty base URL")
|
||||
}
|
||||
|
||||
swapped := make(chan error, 1)
|
||||
go func() { _, e := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/new.gguf"}); swapped <- e }()
|
||||
|
||||
// While the swap waits to drain, a NEW turn is refused immediately rather
|
||||
// than blocked for the length of a model load.
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for {
|
||||
_, rel, aerr := p.acquire()
|
||||
if rel != nil {
|
||||
rel()
|
||||
}
|
||||
if errors.Is(aerr, ErrSwapping) {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("new turns were never refused during a swap (last error: %v)", aerr)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
|
||||
// The swap cannot have completed while our turn was still in flight.
|
||||
select {
|
||||
case e := <-swapped:
|
||||
t.Fatalf("Swap finished before the in-flight turn released: %v", e)
|
||||
case <-time.After(50 * time.Millisecond):
|
||||
}
|
||||
|
||||
release()
|
||||
if e := <-swapped; e != nil {
|
||||
t.Fatalf("Swap after drain: %v", e)
|
||||
}
|
||||
got, err := p.PhraseChat(context.Background(), "привет", nil)
|
||||
if err != nil || got != "new" {
|
||||
t.Fatalf("after swap: %q, %v; want the new model", got, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_RefusedWhenWeDoNotOwnTheServer(t *testing.T) {
|
||||
// NewLLMPhraserAt points at a shared server the eval harness owns. Swapping
|
||||
// there would kill a server another process depends on.
|
||||
p := NewLLMPhraserAt("http://127.0.0.1:1/", DefaultConfig("/m/old.gguf"))
|
||||
if _, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/new.gguf"}); !errors.Is(err, ErrSwapNotOwned) {
|
||||
t.Fatalf("Swap on a borrowed server = %v; want ErrSwapNotOwned", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_SameModelIsANoOp(t *testing.T) {
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
launchesBefore := fl.launch
|
||||
|
||||
res, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/old.gguf"})
|
||||
if err != nil {
|
||||
t.Fatalf("Swap to the live model: %v", err)
|
||||
}
|
||||
if res.Model != "old" {
|
||||
t.Errorf("res.Model = %q; want old", res.Model)
|
||||
}
|
||||
if fl.launch != launchesBefore {
|
||||
t.Errorf("%d extra launches; swapping to the live model must not reload weights", fl.launch-launchesBefore)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSwap_EmptyModelPathRefused(t *testing.T) {
|
||||
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old"}}
|
||||
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
||||
if _, err := p.Swap(context.Background(), SwapSpec{}); err == nil {
|
||||
t.Fatal("Swap with no model path returned nil error")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user