724e90759e
Two fix branches independently added a Gate to internal/llm. One is priority between a voice turn and a background job, the other is admission control while the resident model is swapped. They are orthogonal and both are needed, so the swap one is now SwapGate, with SetSwapGate to install it. Complete takes the priority gate first and the drain second. A background request can wait a long time on priority, and counting it as in flight against the drain that whole time would stall a swap on a request that has not started. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TrVSBKe3RFDF4fGYKWYQnX
128 lines
3.9 KiB
Go
128 lines
3.9 KiB
Go
package phraser
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/kami/maven/internal/llm"
|
|
)
|
|
|
|
// The drain has to cover every holder of the base URL, not only the phrasing
|
|
// paths in this package. The LLM router, the replier, the mail extractor and the
|
|
// memory evaluator all reach llama-server through llm.Client, and a swap that
|
|
// does not count them kills the server mid-turn.
|
|
|
|
// blockingLLM — a completion endpoint that does not answer until the test says
|
|
// so. It stands in for a router call that is generating when the swap arrives.
|
|
func blockingLLM(t *testing.T, release <-chan struct{}) *httptest.Server {
|
|
t.Helper()
|
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
<-release
|
|
w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
|
|
}))
|
|
t.Cleanup(srv.Close)
|
|
return srv
|
|
}
|
|
|
|
func (p *LLMPhraser) inflightCount() int {
|
|
p.mu.Lock()
|
|
defer p.mu.Unlock()
|
|
return p.inflight
|
|
}
|
|
|
|
func TestSwap_WaitsForARouterCallThatWentThroughLLMClient(t *testing.T) {
|
|
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old", "/m/new.gguf": "new"}}
|
|
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
|
|
|
release := make(chan struct{})
|
|
c := llm.New(blockingLLM(t, release).URL, 5*time.Second)
|
|
c.SetSwapGate(p)
|
|
|
|
completed := make(chan error, 1)
|
|
go func() {
|
|
_, err := c.Complete(context.Background(), llm.Req{System: "s", User: "u"})
|
|
completed <- err
|
|
}()
|
|
deadline := time.Now().Add(2 * time.Second)
|
|
for p.inflightCount() == 0 {
|
|
if time.Now().After(deadline) {
|
|
t.Fatal("the llm.Client request never registered with the phraser gate")
|
|
}
|
|
time.Sleep(5 * time.Millisecond)
|
|
}
|
|
|
|
swapped := make(chan error, 1)
|
|
go func() { _, e := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/new.gguf"}); swapped <- e }()
|
|
|
|
select {
|
|
case e := <-swapped:
|
|
t.Fatalf("the swap finished while a router call was still generating (%v); the old server was killed under it", e)
|
|
case <-time.After(200 * time.Millisecond):
|
|
}
|
|
|
|
close(release)
|
|
if e := <-completed; e != nil {
|
|
t.Fatalf("the in-flight call did not finish on the old model: %v", e)
|
|
}
|
|
if e := <-swapped; e != nil {
|
|
t.Fatalf("Swap after the drain: %v", e)
|
|
}
|
|
}
|
|
|
|
func TestSwap_RefusesARouterCallThatArrivesMidSwap(t *testing.T) {
|
|
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old", "/m/new.gguf": "new"}}
|
|
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
|
|
|
// An open server: the refusal has to come from the gate, not from a stall.
|
|
open := make(chan struct{})
|
|
close(open)
|
|
c := llm.New(blockingLLM(t, open).URL, 5*time.Second)
|
|
c.SetSwapGate(p)
|
|
|
|
// Hold the door shut the way quiesce does.
|
|
_, held, err := p.acquire()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
defer held()
|
|
go p.Swap(context.Background(), SwapSpec{ModelPath: "/m/new.gguf"})
|
|
|
|
deadline := time.Now().Add(2 * time.Second)
|
|
for {
|
|
_, err := c.Complete(context.Background(), llm.Req{User: "u"})
|
|
if errors.Is(err, ErrSwapping) {
|
|
return
|
|
}
|
|
if time.Now().After(deadline) {
|
|
t.Fatalf("a router call during a swap was not refused (last error: %v)", err)
|
|
}
|
|
time.Sleep(10 * time.Millisecond)
|
|
}
|
|
}
|
|
|
|
func TestSwap_TotalFailureIsNotReportedAsARollback(t *testing.T) {
|
|
fl := &fakeFleet{models: map[string]string{"/m/old.gguf": "old"}}
|
|
p := newSwapPhraser(t, fl, "/m/old.gguf")
|
|
fl.mu.Lock()
|
|
delete(fl.models, "/m/old.gguf")
|
|
fl.mu.Unlock()
|
|
|
|
res, err := p.Swap(context.Background(), SwapSpec{ModelPath: "/m/broken.gguf"})
|
|
if err == nil {
|
|
t.Fatal("Swap returned nil when both the load and the rollback failed")
|
|
}
|
|
if res.RolledBack {
|
|
t.Error("a total failure set RolledBack; the page then says she is still answering with the old model, and she is not answering at all")
|
|
}
|
|
if !res.NoBackend {
|
|
t.Error("a total failure did not set NoBackend, so nothing distinguishes it from a rolled-back swap")
|
|
}
|
|
if path, _, _ := p.LiveModel(); path != "" {
|
|
t.Errorf("LiveModel = %q after a total failure; nothing is loaded, and naming a gguf makes the page read as half-working", path)
|
|
}
|
|
}
|