fbcca449be
Seven cases. The load-bearing ones are the constraint from 483: an unconfigured deploy never probes and always reaches the floor, a busy card degrades silently with the remote untouched, and a remote that dies between probes still completes the turn and corrects the cached answer on its way out. CompleteRemote is pinned not to fall back, because a named gap that quietly became a 1.7B guess is the failure this whole split exists to prevent. And 1000 Available calls are pinned to make zero probes.
211 lines
6.7 KiB
Go
211 lines
6.7 KiB
Go
package llm
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"sync/atomic"
|
|
"testing"
|
|
"time"
|
|
)
|
|
|
|
// completionServer stands in for a llama-server. It counts what reached it, so
|
|
// a test can say which of the two models answered.
|
|
func completionServer(t *testing.T, reply string, hits *atomic.Int64) *httptest.Server {
|
|
t.Helper()
|
|
s := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
hits.Add(1)
|
|
w.Header().Set("Content-Type", "application/json")
|
|
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"` + reply + `"}}]}`))
|
|
}))
|
|
t.Cleanup(s.Close)
|
|
return s
|
|
}
|
|
|
|
func healthServer(t *testing.T, ok *atomic.Bool) *httptest.Server {
|
|
t.Helper()
|
|
s := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
if !ok.Load() {
|
|
w.WriteHeader(http.StatusServiceUnavailable)
|
|
return
|
|
}
|
|
w.WriteHeader(http.StatusOK)
|
|
}))
|
|
t.Cleanup(s.Close)
|
|
return s
|
|
}
|
|
|
|
// waitFor polls until cond holds or the deadline passes. The prober runs on its
|
|
// own goroutine, so a test has to wait for it rather than assume it has run.
|
|
func waitFor(t *testing.T, cond func() bool) bool {
|
|
t.Helper()
|
|
deadline := time.Now().Add(2 * time.Second)
|
|
for time.Now().Before(deadline) {
|
|
if cond() {
|
|
return true
|
|
}
|
|
time.Sleep(5 * time.Millisecond)
|
|
}
|
|
return false
|
|
}
|
|
|
|
// The unconfigured deploy. No remote, no probing, every call to the floor —
|
|
// exactly what the box does today.
|
|
func TestNoRemoteGoesToTheFloor(t *testing.T) {
|
|
var floorHits atomic.Int64
|
|
floor := completionServer(t, "floor", &floorHits)
|
|
|
|
p := NewPair(nil, New(floor.URL, time.Second), "", time.Second)
|
|
p.Start(context.Background())
|
|
defer p.Stop()
|
|
|
|
if p.Available() {
|
|
t.Fatal("a Pair with no remote reports available")
|
|
}
|
|
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
|
if err != nil {
|
|
t.Fatalf("complete: %v", err)
|
|
}
|
|
if out != "floor" || floorHits.Load() != 1 {
|
|
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
|
}
|
|
}
|
|
|
|
// The workstation is up, so it answers and the resident model is not touched.
|
|
func TestAvailableRemoteAnswers(t *testing.T) {
|
|
var remoteHits, floorHits atomic.Int64
|
|
remote := completionServer(t, "remote", &remoteHits)
|
|
floor := completionServer(t, "floor", &floorHits)
|
|
up := &atomic.Bool{}
|
|
up.Store(true)
|
|
health := healthServer(t, up)
|
|
|
|
p := NewPair(New(remote.URL, time.Second), New(floor.URL, time.Second), health.URL, 20*time.Millisecond)
|
|
p.Start(context.Background())
|
|
defer p.Stop()
|
|
if !waitFor(t, p.Available) {
|
|
t.Fatal("prober never saw the remote come up")
|
|
}
|
|
|
|
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
|
if err != nil {
|
|
t.Fatalf("complete: %v", err)
|
|
}
|
|
if out != "remote" || floorHits.Load() != 0 {
|
|
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
|
}
|
|
}
|
|
|
|
// The card is busy, so /health refuses and Complete degrades silently. This is
|
|
// the constraint from 483: the workstation being down is indistinguishable from
|
|
// today's behaviour.
|
|
func TestBusyCardFallsBackSilently(t *testing.T) {
|
|
var remoteHits, floorHits atomic.Int64
|
|
remote := completionServer(t, "remote", &remoteHits)
|
|
floor := completionServer(t, "floor", &floorHits)
|
|
health := healthServer(t, &atomic.Bool{}) // never ok
|
|
|
|
p := NewPair(New(remote.URL, time.Second), New(floor.URL, time.Second), health.URL, 20*time.Millisecond)
|
|
p.Start(context.Background())
|
|
defer p.Stop()
|
|
time.Sleep(60 * time.Millisecond)
|
|
|
|
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
|
if err != nil {
|
|
t.Fatalf("complete: %v", err)
|
|
}
|
|
if out != "floor" || remoteHits.Load() != 0 {
|
|
t.Fatalf("out = %q, remote hits = %d", out, remoteHits.Load())
|
|
}
|
|
}
|
|
|
|
// The cached admission answer can be one interval out of date, so a remote that
|
|
// dies between probes must still not break the turn.
|
|
func TestRemoteErrorMidRequestFallsBack(t *testing.T) {
|
|
var floorHits atomic.Int64
|
|
dead := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
w.WriteHeader(http.StatusInternalServerError)
|
|
}))
|
|
defer dead.Close()
|
|
floor := completionServer(t, "floor", &floorHits)
|
|
up := &atomic.Bool{}
|
|
up.Store(true)
|
|
health := healthServer(t, up)
|
|
|
|
p := NewPair(New(dead.URL, time.Second), New(floor.URL, time.Second), health.URL, time.Hour)
|
|
p.Start(context.Background())
|
|
defer p.Stop()
|
|
if !waitFor(t, p.Available) {
|
|
t.Fatal("prober never saw the remote come up")
|
|
}
|
|
|
|
out, err := p.Complete(context.Background(), Req{User: "привет"})
|
|
if err != nil {
|
|
t.Fatalf("complete: %v", err)
|
|
}
|
|
if out != "floor" || floorHits.Load() != 1 {
|
|
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
|
}
|
|
// The failed request must have corrected the cached answer, so the next
|
|
// one does not walk into the same hole.
|
|
if p.Available() {
|
|
t.Fatal("a failed remote request left the admission answer up")
|
|
}
|
|
}
|
|
|
|
// The naming half of the degradation rule. A world question must not be handed
|
|
// to the resident model, because it answers by inventing.
|
|
func TestCompleteRemoteNamesTheGap(t *testing.T) {
|
|
var floorHits atomic.Int64
|
|
floor := completionServer(t, "floor", &floorHits)
|
|
health := healthServer(t, &atomic.Bool{}) // never ok
|
|
|
|
p := NewPair(New("http://127.0.0.1:1", time.Second), New(floor.URL, time.Second), health.URL, 20*time.Millisecond)
|
|
p.Start(context.Background())
|
|
defer p.Stop()
|
|
time.Sleep(60 * time.Millisecond)
|
|
|
|
if _, err := p.CompleteRemote(context.Background(), Req{User: "почему небо голубое"}); !errors.Is(err, ErrRemoteUnavailable) {
|
|
t.Fatalf("err = %v, want ErrRemoteUnavailable", err)
|
|
}
|
|
if floorHits.Load() != 0 {
|
|
t.Fatalf("CompleteRemote fell back to the floor %d times", floorHits.Load())
|
|
}
|
|
}
|
|
|
|
// Routing sits on the hot path and must never pay for a health check. Available
|
|
// reads a cached flag, so it costs no network at all.
|
|
func TestAvailableDoesNotProbe(t *testing.T) {
|
|
var probes atomic.Int64
|
|
health := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
probes.Add(1)
|
|
w.WriteHeader(http.StatusOK)
|
|
}))
|
|
defer health.Close()
|
|
|
|
p := NewPair(New("http://127.0.0.1:1", time.Second), New("http://127.0.0.1:1", time.Second), health.URL, time.Hour)
|
|
p.Start(context.Background())
|
|
defer p.Stop()
|
|
if !waitFor(t, p.Available) {
|
|
t.Fatal("prober never ran")
|
|
}
|
|
|
|
before := probes.Load()
|
|
for range 1000 {
|
|
p.Available()
|
|
}
|
|
if got := probes.Load(); got != before {
|
|
t.Fatalf("1000 Available calls made %d probes", got-before)
|
|
}
|
|
}
|
|
|
|
// A Pair with no floor is a configuration mistake, and it must say so rather
|
|
// than silently having nowhere to degrade to.
|
|
func TestNoFloorIsAnError(t *testing.T) {
|
|
p := NewPair(nil, nil, "", time.Second)
|
|
if _, err := p.Complete(context.Background(), Req{User: "привет"}); !errors.Is(err, ErrNoFloor) {
|
|
t.Fatalf("err = %v, want ErrNoFloor", err)
|
|
}
|
|
}
|