Swap the resident model without restarting mavend (#250)
Loading a different gguf was a one-line edit to phraser.model_path plus a
restart. It is now an owner-triggered IPC call, off unless configured.
internal/phraser/swap.go holds the safety properties as code:
- Never two models resident. The old llama-server is killed and reaped
before the new one is launched. One 1.7B fits the Vega iGPU; a
blue/green overlap would OOM the box, so it is not offered.
- Atomic from a turn's point of view. Swap drains the in-flight turns
(they finish on the old model), then refuses arrivals with ErrSwapping
until the new server has answered /v1/models. No turn ever sees half a
swap; refused turns fall back to the classifier cascade.
- A failed load rolls back. If the new model does not start or does not
probe, the previous one is reloaded and the call returns RolledBack
with the error. If the rollback also fails the daemon says so and
degrades to the classifier rather than pretending to serve.
Holders of the completion client are re-pointed, not rebuilt: llm.Client
guards its base URL and LLMPhraser.OnSwap re-points it, so the router, the
replier, the mail extractor and the memory evaluator follow the new port
without knowing a swap happened.
Reach is deliberately narrow. phraser.swap_models is an exact-match
allowlist of absolute paths a human wrote, rejected at startup otherwise,
so "swap the model" can never mean "load any file on my disk"; the running
model is always swappable back to. MethodSwapModel is AuthStepUp, the same
rung as mutating the tool allowlist, and /models gates POST through the
same stepUpOK the tools page uses. Nothing calls Swap on a timer and no
act, intent or utterance reaches it.
Vikunja #250
This commit is contained in:
+1
-2
@@ -27,7 +27,6 @@ import (
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/email"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
@@ -66,7 +65,7 @@ func newMailIntake(st *store.Store, phr phraser.Phraser, cfg *config.Config) *ma
|
||||
if timeout <= 0 {
|
||||
timeout = config.DefaultEmailTimeout
|
||||
}
|
||||
ex := email.NewExtractor(llm.New(lp.BaseURL(), timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
|
||||
ex := email.NewExtractor(llmClientFor(lp, timeout), cfg.Email.MaxTasks, contextBlockFn(cfg, time.Now))
|
||||
log.Printf("mail intake: enabled (max %d candidates per message, timeout %s)", cfg.Email.MaxTasks, timeout)
|
||||
return &mailIntake{st: st, ex: ex, timeout: timeout, now: time.Now}
|
||||
}
|
||||
|
||||
@@ -329,6 +329,7 @@ func run(args []string) error {
|
||||
// ipc.MethodIngestMail reports ErrUnknownMethod.
|
||||
if !locked {
|
||||
wireMailIntake(srv, st, phr, cfg)
|
||||
wireModelSwap(srv, phr, cfg)
|
||||
}
|
||||
|
||||
// WrapKeyFn — wraps the env key with a passkey credential public key and
|
||||
@@ -471,6 +472,7 @@ func run(args []string) error {
|
||||
srv.SetAPI(newAPI)
|
||||
srv.Check = (&auth.Gate{Enrollment: auth.NewFloorEnrollment(), Session: passkeySess}).Check
|
||||
wireMailIntake(srv, st, phr, cfg)
|
||||
wireModelSwap(srv, phr, cfg)
|
||||
|
||||
// Start voice server.
|
||||
if voiceW != nil {
|
||||
|
||||
@@ -16,7 +16,6 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/memeval"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/store"
|
||||
@@ -50,7 +49,7 @@ func newMemoryEvalWorker(st *store.Store, phr phraser.Phraser, cfg *config.Confi
|
||||
}
|
||||
// A generous per-request timeout: this is a long prompt to a Thinking model
|
||||
// and nobody is waiting on the answer.
|
||||
client := llm.New(lp.BaseURL(), 5*time.Minute)
|
||||
client := llmClientFor(lp, 5*time.Minute)
|
||||
ev := memeval.NewEvaluator(st, st, client, memeval.Config{
|
||||
MaxItems: cfg.MemoryEval.MaxItems,
|
||||
MinConfidence: cfg.MemoryEval.MinConfidence,
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
)
|
||||
|
||||
// Swapping the resident model while the daemon runs (Vikunja #250).
|
||||
//
|
||||
// Off unless configured: with no phraser.swap_models allowlist the two IPC
|
||||
// methods are never wired, so they answer ErrUnknownMethod. When it is wired the
|
||||
// swap method is AuthStepUp (internal/auth), which means an authed human surface
|
||||
// only — there is no act, no intent and no timer that reaches it. The daemon
|
||||
// never decides to change its own brain.
|
||||
//
|
||||
// The allowlist is exact-match against paths a human wrote in mavend.json. The
|
||||
// request carries a path and llama-server is started with it as `-m`, so
|
||||
// anything looser would turn "swap the model" into "load any file on my disk".
|
||||
func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
|
||||
if cfg.Phraser == nil || len(cfg.Phraser.SwapModels) == 0 {
|
||||
return
|
||||
}
|
||||
lp, ok := phr.(*phraser.LLMPhraser)
|
||||
if !ok {
|
||||
log.Printf("model swap: phraser.swap_models is set but there is no llama-server phraser — swap disabled")
|
||||
return
|
||||
}
|
||||
allowed := map[string]bool{}
|
||||
for _, m := range cfg.Phraser.SwapModels {
|
||||
allowed[filepath.Clean(m)] = true
|
||||
}
|
||||
// The configured model is always swappable back to, listed or not: the way
|
||||
// out of a bad swap must not depend on remembering to allowlist the model
|
||||
// you are already running.
|
||||
allowed[filepath.Clean(cfg.Phraser.ModelPath)] = true
|
||||
|
||||
srv.SwapModelFn = func(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error) {
|
||||
path := filepath.Clean(req.ModelPath)
|
||||
if !allowed[path] {
|
||||
log.Printf("model swap: REFUSED %q — not in phraser.swap_models", req.ModelPath)
|
||||
return ipc.SwapModelResp{}, fmt.Errorf("%w: %q is not in phraser.swap_models", ipc.ErrForbidden, req.ModelPath)
|
||||
}
|
||||
res, err := lp.Swap(ctx, phraser.SwapSpec{
|
||||
ModelPath: path,
|
||||
NGpuLayers: req.NGpuLayers,
|
||||
NCtx: req.NCtx,
|
||||
})
|
||||
resp := ipc.SwapModelResp{
|
||||
Model: res.Model,
|
||||
ModelPath: res.ModelPath,
|
||||
BaseURL: res.BaseURL,
|
||||
RolledBack: res.RolledBack,
|
||||
TookMs: res.Took.Milliseconds(),
|
||||
}
|
||||
if err != nil {
|
||||
// A rolled-back swap is a failure that left a working daemon behind.
|
||||
// Both halves matter to the caller, so the response is filled in even
|
||||
// though the error is returned.
|
||||
log.Printf("model swap: %v", err)
|
||||
return resp, err
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
srv.ModelStatusFn = func(ctx context.Context) (ipc.ModelStatusResp, error) {
|
||||
path, ngl, nctx := lp.LiveModel()
|
||||
base := lp.BaseURL()
|
||||
resp := ipc.ModelStatusResp{
|
||||
ModelPath: path,
|
||||
BaseURL: base,
|
||||
NGpuLayers: ngl,
|
||||
NCtx: nctx,
|
||||
Swappable: cfg.Phraser.SwapModels,
|
||||
}
|
||||
if base == "" {
|
||||
resp.Model = llm.UnknownModel
|
||||
return resp, nil
|
||||
}
|
||||
id, err := llm.ModelID(ctx, base)
|
||||
if err != nil {
|
||||
// Report the honest "I could not confirm it" rather than echoing the
|
||||
// configured filename as if the server had said it.
|
||||
resp.Model = llm.UnknownModel
|
||||
return resp, nil
|
||||
}
|
||||
resp.Model = id
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
log.Printf("model swap: enabled, %d allowlisted model(s) — step-up required", len(cfg.Phraser.SwapModels))
|
||||
}
|
||||
|
||||
// llmClientFor builds a completion client on the phraser's llama-server and
|
||||
// keeps it pointed at the right one across a model swap.
|
||||
//
|
||||
// Without the OnSwap registration every holder of a base URL — the LLM router,
|
||||
// the replier, the mail extractor, the memory evaluator — would keep talking to
|
||||
// the port of a server that no longer exists, and the daemon would degrade to
|
||||
// the classifier permanently after the first swap. The client is re-pointed, not
|
||||
// rebuilt, so nothing that holds it has to know a swap happened.
|
||||
func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
||||
c := llm.New(lp.BaseURL(), timeout)
|
||||
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
||||
return c
|
||||
}
|
||||
@@ -148,7 +148,9 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem
|
||||
// The replier uses the same llama-server as the phraser.
|
||||
var llmClient *llm.Client
|
||||
if lp, ok := phr.(*phraser.LLMPhraser); ok {
|
||||
llmClient = llm.New(lp.BaseURL(), 60*time.Second)
|
||||
// llmClientFor, not llm.New: this client must follow the phraser onto
|
||||
// the new llama-server when the resident model is swapped (Vikunja #250).
|
||||
llmClient = llmClientFor(lp, 60*time.Second)
|
||||
}
|
||||
// ----- router (the cascade; floor examples seed the classifier) -----
|
||||
// The act matcher's allowlist is exactly the enabled tool names — the
|
||||
|
||||
@@ -124,6 +124,7 @@ var sidebarSections = []struct {
|
||||
Label: "Settings",
|
||||
Pages: []struct{ Label, URL, Key string }{
|
||||
{Label: "Tools", URL: "/tools", Key: "tools"},
|
||||
{Label: "Model", URL: "/models", Key: "models"},
|
||||
{Label: "Passkey", URL: "/auth/passkey", Key: "passkey"},
|
||||
},
|
||||
},
|
||||
@@ -191,6 +192,8 @@ func pageIcon(key string) string {
|
||||
return `<svg class=icon width="14" height="14"><use href="/ethos-icons.svg#i-grid"/></svg>`
|
||||
case "tools":
|
||||
return `<svg class=icon width="14" height="14"><use href="/ethos-icons.svg#i-settings"/></svg>`
|
||||
case "models":
|
||||
return `<svg class=icon width="14" height="14"><use href="/ethos-icons.svg#i-wave"/></svg>`
|
||||
case "passkey":
|
||||
return `<svg class=icon width="14" height="14"><use href="/ethos-icons.svg#i-lock"/></svg>`
|
||||
default:
|
||||
@@ -225,6 +228,8 @@ func pageTitle(key string) string {
|
||||
return "Ecosystem"
|
||||
case "tools":
|
||||
return "Tools"
|
||||
case "models":
|
||||
return "Resident Model"
|
||||
case "passkey":
|
||||
return "Passkey"
|
||||
default:
|
||||
@@ -479,6 +484,12 @@ func main() {
|
||||
mux.HandleFunc("/routines", func(w http.ResponseWriter, r *http.Request) {
|
||||
handleRoutines(w, r, core, stepUpSession, *requireStepUp)
|
||||
})
|
||||
// /models — the resident-model surface (Vikunja #250). Same step-up gate as
|
||||
// /tools, and for a comparable reason: which model is loaded decides how every
|
||||
// utterance is routed and how every reply is worded. GET is read-only.
|
||||
mux.HandleFunc("/models", func(w http.ResponseWriter, r *http.Request) {
|
||||
handleModels(w, r, core, stepUpSession, *requireStepUp)
|
||||
})
|
||||
|
||||
// State-changing routes on this server, and their gate (Vikunja #317):
|
||||
//
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"html/template"
|
||||
"log"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/webauthn"
|
||||
)
|
||||
|
||||
// The resident-model surface (Vikunja #250).
|
||||
//
|
||||
// GET shows which model llama-server actually has loaded and which files the
|
||||
// daemon is configured to allow. POST swaps to one of them, behind the same
|
||||
// step-up gate as POST /tools: the loaded model decides how every utterance is
|
||||
// routed and how every reply is worded, so it is an owner action.
|
||||
//
|
||||
// There is nothing on this page Maven can press. The swap is an IPC method rated
|
||||
// AuthStepUp in internal/auth, unreachable from an act, an intent or a timer.
|
||||
|
||||
// modelController — the two non-CoreAPI methods this page needs. *ipc.Client
|
||||
// satisfies it; a core without a swap allowlist answers ErrUnknownMethod, which
|
||||
// the page renders as "not configured" rather than an error.
|
||||
type modelController interface {
|
||||
ModelStatus(ctx context.Context) (ipc.ModelStatusResp, error)
|
||||
SwapModel(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error)
|
||||
}
|
||||
|
||||
var modelsTmpl = template.Must(template.New("models").Funcs(shellFuncs()).Parse(shellTopHTML + modelsHTML + shellBottomHTML))
|
||||
|
||||
const modelsHTML = `{{template "shellTop" "models"}}
|
||||
<h1>Resident model</h1>
|
||||
<p class=hint>swapping requires step-up — <a href=/auth/passkey>assert a passkey</a> first. The old model is unloaded before the new one is loaded (one model fits the iGPU at a time), so turns during the load are refused and fall back to the classifier.</p>
|
||||
{{if .Msg}}<div class="msg msg-ok">{{.Msg}}</div>{{end}}
|
||||
{{if .Err}}<div class="msg msg-err">{{.Err}}</div>{{end}}
|
||||
{{if .Off}}
|
||||
<section class=card>
|
||||
<h2 class=card-title>swap not configured</h2>
|
||||
<p class=hint>this core has no <code>phraser.swap_models</code> allowlist, so there is nothing to swap to. Add the gguf paths you allow to <code>deploy/mavend.json</code> and restart once.</p>
|
||||
</section>
|
||||
{{else}}
|
||||
<section class=card>
|
||||
<h2 class=card-title>loaded now</h2>
|
||||
<div class=scroll><table>
|
||||
<tr><th>model</th><td><code>{{.Status.Model}}</code></td></tr>
|
||||
<tr><th>file</th><td><code>{{.Status.ModelPath}}</code></td></tr>
|
||||
<tr><th>server</th><td><code>{{.Status.BaseURL}}</code></td></tr>
|
||||
<tr><th>n_ctx</th><td>{{.Status.NCtx}}</td></tr>
|
||||
<tr><th>n_gpu_layers</th><td>{{.Status.NGpuLayers}}</td></tr>
|
||||
</table></div>
|
||||
<p class=hint>the model name is what llama-server reports for itself, not what the config says it should be.</p>
|
||||
</section>
|
||||
<section class=card>
|
||||
<h2 class=card-title>allowed models <span class=badge>{{len .Status.Swappable}}</span></h2>
|
||||
{{if .Status.Swappable}}<div class=scroll><table><tr><th>file</th><th></th></tr>
|
||||
{{range .Status.Swappable}}<tr><td><code>{{.}}</code></td>
|
||||
<td><form method=post action=/models class=inline-form>
|
||||
<input type=hidden name=model_path value="{{.}}">
|
||||
<button class=btn>load this one</button></form></td></tr>{{end}}
|
||||
</table></div>
|
||||
{{else}}<div class=empty><div>no models allowlisted</div></div>{{end}}
|
||||
</section>
|
||||
{{end}}
|
||||
{{template "shellBottom"}}`
|
||||
|
||||
type modelsPage struct {
|
||||
Msg string
|
||||
Err string
|
||||
Off bool
|
||||
Status ipc.ModelStatusResp
|
||||
}
|
||||
|
||||
// handleModels renders the model surface (GET) and applies a swap (POST).
|
||||
//
|
||||
// A failed swap is reported as a failure with the model that is still serving
|
||||
// named, because that is the state the operator needs: the daemon rolled back
|
||||
// and is answering turns, it just is not answering them with what he asked for.
|
||||
func handleModels(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI, session *webauthn.PasskeySession, requireStepUp bool) {
|
||||
if core == nil {
|
||||
http.Error(w, "models disabled (no -core)", http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
mc, ok := core.(modelController)
|
||||
if !ok {
|
||||
http.Error(w, "models unavailable: core connection does not support model swap", http.StatusServiceUnavailable)
|
||||
return
|
||||
}
|
||||
ctx := r.Context()
|
||||
page := modelsPage{}
|
||||
|
||||
if r.Method == http.MethodPost {
|
||||
if !stepUpOK(session, requireStepUp) {
|
||||
http.Error(w, "step-up required: assert a passkey first", http.StatusForbidden)
|
||||
return
|
||||
}
|
||||
path := strings.TrimSpace(r.FormValue("model_path"))
|
||||
if path == "" {
|
||||
http.Error(w, "model_path required", http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
req := ipc.SwapModelReq{ModelPath: path}
|
||||
if v, err := strconv.Atoi(r.FormValue("n_ctx")); err == nil {
|
||||
req.NCtx = v
|
||||
}
|
||||
res, err := mc.SwapModel(ctx, req)
|
||||
switch {
|
||||
case err == nil:
|
||||
page.Msg = "loaded " + res.Model + " (" + strconv.FormatInt(res.TookMs, 10) + "ms)"
|
||||
log.Printf("models: swapped to %s (%s) in %dms", res.ModelPath, res.Model, res.TookMs)
|
||||
case errors.Is(err, ipc.ErrForbidden):
|
||||
http.Error(w, "refused: that model is not in phraser.swap_models, or step-up was not asserted", http.StatusForbidden)
|
||||
return
|
||||
case errors.Is(err, ipc.ErrUnknownMethod):
|
||||
http.Error(w, "swap not configured on this core", http.StatusServiceUnavailable)
|
||||
return
|
||||
case res.RolledBack:
|
||||
page.Err = "swap failed, rolled back to " + res.Model + " — she is still answering, with the old model"
|
||||
log.Printf("models: swap to %s failed, rolled back: %v", path, err)
|
||||
default:
|
||||
page.Err = "swap failed: " + err.Error()
|
||||
log.Printf("models: swap to %s failed: %v", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
st, err := mc.ModelStatus(ctx)
|
||||
if err != nil {
|
||||
if errors.Is(err, ipc.ErrUnknownMethod) {
|
||||
page.Off = true
|
||||
} else {
|
||||
log.Printf("models: status: %v", err)
|
||||
http.Error(w, "core read failed", http.StatusBadGateway)
|
||||
return
|
||||
}
|
||||
}
|
||||
page.Status = st
|
||||
w.Header().Set("Content-Type", "text/html; charset=utf-8")
|
||||
if err := modelsTmpl.Execute(w, page); err != nil {
|
||||
log.Printf("models render: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/webauthn"
|
||||
)
|
||||
|
||||
// fakeModelCore is a core that supports the two model methods. It records what
|
||||
// the page asked for, so the tests can assert the gate rather than the HTML.
|
||||
type fakeModelCore struct {
|
||||
ipc.UnimplementedCoreAPI
|
||||
|
||||
status ipc.ModelStatusResp
|
||||
statusErr error
|
||||
|
||||
swapResp ipc.SwapModelResp
|
||||
swapErr error
|
||||
swapped []ipc.SwapModelReq
|
||||
}
|
||||
|
||||
func (f *fakeModelCore) ModelStatus(ctx context.Context) (ipc.ModelStatusResp, error) {
|
||||
return f.status, f.statusErr
|
||||
}
|
||||
|
||||
func (f *fakeModelCore) SwapModel(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error) {
|
||||
f.swapped = append(f.swapped, req)
|
||||
return f.swapResp, f.swapErr
|
||||
}
|
||||
|
||||
func modelsGET(t *testing.T, core ipc.CoreAPI) *httptest.ResponseRecorder {
|
||||
t.Helper()
|
||||
w := httptest.NewRecorder()
|
||||
handleModels(w, httptest.NewRequest(http.MethodGet, "/models", nil), core, nil, false)
|
||||
return w
|
||||
}
|
||||
|
||||
func modelsPOST(t *testing.T, core ipc.CoreAPI, session *webauthn.PasskeySession, requireStepUp bool, path string) *httptest.ResponseRecorder {
|
||||
t.Helper()
|
||||
r := httptest.NewRequest(http.MethodPost, "/models", strings.NewReader("model_path="+path))
|
||||
r.Header.Set("Content-Type", "application/x-www-form-urlencoded")
|
||||
w := httptest.NewRecorder()
|
||||
handleModels(w, r, core, session, requireStepUp)
|
||||
return w
|
||||
}
|
||||
|
||||
func TestModels_GETShowsTheLoadedModelAndTheAllowlist(t *testing.T) {
|
||||
core := &fakeModelCore{status: ipc.ModelStatusResp{
|
||||
Model: "Qwen3-1.7B-UD-Q4_K_XL",
|
||||
ModelPath: "/opt/maven/models/llm/qwen3.gguf",
|
||||
BaseURL: "http://127.0.0.1:18099",
|
||||
NCtx: 4096,
|
||||
Swappable: []string{"/opt/maven/models/llm/qwen3.gguf", "/opt/maven/models/llm/qwen3-cpt.gguf"},
|
||||
}}
|
||||
w := modelsGET(t, core)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("GET /models = %d; want 200", w.Code)
|
||||
}
|
||||
body := w.Body.String()
|
||||
for _, want := range []string{"Qwen3-1.7B-UD-Q4_K_XL", "qwen3-cpt.gguf", "4096"} {
|
||||
if !strings.Contains(body, want) {
|
||||
t.Errorf("page does not mention %q", want)
|
||||
}
|
||||
}
|
||||
if len(core.swapped) != 0 {
|
||||
t.Errorf("a GET swapped the model: %v", core.swapped)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModels_POSTRequiresStepUpWhenFailingClosed(t *testing.T) {
|
||||
// No WebAuthn configured (nil session) + -require-stepup ⇒ deny, exactly
|
||||
// like POST /tools. Nothing reaches core.
|
||||
core := &fakeModelCore{}
|
||||
w := modelsPOST(t, core, nil, true, "/opt/maven/models/llm/qwen3.gguf")
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("POST /models without assertable step-up = %d; want 403", w.Code)
|
||||
}
|
||||
if len(core.swapped) != 0 {
|
||||
t.Fatalf("a denied POST still called SwapModel: %v", core.swapped)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModels_POSTSwapsAndReportsTheModelThatAnswered(t *testing.T) {
|
||||
core := &fakeModelCore{
|
||||
swapResp: ipc.SwapModelResp{Model: "qwen3-cpt", ModelPath: "/m/cpt.gguf", TookMs: 4200},
|
||||
status: ipc.ModelStatusResp{Model: "qwen3-cpt", ModelPath: "/m/cpt.gguf"},
|
||||
}
|
||||
w := modelsPOST(t, core, nil, false, "/m/cpt.gguf")
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("POST /models = %d; want 200", w.Code)
|
||||
}
|
||||
if len(core.swapped) != 1 || core.swapped[0].ModelPath != "/m/cpt.gguf" {
|
||||
t.Fatalf("SwapModel calls = %v; want one for /m/cpt.gguf", core.swapped)
|
||||
}
|
||||
if !strings.Contains(w.Body.String(), "loaded qwen3-cpt") {
|
||||
t.Errorf("page does not report which model was loaded:\n%s", w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestModels_RolledBackSwapSaysSheIsStillAnswering(t *testing.T) {
|
||||
core := &fakeModelCore{
|
||||
swapResp: ipc.SwapModelResp{Model: "qwen3", ModelPath: "/m/old.gguf", RolledBack: true},
|
||||
swapErr: errBrokenModel{},
|
||||
status: ipc.ModelStatusResp{Model: "qwen3", ModelPath: "/m/old.gguf"},
|
||||
}
|
||||
w := modelsPOST(t, core, nil, false, "/m/cpt.gguf")
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("POST /models after a rollback = %d; want 200 with the failure rendered", w.Code)
|
||||
}
|
||||
body := w.Body.String()
|
||||
if !strings.Contains(body, "rolled back to qwen3") {
|
||||
t.Errorf("page does not say it rolled back:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModels_RefusedPathIs403(t *testing.T) {
|
||||
core := &fakeModelCore{swapErr: ipc.ErrForbidden}
|
||||
w := modelsPOST(t, core, nil, false, "/etc/passwd")
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("POST /models with a non-allowlisted path = %d; want 403", w.Code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestModels_UnconfiguredCoreRendersOff(t *testing.T) {
|
||||
core := &fakeModelCore{statusErr: ipc.ErrUnknownMethod}
|
||||
w := modelsGET(t, core)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("GET /models against a core without the swap = %d; want 200", w.Code)
|
||||
}
|
||||
if !strings.Contains(w.Body.String(), "swap not configured") {
|
||||
t.Errorf("page does not say the capability is off:\n%s", w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestModels_CoreWithoutTheMethodsIs503(t *testing.T) {
|
||||
// An in-process CoreAPI (no swap methods) must not 500 the page.
|
||||
w := modelsGET(t, ipc.UnimplementedCoreAPI{})
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("GET /models on a core without the methods = %d; want 503", w.Code)
|
||||
}
|
||||
}
|
||||
|
||||
type errBrokenModel struct{}
|
||||
|
||||
func (errBrokenModel) Error() string { return "llm: server did not start" }
|
||||
Reference in New Issue
Block a user