Files
Maven/cmd/mavweb/models.go
T
kami ad074cea31 Swap the resident model without restarting mavend (#250)
Loading a different gguf was a one-line edit to phraser.model_path plus a
restart. It is now an owner-triggered IPC call, off unless configured.

internal/phraser/swap.go holds the safety properties as code:

  - Never two models resident. The old llama-server is killed and reaped
    before the new one is launched. One 1.7B fits the Vega iGPU; a
    blue/green overlap would OOM the box, so it is not offered.
  - Atomic from a turn's point of view. Swap drains the in-flight turns
    (they finish on the old model), then refuses arrivals with ErrSwapping
    until the new server has answered /v1/models. No turn ever sees half a
    swap; refused turns fall back to the classifier cascade.
  - A failed load rolls back. If the new model does not start or does not
    probe, the previous one is reloaded and the call returns RolledBack
    with the error. If the rollback also fails the daemon says so and
    degrades to the classifier rather than pretending to serve.

Holders of the completion client are re-pointed, not rebuilt: llm.Client
guards its base URL and LLMPhraser.OnSwap re-points it, so the router, the
replier, the mail extractor and the memory evaluator follow the new port
without knowing a swap happened.

Reach is deliberately narrow. phraser.swap_models is an exact-match
allowlist of absolute paths a human wrote, rejected at startup otherwise,
so "swap the model" can never mean "load any file on my disk"; the running
model is always swappable back to. MethodSwapModel is AuthStepUp, the same
rung as mutating the tool allowlist, and /models gates POST through the
same stepUpOK the tools page uses. Nothing calls Swap on a timer and no
act, intent or utterance reaches it.

Vikunja #250
2026-08-01 03:59:08 +04:00

146 lines
5.5 KiB
Go

package main
import (
"context"
"errors"
"html/template"
"log"
"net/http"
"strconv"
"strings"
"github.com/kami/maven/internal/ipc"
"github.com/kami/maven/internal/webauthn"
)
// The resident-model surface (Vikunja #250).
//
// GET shows which model llama-server actually has loaded and which files the
// daemon is configured to allow. POST swaps to one of them, behind the same
// step-up gate as POST /tools: the loaded model decides how every utterance is
// routed and how every reply is worded, so it is an owner action.
//
// There is nothing on this page Maven can press. The swap is an IPC method rated
// AuthStepUp in internal/auth, unreachable from an act, an intent or a timer.
// modelController — the two non-CoreAPI methods this page needs. *ipc.Client
// satisfies it; a core without a swap allowlist answers ErrUnknownMethod, which
// the page renders as "not configured" rather than an error.
type modelController interface {
ModelStatus(ctx context.Context) (ipc.ModelStatusResp, error)
SwapModel(ctx context.Context, req ipc.SwapModelReq) (ipc.SwapModelResp, error)
}
var modelsTmpl = template.Must(template.New("models").Funcs(shellFuncs()).Parse(shellTopHTML + modelsHTML + shellBottomHTML))
const modelsHTML = `{{template "shellTop" "models"}}
<h1>Resident model</h1>
<p class=hint>swapping requires step-up — <a href=/auth/passkey>assert a passkey</a> first. The old model is unloaded before the new one is loaded (one model fits the iGPU at a time), so turns during the load are refused and fall back to the classifier.</p>
{{if .Msg}}<div class="msg msg-ok">{{.Msg}}</div>{{end}}
{{if .Err}}<div class="msg msg-err">{{.Err}}</div>{{end}}
{{if .Off}}
<section class=card>
<h2 class=card-title>swap not configured</h2>
<p class=hint>this core has no <code>phraser.swap_models</code> allowlist, so there is nothing to swap to. Add the gguf paths you allow to <code>deploy/mavend.json</code> and restart once.</p>
</section>
{{else}}
<section class=card>
<h2 class=card-title>loaded now</h2>
<div class=scroll><table>
<tr><th>model</th><td><code>{{.Status.Model}}</code></td></tr>
<tr><th>file</th><td><code>{{.Status.ModelPath}}</code></td></tr>
<tr><th>server</th><td><code>{{.Status.BaseURL}}</code></td></tr>
<tr><th>n_ctx</th><td>{{.Status.NCtx}}</td></tr>
<tr><th>n_gpu_layers</th><td>{{.Status.NGpuLayers}}</td></tr>
</table></div>
<p class=hint>the model name is what llama-server reports for itself, not what the config says it should be.</p>
</section>
<section class=card>
<h2 class=card-title>allowed models <span class=badge>{{len .Status.Swappable}}</span></h2>
{{if .Status.Swappable}}<div class=scroll><table><tr><th>file</th><th></th></tr>
{{range .Status.Swappable}}<tr><td><code>{{.}}</code></td>
<td><form method=post action=/models class=inline-form>
<input type=hidden name=model_path value="{{.}}">
<button class=btn>load this one</button></form></td></tr>{{end}}
</table></div>
{{else}}<div class=empty><div>no models allowlisted</div></div>{{end}}
</section>
{{end}}
{{template "shellBottom"}}`
type modelsPage struct {
Msg string
Err string
Off bool
Status ipc.ModelStatusResp
}
// handleModels renders the model surface (GET) and applies a swap (POST).
//
// A failed swap is reported as a failure with the model that is still serving
// named, because that is the state the operator needs: the daemon rolled back
// and is answering turns, it just is not answering them with what he asked for.
func handleModels(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI, session *webauthn.PasskeySession, requireStepUp bool) {
if core == nil {
http.Error(w, "models disabled (no -core)", http.StatusServiceUnavailable)
return
}
mc, ok := core.(modelController)
if !ok {
http.Error(w, "models unavailable: core connection does not support model swap", http.StatusServiceUnavailable)
return
}
ctx := r.Context()
page := modelsPage{}
if r.Method == http.MethodPost {
if !stepUpOK(session, requireStepUp) {
http.Error(w, "step-up required: assert a passkey first", http.StatusForbidden)
return
}
path := strings.TrimSpace(r.FormValue("model_path"))
if path == "" {
http.Error(w, "model_path required", http.StatusBadRequest)
return
}
req := ipc.SwapModelReq{ModelPath: path}
if v, err := strconv.Atoi(r.FormValue("n_ctx")); err == nil {
req.NCtx = v
}
res, err := mc.SwapModel(ctx, req)
switch {
case err == nil:
page.Msg = "loaded " + res.Model + " (" + strconv.FormatInt(res.TookMs, 10) + "ms)"
log.Printf("models: swapped to %s (%s) in %dms", res.ModelPath, res.Model, res.TookMs)
case errors.Is(err, ipc.ErrForbidden):
http.Error(w, "refused: that model is not in phraser.swap_models, or step-up was not asserted", http.StatusForbidden)
return
case errors.Is(err, ipc.ErrUnknownMethod):
http.Error(w, "swap not configured on this core", http.StatusServiceUnavailable)
return
case res.RolledBack:
page.Err = "swap failed, rolled back to " + res.Model + " — she is still answering, with the old model"
log.Printf("models: swap to %s failed, rolled back: %v", path, err)
default:
page.Err = "swap failed: " + err.Error()
log.Printf("models: swap to %s failed: %v", path, err)
}
}
st, err := mc.ModelStatus(ctx)
if err != nil {
if errors.Is(err, ipc.ErrUnknownMethod) {
page.Off = true
} else {
log.Printf("models: status: %v", err)
http.Error(w, "core read failed", http.StatusBadGateway)
return
}
}
page.Status = st
w.Header().Set("Content-Type", "text/html; charset=utf-8")
if err := modelsTmpl.Execute(w, page); err != nil {
log.Printf("models render: %v", err)
}
}