Swap the resident model without restarting mavend (#250)
Loading a different gguf was a one-line edit to phraser.model_path plus a
restart. It is now an owner-triggered IPC call, off unless configured.
internal/phraser/swap.go holds the safety properties as code:
- Never two models resident. The old llama-server is killed and reaped
before the new one is launched. One 1.7B fits the Vega iGPU; a
blue/green overlap would OOM the box, so it is not offered.
- Atomic from a turn's point of view. Swap drains the in-flight turns
(they finish on the old model), then refuses arrivals with ErrSwapping
until the new server has answered /v1/models. No turn ever sees half a
swap; refused turns fall back to the classifier cascade.
- A failed load rolls back. If the new model does not start or does not
probe, the previous one is reloaded and the call returns RolledBack
with the error. If the rollback also fails the daemon says so and
degrades to the classifier rather than pretending to serve.
Holders of the completion client are re-pointed, not rebuilt: llm.Client
guards its base URL and LLMPhraser.OnSwap re-points it, so the router, the
replier, the mail extractor and the memory evaluator follow the new port
without knowing a swap happened.
Reach is deliberately narrow. phraser.swap_models is an exact-match
allowlist of absolute paths a human wrote, rejected at startup otherwise,
so "swap the model" can never mean "load any file on my disk"; the running
model is always swappable back to. MethodSwapModel is AuthStepUp, the same
rung as mutating the tool allowlist, and /models gates POST through the
same stepUpOK the tools page uses. Nothing calls Swap on a timer and no
act, intent or utterance reaches it.
Vikunja #250
This commit is contained in:
@@ -176,6 +176,50 @@ type IngestMailResp struct {
|
||||
Skipped bool `json:"skipped,omitempty"`
|
||||
}
|
||||
|
||||
// SwapModelReq — load another resident model without restarting the daemon
|
||||
// (Vikunja #250). ModelPath must be one of the paths in phraser.swap_models;
|
||||
// anything else is ErrForbidden, and an unconfigured allowlist makes the whole
|
||||
// method ErrUnknownMethod.
|
||||
//
|
||||
// NGpuLayers and NCtx are zero for "keep what is loaded now", which is the
|
||||
// normal case — the same laptop iGPU, a different gguf.
|
||||
//
|
||||
// This is an owner action. It is AuthStepUp in the authority table, it is not on
|
||||
// CoreAPI, and no act, intent or timer can reach it: swapping the model is not
|
||||
// something Maven does to herself.
|
||||
type SwapModelReq struct {
|
||||
ModelPath string `json:"model_path"`
|
||||
NGpuLayers int `json:"n_gpu_layers,omitempty"`
|
||||
NCtx int `json:"n_ctx,omitempty"`
|
||||
}
|
||||
|
||||
// SwapModelResp — what the daemon ended up serving. Model is the identity the
|
||||
// new llama-server reported for itself, not an echo of the request: if the file
|
||||
// was not the model the operator thought it was, this is where it shows.
|
||||
//
|
||||
// RolledBack is true when the requested model failed to load or would not answer
|
||||
// and the previous one was put back. In that case the call also returns an error
|
||||
// — the swap did not happen — and Model names the model still serving.
|
||||
type SwapModelResp struct {
|
||||
Model string `json:"model"`
|
||||
ModelPath string `json:"model_path"`
|
||||
BaseURL string `json:"base_url"`
|
||||
RolledBack bool `json:"rolled_back,omitempty"`
|
||||
TookMs int64 `json:"took_ms"`
|
||||
}
|
||||
|
||||
// ModelStatusResp — which model is resident and which ones may be swapped in.
|
||||
// Read-only; the authed page renders it. Swappable is the configured allowlist,
|
||||
// so an empty list means the capability is off.
|
||||
type ModelStatusResp struct {
|
||||
Model string `json:"model"`
|
||||
ModelPath string `json:"model_path"`
|
||||
BaseURL string `json:"base_url"`
|
||||
NGpuLayers int `json:"n_gpu_layers"`
|
||||
NCtx int `json:"n_ctx"`
|
||||
Swappable []string `json:"swappable,omitempty"`
|
||||
}
|
||||
|
||||
type listTasksReq struct {
|
||||
Status string `json:"status"` // "" all | "live" | candidate|open|done|dropped
|
||||
}
|
||||
|
||||
@@ -459,6 +459,28 @@ func (c *Client) IngestMail(ctx context.Context, req IngestMailReq) (IngestMailR
|
||||
return r, nil
|
||||
}
|
||||
|
||||
// SwapModel asks core to load another resident model (Vikunja #250).
|
||||
// ErrUnknownMethod means core has no phraser.swap_models allowlist configured;
|
||||
// ErrForbidden means the path is not on it, or step-up was not asserted. A
|
||||
// non-nil error with RolledBack set means nothing changed — the old model is
|
||||
// still serving.
|
||||
func (c *Client) SwapModel(ctx context.Context, req SwapModelReq) (SwapModelResp, error) {
|
||||
var r SwapModelResp
|
||||
if err := c.call(ctx, MethodSwapModel, req, &r); err != nil {
|
||||
return SwapModelResp{}, err
|
||||
}
|
||||
return r, nil
|
||||
}
|
||||
|
||||
// ModelStatus reports the resident model and the swap allowlist. Read-only.
|
||||
func (c *Client) ModelStatus(ctx context.Context) (ModelStatusResp, error) {
|
||||
var r ModelStatusResp
|
||||
if err := c.call(ctx, MethodModelStatus, nil, &r); err != nil {
|
||||
return ModelStatusResp{}, err
|
||||
}
|
||||
return r, nil
|
||||
}
|
||||
|
||||
func (c *Client) DismissProposedRoutine(ctx context.Context, id int64) error {
|
||||
return c.call(ctx, MethodDismissProposedRoutine, dismissProposedRoutineReq{ID: id}, nil)
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"encoding/binary"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"os"
|
||||
@@ -598,3 +599,44 @@ func TestIngestMail_Hook(t *testing.T) {
|
||||
t.Errorf("req across the wire = %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSwapModel_OffUnlessConfigured — no allowlist in the config means the
|
||||
// daemon never sets the hook, and the method does not exist. That is what "off
|
||||
// unless configured" looks like at the wire for the model swap (Vikunja #250).
|
||||
func TestSwapModel_OffUnlessConfigured(t *testing.T) {
|
||||
_, _, cli, _ := newServerWithStore(t)
|
||||
if _, err := cli.SwapModel(context.Background(), SwapModelReq{ModelPath: "/m/x.gguf"}); !errors.Is(err, ErrUnknownMethod) {
|
||||
t.Fatalf("SwapModel error = %v, want ErrUnknownMethod", err)
|
||||
}
|
||||
if _, err := cli.ModelStatus(context.Background()); !errors.Is(err, ErrUnknownMethod) {
|
||||
t.Fatalf("ModelStatus error = %v, want ErrUnknownMethod", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSwapModel_Hook — the request crosses the boundary intact and the reported
|
||||
// identity comes back. A refusal from the daemon's allowlist arrives as
|
||||
// ErrForbidden, which is what a caller keys its error message off.
|
||||
func TestSwapModel_Hook(t *testing.T) {
|
||||
_, srv, cli, _ := newServerWithStore(t)
|
||||
var got SwapModelReq
|
||||
srv.SwapModelFn = func(_ context.Context, req SwapModelReq) (SwapModelResp, error) {
|
||||
got = req
|
||||
if req.ModelPath != "/m/allowed.gguf" {
|
||||
return SwapModelResp{}, fmt.Errorf("%w: not allowlisted", ErrForbidden)
|
||||
}
|
||||
return SwapModelResp{Model: "allowed", ModelPath: req.ModelPath, BaseURL: "http://127.0.0.1:9", TookMs: 12}, nil
|
||||
}
|
||||
resp, err := cli.SwapModel(context.Background(), SwapModelReq{ModelPath: "/m/allowed.gguf", NCtx: 4096})
|
||||
if err != nil {
|
||||
t.Fatalf("SwapModel: %v", err)
|
||||
}
|
||||
if resp.Model != "allowed" || resp.TookMs != 12 {
|
||||
t.Errorf("resp = %+v", resp)
|
||||
}
|
||||
if got.NCtx != 4096 {
|
||||
t.Errorf("req across the wire = %+v", got)
|
||||
}
|
||||
if _, err := cli.SwapModel(context.Background(), SwapModelReq{ModelPath: "/etc/shadow"}); !errors.Is(err, ErrForbidden) {
|
||||
t.Fatalf("swap to a non-allowlisted path = %v; want ErrForbidden", err)
|
||||
}
|
||||
}
|
||||
|
||||
+46
-2
@@ -432,6 +432,19 @@ type Server struct {
|
||||
// every CoreAPI implementation has to carry.
|
||||
IngestMailFn IngestMailFunc
|
||||
|
||||
// SwapModelFn / ModelStatusFn — the on-the-fly resident model swap (Vikunja
|
||||
// #250) and its read side. Set by the daemon only when phraser.swap_models
|
||||
// lists at least one model AND the phraser owns a llama-server; nil ⇒ both
|
||||
// methods answer ErrUnknownMethod, which is what "off unless configured"
|
||||
// looks like at the wire.
|
||||
//
|
||||
// They bypass CoreAPI for the same reason IngestMailFn does: this is not a
|
||||
// store operation, it needs the daemon's llama-server, and no other CoreAPI
|
||||
// implementation should have to carry it. MethodSwapModel is AuthStepUp in
|
||||
// internal/auth — owner-triggered, never an act and never a timer.
|
||||
SwapModelFn SwapModelFunc
|
||||
ModelStatusFn ModelStatusFunc
|
||||
|
||||
// UnlockFn — unwraps the store encryption key from the wrapped blob using
|
||||
// the passkey credential public key, opens the encrypted store, and wires
|
||||
// the rest of the daemon (voice, loop, delivery). Set by the daemon when
|
||||
@@ -450,6 +463,12 @@ type WrapKeyFunc func(ctx context.Context, publicKey []byte) error
|
||||
// public key and completes daemon initialization.
|
||||
type UnlockFunc func(ctx context.Context, publicKey []byte) error
|
||||
|
||||
// SwapModelFunc — loads another resident model in place of the live one.
|
||||
type SwapModelFunc func(ctx context.Context, req SwapModelReq) (SwapModelResp, error)
|
||||
|
||||
// ModelStatusFunc — reports the resident model and the swap allowlist.
|
||||
type ModelStatusFunc func(ctx context.Context) (ModelStatusResp, error)
|
||||
|
||||
// IngestMailFunc — core-side mail extraction. Returns what was captured.
|
||||
type IngestMailFunc func(ctx context.Context, req IngestMailReq) (IngestMailResp, error)
|
||||
|
||||
@@ -620,8 +639,9 @@ func withoutParams[R any](fn func(ctx context.Context, api CoreAPI) (R, error))
|
||||
// existed) as an argument — so SetAPI's runtime swap (the unlock transition)
|
||||
// is still honored on the very next request with no extra plumbing here.
|
||||
//
|
||||
// MethodAssertStepUp, MethodStoreEncryptionKey, MethodUnlock and
|
||||
// MethodIngestMail are NOT in this table: they bypass CoreAPI entirely
|
||||
// MethodAssertStepUp, MethodStoreEncryptionKey, MethodUnlock,
|
||||
// MethodIngestMail, MethodSwapModel and MethodModelStatus are NOT in this
|
||||
// table: they bypass CoreAPI entirely
|
||||
// (s.StepUp / s.WrapKeyFn / s.UnlockFn / s.IngestMailFn), so dispatch
|
||||
// special-cases them before consulting the table.
|
||||
var methodTable = map[Method]handlerFunc{
|
||||
@@ -874,6 +894,30 @@ func (s *Server) dispatch(ctx context.Context, req Request) (json.RawMessage, er
|
||||
return marshalResult(resp), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
|
||||
case MethodSwapModel:
|
||||
if s.SwapModelFn != nil {
|
||||
var p SwapModelReq
|
||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := s.SwapModelFn(ctx, p)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return marshalResult(resp), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
|
||||
case MethodModelStatus:
|
||||
if s.ModelStatusFn != nil {
|
||||
resp, err := s.ModelStatusFn(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return marshalResult(resp), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
}
|
||||
|
||||
h, ok := methodTable[req.Method]
|
||||
|
||||
@@ -51,6 +51,8 @@ const (
|
||||
MethodListTasks Method = "list_tasks"
|
||||
MethodSetTaskStatus Method = "set_task_status"
|
||||
MethodIngestMail Method = "ingest_mail"
|
||||
MethodSwapModel Method = "swap_model"
|
||||
MethodModelStatus Method = "model_status"
|
||||
)
|
||||
|
||||
// Request — one frame from module to core. Params is the JSON-encoded argument
|
||||
|
||||
Reference in New Issue
Block a user