Swap the resident model without restarting mavend (#250)
Loading a different gguf was a one-line edit to phraser.model_path plus a
restart. It is now an owner-triggered IPC call, off unless configured.
internal/phraser/swap.go holds the safety properties as code:
- Never two models resident. The old llama-server is killed and reaped
before the new one is launched. One 1.7B fits the Vega iGPU; a
blue/green overlap would OOM the box, so it is not offered.
- Atomic from a turn's point of view. Swap drains the in-flight turns
(they finish on the old model), then refuses arrivals with ErrSwapping
until the new server has answered /v1/models. No turn ever sees half a
swap; refused turns fall back to the classifier cascade.
- A failed load rolls back. If the new model does not start or does not
probe, the previous one is reloaded and the call returns RolledBack
with the error. If the rollback also fails the daemon says so and
degrades to the classifier rather than pretending to serve.
Holders of the completion client are re-pointed, not rebuilt: llm.Client
guards its base URL and LLMPhraser.OnSwap re-points it, so the router, the
replier, the mail extractor and the memory evaluator follow the new port
without knowing a swap happened.
Reach is deliberately narrow. phraser.swap_models is an exact-match
allowlist of absolute paths a human wrote, rejected at startup otherwise,
so "swap the model" can never mean "load any file on my disk"; the running
model is always swappable back to. MethodSwapModel is AuthStepUp, the same
rung as mutating the tool allowlist, and /models gates POST through the
same stepUpOK the tools page uses. Nothing calls Swap on a timer and no
act, intent or utterance reaches it.
Vikunja #250
This commit is contained in:
+27
-1
@@ -10,10 +10,17 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
type Client struct {
|
||||
// mu guards base only. The base URL changes when the daemon swaps the
|
||||
// resident model (Vikunja #250): llama-server is relaunched on a fresh
|
||||
// port, and every holder of this client — the LLM router, the replier, the
|
||||
// mail extractor — must follow without being rebuilt. One mutexed field is
|
||||
// the whole mechanism; a swap re-points the client, it does not replace it.
|
||||
mu sync.RWMutex
|
||||
base string
|
||||
http *http.Client
|
||||
}
|
||||
@@ -22,6 +29,25 @@ func New(baseURL string, timeout time.Duration) *Client {
|
||||
return &Client{base: baseURL, http: &http.Client{Timeout: timeout}}
|
||||
}
|
||||
|
||||
// SetBaseURL re-points the client at another llama-server. Safe to call while
|
||||
// requests are in flight: a request that already read the old base finishes
|
||||
// against the old base (or fails, and every caller of Complete has a fallback),
|
||||
// and the next one uses the new base. It is deliberately NOT a queue-and-retry —
|
||||
// the phraser quiesces around a swap, so the window is small and a lost turn
|
||||
// degrades to the classifier rather than hanging.
|
||||
func (c *Client) SetBaseURL(base string) {
|
||||
c.mu.Lock()
|
||||
c.base = base
|
||||
c.mu.Unlock()
|
||||
}
|
||||
|
||||
// BaseURL is the server this client currently talks to.
|
||||
func (c *Client) BaseURL() string {
|
||||
c.mu.RLock()
|
||||
defer c.mu.RUnlock()
|
||||
return c.base
|
||||
}
|
||||
|
||||
type Req struct {
|
||||
System string
|
||||
User string
|
||||
@@ -63,7 +89,7 @@ func (c *Client) Complete(ctx context.Context, r Req) (string, error) {
|
||||
RepeatPenalty: r.RepeatPenalty,
|
||||
Stop: r.Stop,
|
||||
})
|
||||
req, err := http.NewRequestWithContext(ctx, "POST", c.base+"/v1/chat/completions", bytes.NewReader(b))
|
||||
req, err := http.NewRequestWithContext(ctx, "POST", c.BaseURL()+"/v1/chat/completions", bytes.NewReader(b))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
@@ -57,3 +57,37 @@ func TestComplete(t *testing.T) {
|
||||
t.Errorf("got %q, want %q", got, "ok")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSetBaseURL — a model swap re-points every holder of the client rather than
|
||||
// rebuilding the router, the replier and the extractors (Vikunja #250).
|
||||
func TestSetBaseURL(t *testing.T) {
|
||||
var hit string
|
||||
srvA := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hit = "A"
|
||||
w.Write([]byte(`{"choices":[{"message":{"content":"a"}}]}`))
|
||||
}))
|
||||
defer srvA.Close()
|
||||
srvB := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hit = "B"
|
||||
w.Write([]byte(`{"choices":[{"message":{"content":"b"}}]}`))
|
||||
}))
|
||||
defer srvB.Close()
|
||||
|
||||
c := New(srvA.URL, 5*time.Second)
|
||||
if _, err := c.Complete(context.Background(), Req{User: "x"}); err != nil {
|
||||
t.Fatalf("Complete against A: %v", err)
|
||||
}
|
||||
if hit != "A" {
|
||||
t.Fatalf("first request went to %q; want A", hit)
|
||||
}
|
||||
c.SetBaseURL(srvB.URL)
|
||||
if got := c.BaseURL(); got != srvB.URL {
|
||||
t.Errorf("BaseURL = %q; want %q", got, srvB.URL)
|
||||
}
|
||||
if _, err := c.Complete(context.Background(), Req{User: "x"}); err != nil {
|
||||
t.Fatalf("Complete against B: %v", err)
|
||||
}
|
||||
if hit != "B" {
|
||||
t.Errorf("request after the swap went to %q; want B", hit)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user