diff --git a/cmd/mavend/modelseam_test.go b/cmd/mavend/modelseam_test.go new file mode 100644 index 0000000..553b8ac --- /dev/null +++ b/cmd/mavend/modelseam_test.go @@ -0,0 +1,91 @@ +package main + +import ( + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/kami/maven/internal/config" + "github.com/kami/maven/internal/llm" +) + +// No `workstation` block is the shipping deploy. The seam must then be the +// resident client itself, with nothing probing anything. +func TestModelSeamUnconfiguredIsResidentOnly(t *testing.T) { + resident := llm.New("http://127.0.0.1:1", time.Second) + hot, pair := modelSeam(&config.Config{}, resident) + if pair != nil { + t.Error("built a pair with no workstation configured") + } + if hot == nil { + t.Fatal("no seam at all, so the cascade would route with the classifier") + } +} + +// A workstation with no resident model behind it has no floor, and a Pair with +// no floor is a configuration mistake rather than a degraded mode. +func TestModelSeamWithoutResidentIsNil(t *testing.T) { + cfg := &config.Config{Workstation: &config.WorkstationConfig{URL: "http://127.0.0.1:1"}} + cfg.Workstation.Health = strings.TrimRight(cfg.Workstation.URL, "/") + "/health" + hot, pair := modelSeam(cfg, nil) + if hot != nil || pair != nil { + t.Errorf("built a seam with no floor: hot=%v pair=%v", hot, pair) + } +} + +// The configured case: the seam is the pair, and the pair notices a workstation +// that answers /health. +func TestModelSeamPrefersAnAnsweringWorkstation(t *testing.T) { + up := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusOK) + })) + defer up.Close() + + cfg := &config.Config{Workstation: &config.WorkstationConfig{ + URL: up.URL, + Probe: config.Duration(10 * time.Millisecond), + }} + cfg.Workstation.Health = strings.TrimRight(cfg.Workstation.URL, "/") + "/health" + + hot, pair := modelSeam(cfg, llm.New("http://127.0.0.1:1", time.Second)) + if pair == nil || hot == nil { + t.Fatal("no pair built for a configured workstation") + } + defer pair.Stop() + + deadline := time.Now().Add(2 * time.Second) + for !pair.Available() && time.Now().Before(deadline) { + time.Sleep(5 * time.Millisecond) + } + if !pair.Available() { + t.Fatal("the pair never saw a workstation that answers /health") + } +} + +// A card held by a CPT run answers 503, and that must read as unavailable +// rather than as an error a turn has to handle. +func TestModelSeamHeldCardIsUnavailable(t *testing.T) { + busy := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + http.Error(w, "model not loaded", http.StatusServiceUnavailable) + })) + defer busy.Close() + + cfg := &config.Config{Workstation: &config.WorkstationConfig{ + URL: busy.URL, + Probe: config.Duration(10 * time.Millisecond), + }} + cfg.Workstation.Health = strings.TrimRight(cfg.Workstation.URL, "/") + "/health" + + _, pair := modelSeam(cfg, llm.New("http://127.0.0.1:1", time.Second)) + if pair == nil { + t.Fatal("no pair built for a configured workstation") + } + defer pair.Stop() + + time.Sleep(50 * time.Millisecond) + if pair.Available() { + t.Error("a 503 from the supervisor read as available") + } +} diff --git a/cmd/mavend/voicewire.go b/cmd/mavend/voicewire.go index f382829..41a6911 100644 --- a/cmd/mavend/voicewire.go +++ b/cmd/mavend/voicewire.go @@ -48,7 +48,11 @@ type voiceWiring struct { // mcp — the MCP client, nil unless the `mcp` block configures an enabled // server (Vikunja #251). Its tools land in the same allowlist as every // other act, so nothing else here has to know about it. - mcp *mcpWiring + // pair — the workstation model with the resident one as the floor, nil + // unless a `workstation` block names an address. Held here only so the + // prober is stopped on shutdown; callers were handed it at build time. + pair *llm.Pair + mcp *mcpWiring // home — the Home Assistant client, nil unless the `smarthome` block is // enabled (Vikunja #256). Its devices land in the same allowlist as every // other act, so nothing else here has to know about it. @@ -76,6 +80,9 @@ func (w *voiceWiring) close() { if w.ttsClient != nil { _ = w.ttsClient.Close() } + if w.pair != nil { + w.pair.Stop() + } w.mcp.close() } @@ -188,6 +195,11 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem // the new llama-server when the resident model is swapped (Vikunja #250). llmClient = llmClientFor(lp, 60*time.Second) } + // The workstation model sits above that one when it is configured and its + // card is free. hot is what the router and the replier complete through: + // either the pair, or the resident client alone, or nothing at all. + hot, pair := modelSeam(cfg, llmClient) + w.pair = pair // ----- router (the cascade; floor examples seed the classifier) ----- // The act matcher's allowlist is exactly the enabled tool names — the // router only matches acts the executor can run (one source of truth). @@ -199,7 +211,7 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem // against the classifier's 50.0%, at about 1s a turn instead of 30ms (see // config.VoiceConfig.LLMRouter). The classifier always stays wired as the // fallback, so a model error never breaks a turn. - rtr := buildRouter(emb, matcher, threshold, pickLLMRouter(cfg.Voice.UseLLMRouter(), llmClient)) + rtr := buildRouter(emb, matcher, threshold, pickLLMRouter(cfg.Voice.UseLLMRouter(), hot)) // ----- sessions registry (shared with voicesink) ----- sessions := voice.NewSessions() @@ -233,8 +245,8 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem // ----- replier (LLM-backed when the engine is on, Stub floor otherwise) ----- replier := voice.Replier(voice.NewStubReplier()) - if llmClient != nil { - replier = newLLMReplier(llmClient, contextBlockFn(cfg, time.Now)) + if hot != nil { + replier = newLLMReplier(hot, contextBlockFn(cfg, time.Now)) } // ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) ----- @@ -291,7 +303,42 @@ func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, mem // pickLLMRouter returns the LLM router when the operator asked for it and there // is a llama-server to talk to, and nil otherwise. nil is safe: the cascade then // routes with the classifier, so an unusable setting costs accuracy, not turns. -func pickLLMRouter(enabled bool, c *llm.Client) *router.LLMRouter { +// modelSeam builds the completion seam the hot paths use: routing and replies. +// +// With no `workstation` block it is the resident client and nothing probes +// anything, which is today's deploy exactly. With one, it is an llm.Pair that +// prefers the workstation and falls back to the resident model silently — the +// silent half of the degradation rule (docs/offload.md), because the big model +// is only better here and the 1.7B is today's shipping quality. He is never +// told which of the two phrased his reply. +// +// A nil resident client means the phraser is not an LLM phraser. There is then +// no floor, and a Pair with no floor is a configuration mistake rather than a +// degraded mode, so the seam is nil and the cascade routes with the classifier. +func modelSeam(cfg *config.Config, resident *llm.Client) (router.Completer, *llm.Pair) { + if resident == nil { + if cfg.Workstation != nil { + log.Printf("voice: a workstation is configured but there is no resident model to floor it with — ignoring the block") + } + return nil, nil + } + if cfg.Workstation == nil { + return resident, nil + } + ws := cfg.Workstation + pair := llm.NewPair( + llm.New(ws.URL, time.Duration(ws.Timeout)), + resident, + ws.Health, + time.Duration(ws.Probe), + ) + pair.Start(context.Background()) + log.Printf("voice: workstation model at %s, probed every %s, resident model as the floor", + ws.URL, time.Duration(ws.Probe)) + return pair, pair +} + +func pickLLMRouter(enabled bool, c router.Completer) *router.LLMRouter { if !enabled { return nil }