diff --git a/CLAUDE.md b/CLAUDE.md index 558f6e2..6fed4ab 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -47,6 +47,31 @@ free — `worldGap` in `cmd/mavend/worldmodel.go`, which the owner hears instead answer. A box with no `workstation` block behaves exactly as it did before the seam: naming a gap requires a gap. The offload table in `docs/offload.md` says which caller is which. +**Speech-to-text moved on 2026-08-09** (V-486). `sttSeam` in `cmd/mavend/voicewire.go` +builds an `stt.Pair` beside `modelSeam`, preferring CrisperWhisper 2.0 turbo on workpc +with mavsttd as the floor. It takes only the silent half of the rule. A worse +transcript is still a turn, so `stt.Pair` has no `TranscribeRemote`. The fallback is +never spoken. CW2 turbo scores **10.4% WER in Russian against 27.5%** for the `ggml-small.bin` +mavsttd loads, over 200 Golos clips +(`docs/evals/2026-08-09-crisperwhisper2-russian-wer.md`). It runs in Intended mode, not +Verbatim, though that corpus cannot separate the two. +**whisper.cpp cannot load CW2 at all.** It reads its language count off the vocabulary +size, and CW2's 51897 tokens shift seven special token ids. So it is not a second +endpoint on mavgpud. It is its own transformers service on port 8081 +(`deploy/cw2/serve.py`), which Maven reaches directly. `stt.HTTPTranscriber` +posts raw PCM to it with a bearer token, because audio is the most sensitive thing that +crosses this seam. The switch is `workstation.stt` in +`deploy/mavend.json`, and deleting the block sends every utterance to mavsttd. +**mavgpud runs that service as a second child.** That is not an optimisation. CW2 is a +ROCm process on the same card, so it registers on the KFD like any contender. Under its own +systemd unit it made mavgpud evict llama-server every few seconds. That took the +gemma-4-12b arm down for eight minutes on 2026-08-09 before anyone noticed. The card needs +one owner. Any GPU service added beside this daemon has the same defect, so add it to +`cmd/mavgpud` and not to systemd. CW2 is on the yield clock and not the idle one. At 1.6GB +it denies the card to nobody, and unloading it would only send the next voice turn to the +homesrv floor. +Text-to-speech has not moved and piper on homesrv is still the only synthesizer. + ## Build & test CGO daemons (`mavend`, `mavsttd`, `mavttsd`, `mavenclient`) need the vendored toolchain @@ -230,11 +255,21 @@ re-run it, start a **second** llama-server on a fixed host port — the resident `--port 0` inside the container and no host process can reach it. **The numbers above are the homesrv floor, not the ceiling.** With the workstation up, routing -completes through `llm.Pair` against gemma-4-12b and scores **84.4% full / 93.5% intent-only at -p50 329ms** — better than the resident model and about 2.5× faster (`docs/evals/2026-08-02-workstation-gemma4-12b.md`, -Vikunja #485). The workstation is never assumed up, so both sets of numbers are live. Judge a +completes through `llm.Pair` against the model mavgpud holds, which is better than the resident +model and about 2.5× faster. gemma-4-12b scored **84.4% full / 93.5% intent-only at p50 329ms** +(`docs/evals/2026-08-02-workstation-gemma4-12b.md`, Vikunja #485). The workstation is never +assumed up, so both sets of numbers are live. Judge a routing change against the classifier and the resident model, since those are what always answer. +**The workstation runs gemma-4-E4B since 2026-08-09** (owner's call), and it is a +step down measured the same day (`docs/evals/2026-08-09-e4b-vs-12b-routing.md`). +Against a same-session 12B control it scores **83.3% full / 89.6% intent-only, +destination 19/33 against 23/33, at p50 294ms against 344ms**. So it costs four +destination cases and buys 50ms. Read destination as the finding: it names nothing +where the 12B names `recall` or `calendar`, which is safe but walks the whole chain. +It also has no MTP and cannot be given any here. The only `gemma4-assistant` +draft on disk is trained against the 12B's hidden states. + **The intended third engine is not a generative model** (owner's call, 05-08-2026, V-546, `docs/plans/18-routing-heads-on-e5-small.md`). Routing has a bounded output space, so it is classification, and the 118M multilingual-e5-small is already resident. Three heads on one diff --git a/cmd/mavgpud/gpu.go b/cmd/mavgpud/gpu.go index 89120c6..989c2b9 100644 --- a/cmd/mavgpud/gpu.go +++ b/cmd/mavgpud/gpu.go @@ -34,22 +34,31 @@ type probe struct { drmDev string } -// foreign lists every ROCm process that is not ours. selfPID is the supervisor's -// llama-server child, or 0 when it is not running. +// foreign lists every ROCm process that is not ours. self holds the pids of the +// supervisor's own children, and a child that is not running contributes 0. +// +// There is more than one child since 09-08-2026. CW2 registers on the KFD like +// any ROCm job, so a supervisor that excluded only llama-server would read its +// own transcriber as a contender, yield the card to it, and never keep a model +// loaded again. // // An unreadable kfd tree returns no processes and no error. That is deliberate // and it is the safe direction only because startVRAM also has to agree before // anything launches: a supervisor that cannot see the KFD never sees free VRAM // either, because the CPT run holding the card shows up in the drm totals. -func (p probe) foreign(selfPID int) []gpuProc { +func (p probe) foreign(self ...int) []gpuProc { entries, err := os.ReadDir(p.kfdRoot) if err != nil { return nil } + mine := make(map[int]bool, len(self)) + for _, pid := range self { + mine[pid] = true + } var out []gpuProc for _, e := range entries { pid, err := strconv.Atoi(e.Name()) - if err != nil || pid == selfPID { + if err != nil || mine[pid] { continue } out = append(out, gpuProc{ diff --git a/cmd/mavgpud/gpu_test.go b/cmd/mavgpud/gpu_test.go index 073874b..52dc07d 100644 --- a/cmd/mavgpud/gpu_test.go +++ b/cmd/mavgpud/gpu_test.go @@ -1,6 +1,7 @@ package main import ( + "context" "net/http" "net/http/httptest" "net/url" @@ -8,6 +9,7 @@ import ( "path/filepath" "strconv" "testing" + "time" ) // fakeKFD builds the sysfs shape the workstation actually has: one directory @@ -47,6 +49,24 @@ func TestForeignExcludesOurChild(t *testing.T) { } } +// The transcriber is a ROCm process on the same card, so it registers on the +// KFD exactly like a contender does. Reading it as one is what happened on +// 2026-08-09 while CW2 ran under its own systemd unit: mavgpud yielded, waited +// five polls, loaded the model, yielded again, and never held it for a whole +// minute. Excluding every child is the fix and this is the test of it. +func TestForeignExcludesEveryChild(t *testing.T) { + p := probe{kfdRoot: fakeKFD(t, map[int]int64{478104: 12791693312, 999: 4096, 1001: 1717986918})} + + ours := p.foreign(999, 1001) + if len(ours) != 1 || ours[0].PID != 478104 { + t.Fatalf("only the CPT run is a contender, got %+v", ours) + } + // A child that is not running reports pid 0, which must exclude nothing. + if got := p.foreign(999, 0); len(got) != 2 { + t.Errorf("a stopped child excludes nobody: got %d contenders, want 2", len(got)) + } +} + // An empty KFD tree is the state that permits a start, so it must read as empty // rather than as an error the caller has to interpret. func TestForeignEmptyAndMissing(t *testing.T) { @@ -81,7 +101,7 @@ func TestFreeVRAM(t *testing.T) { // rather than hanging or proxying into a closed port. Maven reads this endpoint // on a timer forever, including while the workstation is busy. func TestHealthAndProxyRefuseWhenNotReady(t *testing.T) { - s := &supervisor{run: newRunner("/bin/true", nil, "")} + s := &supervisor{run: newRunner("fake", "/bin/true", nil, "")} h := s.handler(mustURL(t, "http://127.0.0.1:1")) for _, path := range []string{"/health", "/v1/chat/completions"} { @@ -101,3 +121,28 @@ func mustURL(t *testing.T, s string) *url.URL { } return u } + +// Yielding is all or nothing. A CPT run wants the whole card, so handing back +// the language model while the transcriber keeps 1.6GB mapped would leave the +// other job failing its allocation, which is the outcome yielding exists to +// prevent. +func TestYieldStopsEveryChild(t *testing.T) { + idle := "while : ; do sleep 1 ; done" + s := &supervisor{ + cfg: config{EvictAfter: 1, StopGrace: duration(2 * time.Second)}, + probe: probe{kfdRoot: fakeKFD(t, map[int]int64{478104: 12791693312})}, + run: newRunner("llama-server", fakeServer(t, idle), nil, ""), + stt: newRunner("cw2", fakeServer(t, idle), nil, ""), + } + for _, r := range s.children() { + if err := r.start(); err != nil { + t.Fatal(err) + } + } + s.tick(context.Background()) + for _, r := range s.children() { + if r.running() { + t.Errorf("%s outlived the yield", r.name) + } + } +} diff --git a/cmd/mavgpud/main.go b/cmd/mavgpud/main.go index 04c6123..d2cb76e 100644 --- a/cmd/mavgpud/main.go +++ b/cmd/mavgpud/main.go @@ -10,6 +10,11 @@ // the card. Not on demand, because a 7-14B takes tens of seconds to load and a // world question would be answered by a gap every time the card had been quiet. // Not always on, because that holds 16GB against the owner's own jobs. +// +// It supervises a second child since 09-08-2026, the CW2 transcriber, and for +// one reason only: it is a ROCm process on the same card. Any GPU service the +// owner leaves running beside this daemon reads as a contender and evicts the +// model, so the card needs one owner rather than two neighbours. package main import ( @@ -36,6 +41,10 @@ type config struct { // owner's business and not this daemon's schema. LlamaArgs []string `json:"llama_args"` + // Stt is optional. Without it mavgpud supervises llama-server alone, which + // is everything it did before 09-08-2026. + Stt *sttConfig `json:"stt,omitempty"` + KFDRoot string `json:"kfd_root"` DRMDevice string `json:"drm_device"` @@ -51,6 +60,22 @@ type config struct { StartAfter int `json:"start_after_polls"` } +// sttConfig is the CW2 transcriber, which mavgpud runs for one reason: it is a +// ROCm process on this card. Left to its own systemd unit it registers on the +// KFD, the supervisor reads it as a contender, and llama-server is evicted +// within two polls and restarted five polls later, forever. That thrash was +// observed on 2026-08-09 and it is what folded the service in here. +// +// Maven talks to it directly, not through this daemon. There is no proxy and no +// idle timer: at 1.6GB it denies the card to nobody, and unloading it would only +// send the next voice turn to the homesrv floor for no gain. +type sttConfig struct { + // Addr is where the service binds, and it is read only to probe /health. + Addr string `json:"addr"` + Bin string `json:"bin"` + Args []string `json:"args"` +} + func defaults() config { return config{ Listen: ":8080", @@ -99,12 +124,18 @@ func main() { } base := "http://" + cfg.LlamaAddr - run := newRunner(cfg.LlamaBin, cfg.LlamaArgs, base+"/health") + run := newRunner("llama-server", cfg.LlamaBin, cfg.LlamaArgs, base+"/health") sup := &supervisor{ cfg: cfg, probe: probe{kfdRoot: cfg.KFDRoot, drmDev: cfg.DRMDevice}, run: run, } + if s := cfg.Stt; s != nil { + if s.Bin == "" || s.Addr == "" { + log.Fatal("mavgpud: stt needs both bin and addr") + } + sup.stt = newRunner("cw2", s.Bin, s.Args, "http://"+s.Addr+"/health") + } sup.touch() ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) @@ -129,13 +160,17 @@ func main() { shut, done := context.WithTimeout(context.Background(), 5*time.Second) defer done() _ = srv.Shutdown(shut) - run.stop(time.Duration(cfg.StopGrace)) + for _, r := range sup.children() { + r.stop(time.Duration(cfg.StopGrace)) + } } type supervisor struct { cfg config probe probe run *runner + // stt is the CW2 transcriber, or nil when the config names none. + stt *runner lastReq atomic.Int64 // unix nanos of the last request Maven sent @@ -198,7 +233,11 @@ func (s *supervisor) loop(ctx context.Context) { // allocates, so we see a contender during its startup rather than after it has // already failed to get the memory it wanted. func (s *supervisor) tick(ctx context.Context) { - others := s.probe.foreign(s.run.pid()) + var pids []int + for _, r := range s.children() { + pids = append(pids, r.pid()) + } + others := s.probe.foreign(pids...) if len(others) > 0 { s.foreignStreak++ s.clearStreak = 0 @@ -207,31 +246,65 @@ func (s *supervisor) tick(ctx context.Context) { s.clearStreak++ } - if s.run.running() { - s.run.refreshReady(ctx) - switch { - case s.foreignStreak >= s.cfg.EvictAfter: - log.Printf("mavgpud: yielding the card to %s", describe(others)) - s.run.stop(time.Duration(s.cfg.StopGrace)) - case s.idle() > time.Duration(s.cfg.IdleTimeout): - log.Printf("mavgpud: idle for %s, unloading", s.idle().Round(time.Second)) - s.run.stop(time.Duration(s.cfg.StopGrace)) + // Yielding is all or nothing. A CPT run wants the whole card, and handing + // back 8GB while holding 1.6GB is the shape of a failed allocation. + if s.foreignStreak >= s.cfg.EvictAfter && s.anyRunning() { + log.Printf("mavgpud: yielding the card to %s", describe(others)) + for _, r := range s.children() { + r.stop(time.Duration(s.cfg.StopGrace)) } return } - if s.clearStreak < s.cfg.StartAfter { + clear := s.clearStreak >= s.cfg.StartAfter + + if s.run.running() { + s.run.refreshReady(ctx) + if s.idle() > time.Duration(s.cfg.IdleTimeout) { + log.Printf("mavgpud: idle for %s, unloading", s.idle().Round(time.Second)) + s.run.stop(time.Duration(s.cfg.StopGrace)) + } + } else if clear && s.probe.freeVRAM() >= s.cfg.MinFreeVRAM { + s.touch() // the idle clock starts at load, not at the last request before it + if err := s.run.start(); err != nil { + log.Printf("mavgpud: start llama-server: %v", err) + } + } + + if s.stt == nil { return } - if free := s.probe.freeVRAM(); free < s.cfg.MinFreeVRAM { + if s.stt.running() { + s.stt.refreshReady(ctx) return } - s.touch() // the idle clock starts at load, not at the last request before it - if err := s.run.start(); err != nil { - log.Printf("mavgpud: start llama-server: %v", err) + // No VRAM precondition here, unlike llama-server. That check exists because + // a 12B refuses to load when the card is short, and 1.6GB fits wherever the + // KFD is clear. Reading free VRAM would also block the transcriber for good + // once the language model was resident, since it holds more than the floor. + if clear { + if err := s.stt.start(); err != nil { + log.Printf("mavgpud: start cw2: %v", err) + } } } +func (s *supervisor) children() []*runner { + if s.stt == nil { + return []*runner{s.run} + } + return []*runner{s.run, s.stt} +} + +func (s *supervisor) anyRunning() bool { + for _, r := range s.children() { + if r.running() { + return true + } + } + return false +} + // describe names the contenders in the log. This log is the instrument for the // open question in #488: whether polling the KFD misses a job that wants the // card without registering there. diff --git a/cmd/mavgpud/runner.go b/cmd/mavgpud/runner.go index 83a68b5..ffdb78d 100644 --- a/cmd/mavgpud/runner.go +++ b/cmd/mavgpud/runner.go @@ -10,14 +10,18 @@ import ( "time" ) -// runner owns one llama-server process. Owning it is the point of the daemon: -// the workstation cannot keep a 7-14B resident, because that holds 16GB against +// runner owns one GPU process. Owning it is the point of the daemon: the +// workstation cannot keep a 7-14B resident, because that holds 16GB against // the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that // stays up is this, which costs no VRAM, and the model comes and goes under it. +// +// There are two of them since 09-08-2026: llama-server and the CW2 transcriber. +// name is what the log calls this one. type runner struct { + name string bin string args []string - // ready is llama-server's own /health, which answers "is a model loaded". + // ready is the child's own /health, which answers "is a model loaded". // Loading a 7-14B takes tens of seconds, so started is not ready. readyURL string @@ -32,9 +36,9 @@ type runner struct { http *http.Client } -func newRunner(bin string, args []string, readyURL string) *runner { +func newRunner(name, bin string, args []string, readyURL string) *runner { return &runner{ - bin: bin, args: args, readyURL: readyURL, + name: name, bin: bin, args: args, readyURL: readyURL, http: &http.Client{Timeout: 2 * time.Second}, } } @@ -60,7 +64,7 @@ func (r *runner) isReady() bool { return r.ready } -// start launches llama-server. It returns as soon as the process exists, not +// start launches the child. It returns as soon as the process exists, not // when the model is loaded. func (r *runner) start() error { r.mu.Lock() @@ -76,7 +80,7 @@ func (r *runner) start() error { return err } r.cmd, r.ready, r.yielding = cmd, false, false - log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid) + log.Printf("mavgpud: started %s pid=%d", r.name, cmd.Process.Pid) go func() { err := cmd.Wait() r.mu.Lock() @@ -84,15 +88,15 @@ func (r *runner) start() error { r.cmd, r.ready, r.yielding = nil, false, false r.mu.Unlock() if yielded { - log.Printf("mavgpud: llama-server stopped, card yielded (%v)", err) + log.Printf("mavgpud: %s stopped, card yielded (%v)", r.name, err) return } - log.Printf("mavgpud: llama-server exited: %v", err) + log.Printf("mavgpud: %s exited: %v", r.name, err) }() return nil } -// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so +// stop ends the child and waits for the VRAM to come back. SIGTERM first so // it unmaps cleanly, SIGKILL after the grace window. Returning before the // process is gone would let the supervisor report a free card while 14GB is // still mapped, which is the one lie that would make yielding useless. @@ -117,11 +121,11 @@ func (r *runner) stop(grace time.Duration) { } time.Sleep(100 * time.Millisecond) } - log.Printf("mavgpud: llama-server did not exit in %s, killing", grace) + log.Printf("mavgpud: %s did not exit in %s, killing", r.name, grace) _ = syscall.Kill(pgid, syscall.SIGKILL) } -// refreshReady asks llama-server whether the model is loaded. Called once per +// refreshReady asks the child whether the model is loaded. Called once per // supervisor tick, never per request. func (r *runner) refreshReady(ctx context.Context) { if !r.running() { @@ -141,6 +145,6 @@ func (r *runner) refreshReady(ctx context.Context) { r.ready = ok r.mu.Unlock() if ok && !was { - log.Printf("mavgpud: model ready") + log.Printf("mavgpud: %s ready", r.name) } } diff --git a/cmd/mavgpud/runner_test.go b/cmd/mavgpud/runner_test.go index e9cae0d..7582b70 100644 --- a/cmd/mavgpud/runner_test.go +++ b/cmd/mavgpud/runner_test.go @@ -25,7 +25,7 @@ func fakeServer(t *testing.T, body string) string { // status of a routine yield is identical to that of a real crash. Reading the // mavgpud log, the two were indistinguishable (Vikunja #491). func TestStopMarksTheExitAsAYield(t *testing.T) { - r := newRunner(fakeServer(t, "while : ; do sleep 1 ; done"), nil, "") + r := newRunner("fake", fakeServer(t, "while : ; do sleep 1 ; done"), nil, "") if err := r.start(); err != nil { t.Fatalf("start: %v", err) } @@ -49,7 +49,7 @@ func TestStopMarksTheExitAsAYield(t *testing.T) { // Stopping when nothing is running must not arm the flag for the next child. // The next exit after that would be a real crash logged as a yield. func TestStopWithNoChildDoesNotArmTheFlag(t *testing.T) { - r := newRunner("/nonexistent", nil, "") + r := newRunner("fake", "/nonexistent", nil, "") r.stop(10 * time.Millisecond) r.mu.Lock() defer r.mu.Unlock() diff --git a/deploy/cw2/serve.py b/deploy/cw2/serve.py new file mode 100644 index 0000000..89da08c --- /dev/null +++ b/deploy/cw2/serve.py @@ -0,0 +1,156 @@ +"""CrisperWhisper 2.0 turbo as an HTTP service, for Maven's stt.Pair. + +Two endpoints and no framework. + + GET /health 200 once the model is loaded, 503 while it is loading. + POST /transcribe raw 16kHz mono PCM in, {"text","confidence"} out. + +The body is the PCM itself rather than JSON. A minute of 16kHz mono is under +2MB raw and about 2.6MB base64, and the format is fixed at the Maven seam, so +headers carry it more cheaply than an envelope. + +Why this exists at all: whisper.cpp cannot load CW2. It derives its language +count from the vocabulary size, and CW2's 51897 tokens shift seven special +token ids. So mavsttd stays whisper.cpp on homesrv and this runs beside the +model on workpc, where it scores 10.4% WER in Russian against the floor's 27.5% +(docs/evals/2026-08-09-crisperwhisper2-russian-wer.md in the Maven repo). + +Intended mode, not verbatim. The owner asked for what he meant to say, not +every stutter on the way there. +""" + +import hmac +import json +import logging +import os +import sys +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +import numpy as np + +HOST = os.environ.get("CW2_HOST", "0.0.0.0") +PORT = int(os.environ.get("CW2_PORT", "8081")) +SIZE = os.environ.get("CW2_SIZE", "turbo") +MODE = os.environ.get("CW2_MODE", "intended") +TOKEN = os.environ.get("CW2_TOKEN", "") +# 25MB is about thirteen minutes of 16kHz mono. Longer than any utterance and +# short enough that a wrong caller cannot exhaust memory. +MAX_BODY = int(os.environ.get("CW2_MAX_BODY", str(25 * 1024 * 1024))) + +logging.basicConfig( + level=logging.INFO, format="%(asctime)s cw2: %(message)s", stream=sys.stderr +) +log = logging.getLogger("cw2") + +_model = None +# The card holds one model and transcribes one utterance at a time. The lock is +# what makes a second caller wait rather than corrupt the first. +_lock = threading.Lock() + + +def load_model(): + global _model + from crisperwhisper import CrisperWhisperModel + + t0 = time.perf_counter() + # backend is forced. With ctranslate2 importable, "auto" picks ct2, which is + # CUDA-only and this card is AMD. + m = CrisperWhisperModel( + SIZE, backend="transformers", compute_type="float16", device="cuda" + ) + _model = m + log.info("loaded %s in %.1fs, mode=%s", SIZE, time.perf_counter() - t0, MODE) + + +def authorised(headers): + if not TOKEN: + return True + got = headers.get("Authorization", "") + return hmac.compare_digest(got, "Bearer " + TOKEN) + + +class Handler(BaseHTTPRequestHandler): + protocol_version = "HTTP/1.1" + + def log_message(self, fmt, *args): + log.info(fmt, *args) + + def _send(self, code, payload): + body = json.dumps(payload, ensure_ascii=False).encode("utf-8") + self.send_response(code) + self.send_header("Content-Type", "application/json; charset=utf-8") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_GET(self): + if self.path.rstrip("/") != "/health": + self._send(404, {"error": "not found"}) + return + if _model is None: + self._send(503, {"status": "loading"}) + return + self._send(200, {"status": "ok", "model": SIZE, "mode": MODE}) + + def do_POST(self): + if self.path.rstrip("/") != "/transcribe": + self._send(404, {"error": "not found"}) + return + if not authorised(self.headers): + self._send(401, {"error": "unauthorised"}) + return + if _model is None: + self._send(503, {"error": "loading"}) + return + + length = int(self.headers.get("Content-Length", "0")) + if length <= 0 or length > MAX_BODY: + self._send(413, {"error": "bad body length"}) + return + raw = self.rfile.read(length) + + rate = int(self.headers.get("X-Sample-Rate", "16000")) + channels = int(self.headers.get("X-Channels", "1")) + bits = int(self.headers.get("X-Sample-Bits", "16")) + lang = self.headers.get("X-Language", "ru") or "ru" + if channels != 1 or bits != 16: + self._send(400, {"error": "want 16-bit mono pcm"}) + return + + # int16 little-endian to the float32 the encoder wants. + wav = np.frombuffer(raw, dtype="