diff --git a/cmd/mavgpud/gpu.go b/cmd/mavgpud/gpu.go index 89120c6..989c2b9 100644 --- a/cmd/mavgpud/gpu.go +++ b/cmd/mavgpud/gpu.go @@ -34,22 +34,31 @@ type probe struct { drmDev string } -// foreign lists every ROCm process that is not ours. selfPID is the supervisor's -// llama-server child, or 0 when it is not running. +// foreign lists every ROCm process that is not ours. self holds the pids of the +// supervisor's own children, and a child that is not running contributes 0. +// +// There is more than one child since 09-08-2026. CW2 registers on the KFD like +// any ROCm job, so a supervisor that excluded only llama-server would read its +// own transcriber as a contender, yield the card to it, and never keep a model +// loaded again. // // An unreadable kfd tree returns no processes and no error. That is deliberate // and it is the safe direction only because startVRAM also has to agree before // anything launches: a supervisor that cannot see the KFD never sees free VRAM // either, because the CPT run holding the card shows up in the drm totals. -func (p probe) foreign(selfPID int) []gpuProc { +func (p probe) foreign(self ...int) []gpuProc { entries, err := os.ReadDir(p.kfdRoot) if err != nil { return nil } + mine := make(map[int]bool, len(self)) + for _, pid := range self { + mine[pid] = true + } var out []gpuProc for _, e := range entries { pid, err := strconv.Atoi(e.Name()) - if err != nil || pid == selfPID { + if err != nil || mine[pid] { continue } out = append(out, gpuProc{ diff --git a/cmd/mavgpud/gpu_test.go b/cmd/mavgpud/gpu_test.go index 073874b..52dc07d 100644 --- a/cmd/mavgpud/gpu_test.go +++ b/cmd/mavgpud/gpu_test.go @@ -1,6 +1,7 @@ package main import ( + "context" "net/http" "net/http/httptest" "net/url" @@ -8,6 +9,7 @@ import ( "path/filepath" "strconv" "testing" + "time" ) // fakeKFD builds the sysfs shape the workstation actually has: one directory @@ -47,6 +49,24 @@ func TestForeignExcludesOurChild(t *testing.T) { } } +// The transcriber is a ROCm process on the same card, so it registers on the +// KFD exactly like a contender does. Reading it as one is what happened on +// 2026-08-09 while CW2 ran under its own systemd unit: mavgpud yielded, waited +// five polls, loaded the model, yielded again, and never held it for a whole +// minute. Excluding every child is the fix and this is the test of it. +func TestForeignExcludesEveryChild(t *testing.T) { + p := probe{kfdRoot: fakeKFD(t, map[int]int64{478104: 12791693312, 999: 4096, 1001: 1717986918})} + + ours := p.foreign(999, 1001) + if len(ours) != 1 || ours[0].PID != 478104 { + t.Fatalf("only the CPT run is a contender, got %+v", ours) + } + // A child that is not running reports pid 0, which must exclude nothing. + if got := p.foreign(999, 0); len(got) != 2 { + t.Errorf("a stopped child excludes nobody: got %d contenders, want 2", len(got)) + } +} + // An empty KFD tree is the state that permits a start, so it must read as empty // rather than as an error the caller has to interpret. func TestForeignEmptyAndMissing(t *testing.T) { @@ -81,7 +101,7 @@ func TestFreeVRAM(t *testing.T) { // rather than hanging or proxying into a closed port. Maven reads this endpoint // on a timer forever, including while the workstation is busy. func TestHealthAndProxyRefuseWhenNotReady(t *testing.T) { - s := &supervisor{run: newRunner("/bin/true", nil, "")} + s := &supervisor{run: newRunner("fake", "/bin/true", nil, "")} h := s.handler(mustURL(t, "http://127.0.0.1:1")) for _, path := range []string{"/health", "/v1/chat/completions"} { @@ -101,3 +121,28 @@ func mustURL(t *testing.T, s string) *url.URL { } return u } + +// Yielding is all or nothing. A CPT run wants the whole card, so handing back +// the language model while the transcriber keeps 1.6GB mapped would leave the +// other job failing its allocation, which is the outcome yielding exists to +// prevent. +func TestYieldStopsEveryChild(t *testing.T) { + idle := "while : ; do sleep 1 ; done" + s := &supervisor{ + cfg: config{EvictAfter: 1, StopGrace: duration(2 * time.Second)}, + probe: probe{kfdRoot: fakeKFD(t, map[int]int64{478104: 12791693312})}, + run: newRunner("llama-server", fakeServer(t, idle), nil, ""), + stt: newRunner("cw2", fakeServer(t, idle), nil, ""), + } + for _, r := range s.children() { + if err := r.start(); err != nil { + t.Fatal(err) + } + } + s.tick(context.Background()) + for _, r := range s.children() { + if r.running() { + t.Errorf("%s outlived the yield", r.name) + } + } +} diff --git a/cmd/mavgpud/main.go b/cmd/mavgpud/main.go index 04c6123..d2cb76e 100644 --- a/cmd/mavgpud/main.go +++ b/cmd/mavgpud/main.go @@ -10,6 +10,11 @@ // the card. Not on demand, because a 7-14B takes tens of seconds to load and a // world question would be answered by a gap every time the card had been quiet. // Not always on, because that holds 16GB against the owner's own jobs. +// +// It supervises a second child since 09-08-2026, the CW2 transcriber, and for +// one reason only: it is a ROCm process on the same card. Any GPU service the +// owner leaves running beside this daemon reads as a contender and evicts the +// model, so the card needs one owner rather than two neighbours. package main import ( @@ -36,6 +41,10 @@ type config struct { // owner's business and not this daemon's schema. LlamaArgs []string `json:"llama_args"` + // Stt is optional. Without it mavgpud supervises llama-server alone, which + // is everything it did before 09-08-2026. + Stt *sttConfig `json:"stt,omitempty"` + KFDRoot string `json:"kfd_root"` DRMDevice string `json:"drm_device"` @@ -51,6 +60,22 @@ type config struct { StartAfter int `json:"start_after_polls"` } +// sttConfig is the CW2 transcriber, which mavgpud runs for one reason: it is a +// ROCm process on this card. Left to its own systemd unit it registers on the +// KFD, the supervisor reads it as a contender, and llama-server is evicted +// within two polls and restarted five polls later, forever. That thrash was +// observed on 2026-08-09 and it is what folded the service in here. +// +// Maven talks to it directly, not through this daemon. There is no proxy and no +// idle timer: at 1.6GB it denies the card to nobody, and unloading it would only +// send the next voice turn to the homesrv floor for no gain. +type sttConfig struct { + // Addr is where the service binds, and it is read only to probe /health. + Addr string `json:"addr"` + Bin string `json:"bin"` + Args []string `json:"args"` +} + func defaults() config { return config{ Listen: ":8080", @@ -99,12 +124,18 @@ func main() { } base := "http://" + cfg.LlamaAddr - run := newRunner(cfg.LlamaBin, cfg.LlamaArgs, base+"/health") + run := newRunner("llama-server", cfg.LlamaBin, cfg.LlamaArgs, base+"/health") sup := &supervisor{ cfg: cfg, probe: probe{kfdRoot: cfg.KFDRoot, drmDev: cfg.DRMDevice}, run: run, } + if s := cfg.Stt; s != nil { + if s.Bin == "" || s.Addr == "" { + log.Fatal("mavgpud: stt needs both bin and addr") + } + sup.stt = newRunner("cw2", s.Bin, s.Args, "http://"+s.Addr+"/health") + } sup.touch() ctx, cancel := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) @@ -129,13 +160,17 @@ func main() { shut, done := context.WithTimeout(context.Background(), 5*time.Second) defer done() _ = srv.Shutdown(shut) - run.stop(time.Duration(cfg.StopGrace)) + for _, r := range sup.children() { + r.stop(time.Duration(cfg.StopGrace)) + } } type supervisor struct { cfg config probe probe run *runner + // stt is the CW2 transcriber, or nil when the config names none. + stt *runner lastReq atomic.Int64 // unix nanos of the last request Maven sent @@ -198,7 +233,11 @@ func (s *supervisor) loop(ctx context.Context) { // allocates, so we see a contender during its startup rather than after it has // already failed to get the memory it wanted. func (s *supervisor) tick(ctx context.Context) { - others := s.probe.foreign(s.run.pid()) + var pids []int + for _, r := range s.children() { + pids = append(pids, r.pid()) + } + others := s.probe.foreign(pids...) if len(others) > 0 { s.foreignStreak++ s.clearStreak = 0 @@ -207,31 +246,65 @@ func (s *supervisor) tick(ctx context.Context) { s.clearStreak++ } - if s.run.running() { - s.run.refreshReady(ctx) - switch { - case s.foreignStreak >= s.cfg.EvictAfter: - log.Printf("mavgpud: yielding the card to %s", describe(others)) - s.run.stop(time.Duration(s.cfg.StopGrace)) - case s.idle() > time.Duration(s.cfg.IdleTimeout): - log.Printf("mavgpud: idle for %s, unloading", s.idle().Round(time.Second)) - s.run.stop(time.Duration(s.cfg.StopGrace)) + // Yielding is all or nothing. A CPT run wants the whole card, and handing + // back 8GB while holding 1.6GB is the shape of a failed allocation. + if s.foreignStreak >= s.cfg.EvictAfter && s.anyRunning() { + log.Printf("mavgpud: yielding the card to %s", describe(others)) + for _, r := range s.children() { + r.stop(time.Duration(s.cfg.StopGrace)) } return } - if s.clearStreak < s.cfg.StartAfter { + clear := s.clearStreak >= s.cfg.StartAfter + + if s.run.running() { + s.run.refreshReady(ctx) + if s.idle() > time.Duration(s.cfg.IdleTimeout) { + log.Printf("mavgpud: idle for %s, unloading", s.idle().Round(time.Second)) + s.run.stop(time.Duration(s.cfg.StopGrace)) + } + } else if clear && s.probe.freeVRAM() >= s.cfg.MinFreeVRAM { + s.touch() // the idle clock starts at load, not at the last request before it + if err := s.run.start(); err != nil { + log.Printf("mavgpud: start llama-server: %v", err) + } + } + + if s.stt == nil { return } - if free := s.probe.freeVRAM(); free < s.cfg.MinFreeVRAM { + if s.stt.running() { + s.stt.refreshReady(ctx) return } - s.touch() // the idle clock starts at load, not at the last request before it - if err := s.run.start(); err != nil { - log.Printf("mavgpud: start llama-server: %v", err) + // No VRAM precondition here, unlike llama-server. That check exists because + // a 12B refuses to load when the card is short, and 1.6GB fits wherever the + // KFD is clear. Reading free VRAM would also block the transcriber for good + // once the language model was resident, since it holds more than the floor. + if clear { + if err := s.stt.start(); err != nil { + log.Printf("mavgpud: start cw2: %v", err) + } } } +func (s *supervisor) children() []*runner { + if s.stt == nil { + return []*runner{s.run} + } + return []*runner{s.run, s.stt} +} + +func (s *supervisor) anyRunning() bool { + for _, r := range s.children() { + if r.running() { + return true + } + } + return false +} + // describe names the contenders in the log. This log is the instrument for the // open question in #488: whether polling the KFD misses a job that wants the // card without registering there. diff --git a/cmd/mavgpud/runner.go b/cmd/mavgpud/runner.go index 83a68b5..ffdb78d 100644 --- a/cmd/mavgpud/runner.go +++ b/cmd/mavgpud/runner.go @@ -10,14 +10,18 @@ import ( "time" ) -// runner owns one llama-server process. Owning it is the point of the daemon: -// the workstation cannot keep a 7-14B resident, because that holds 16GB against +// runner owns one GPU process. Owning it is the point of the daemon: the +// workstation cannot keep a 7-14B resident, because that holds 16GB against // the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that // stays up is this, which costs no VRAM, and the model comes and goes under it. +// +// There are two of them since 09-08-2026: llama-server and the CW2 transcriber. +// name is what the log calls this one. type runner struct { + name string bin string args []string - // ready is llama-server's own /health, which answers "is a model loaded". + // ready is the child's own /health, which answers "is a model loaded". // Loading a 7-14B takes tens of seconds, so started is not ready. readyURL string @@ -32,9 +36,9 @@ type runner struct { http *http.Client } -func newRunner(bin string, args []string, readyURL string) *runner { +func newRunner(name, bin string, args []string, readyURL string) *runner { return &runner{ - bin: bin, args: args, readyURL: readyURL, + name: name, bin: bin, args: args, readyURL: readyURL, http: &http.Client{Timeout: 2 * time.Second}, } } @@ -60,7 +64,7 @@ func (r *runner) isReady() bool { return r.ready } -// start launches llama-server. It returns as soon as the process exists, not +// start launches the child. It returns as soon as the process exists, not // when the model is loaded. func (r *runner) start() error { r.mu.Lock() @@ -76,7 +80,7 @@ func (r *runner) start() error { return err } r.cmd, r.ready, r.yielding = cmd, false, false - log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid) + log.Printf("mavgpud: started %s pid=%d", r.name, cmd.Process.Pid) go func() { err := cmd.Wait() r.mu.Lock() @@ -84,15 +88,15 @@ func (r *runner) start() error { r.cmd, r.ready, r.yielding = nil, false, false r.mu.Unlock() if yielded { - log.Printf("mavgpud: llama-server stopped, card yielded (%v)", err) + log.Printf("mavgpud: %s stopped, card yielded (%v)", r.name, err) return } - log.Printf("mavgpud: llama-server exited: %v", err) + log.Printf("mavgpud: %s exited: %v", r.name, err) }() return nil } -// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so +// stop ends the child and waits for the VRAM to come back. SIGTERM first so // it unmaps cleanly, SIGKILL after the grace window. Returning before the // process is gone would let the supervisor report a free card while 14GB is // still mapped, which is the one lie that would make yielding useless. @@ -117,11 +121,11 @@ func (r *runner) stop(grace time.Duration) { } time.Sleep(100 * time.Millisecond) } - log.Printf("mavgpud: llama-server did not exit in %s, killing", grace) + log.Printf("mavgpud: %s did not exit in %s, killing", r.name, grace) _ = syscall.Kill(pgid, syscall.SIGKILL) } -// refreshReady asks llama-server whether the model is loaded. Called once per +// refreshReady asks the child whether the model is loaded. Called once per // supervisor tick, never per request. func (r *runner) refreshReady(ctx context.Context) { if !r.running() { @@ -141,6 +145,6 @@ func (r *runner) refreshReady(ctx context.Context) { r.ready = ok r.mu.Unlock() if ok && !was { - log.Printf("mavgpud: model ready") + log.Printf("mavgpud: %s ready", r.name) } } diff --git a/cmd/mavgpud/runner_test.go b/cmd/mavgpud/runner_test.go index e9cae0d..7582b70 100644 --- a/cmd/mavgpud/runner_test.go +++ b/cmd/mavgpud/runner_test.go @@ -25,7 +25,7 @@ func fakeServer(t *testing.T, body string) string { // status of a routine yield is identical to that of a real crash. Reading the // mavgpud log, the two were indistinguishable (Vikunja #491). func TestStopMarksTheExitAsAYield(t *testing.T) { - r := newRunner(fakeServer(t, "while : ; do sleep 1 ; done"), nil, "") + r := newRunner("fake", fakeServer(t, "while : ; do sleep 1 ; done"), nil, "") if err := r.start(); err != nil { t.Fatalf("start: %v", err) } @@ -49,7 +49,7 @@ func TestStopMarksTheExitAsAYield(t *testing.T) { // Stopping when nothing is running must not arm the flag for the next child. // The next exit after that would be a real crash logged as a yield. func TestStopWithNoChildDoesNotArmTheFlag(t *testing.T) { - r := newRunner("/nonexistent", nil, "") + r := newRunner("fake", "/nonexistent", nil, "") r.stop(10 * time.Millisecond) r.mu.Lock() defer r.mu.Unlock()