diff --git a/cmd/mavgpud/gpu.go b/cmd/mavgpud/gpu.go new file mode 100644 index 0000000..89120c6 --- /dev/null +++ b/cmd/mavgpud/gpu.go @@ -0,0 +1,117 @@ +package main + +import ( + "os" + "path/filepath" + "strconv" + "strings" +) + +// The card is an AMD 7900 GRE with 16GB, driven by amdgpu and ROCm. Everything +// here reads sysfs and forks nothing: rocm-smi is not even installed on the +// workstation, and a poll that costs a subprocess every second is a poll that +// gets tuned down until it is useless. + +// gpuProc — one process holding the compute engine. +type gpuProc struct { + PID int + Comm string + VRAM int64 // bytes, as the kernel accounts them to this process +} + +// probe reads the two sysfs trees the supervisor decides from. +// +// kfdRoot is /sys/class/kfd/kfd/proc, one directory per ROCm process. The +// directory appears when the process initialises HIP, which is well before it +// allocates anything large. That is the whole reason this works: the job that +// is about to want the card announces itself while it is still starting up, +// so we see the contender rather than only the winner of an allocation race. +// +// drmDev is /sys/class/drm/cardN/device, which reports total and used VRAM for +// the card as a whole. +type probe struct { + kfdRoot string + drmDev string +} + +// foreign lists every ROCm process that is not ours. selfPID is the supervisor's +// llama-server child, or 0 when it is not running. +// +// An unreadable kfd tree returns no processes and no error. That is deliberate +// and it is the safe direction only because startVRAM also has to agree before +// anything launches: a supervisor that cannot see the KFD never sees free VRAM +// either, because the CPT run holding the card shows up in the drm totals. +func (p probe) foreign(selfPID int) []gpuProc { + entries, err := os.ReadDir(p.kfdRoot) + if err != nil { + return nil + } + var out []gpuProc + for _, e := range entries { + pid, err := strconv.Atoi(e.Name()) + if err != nil || pid == selfPID { + continue + } + out = append(out, gpuProc{ + PID: pid, + Comm: readComm(pid), + VRAM: p.procVRAM(e.Name()), + }) + } + return out +} + +// procVRAM sums the per-node vram_* files under one process directory. The +// suffix is the KFD topology node id (vram_35881 on this card), so it is +// globbed rather than named, and a machine with two cards sums both. +func (p probe) procVRAM(pid string) int64 { + matches, err := filepath.Glob(filepath.Join(p.kfdRoot, pid, "vram_*")) + if err != nil { + return 0 + } + var total int64 + for _, m := range matches { + total += readInt(m) + } + return total +} + +// freeVRAM reports the bytes the card has left. Used only to decide whether to +// start: a shortfall here means llama-server would refuse to load anyway. It is +// never used to decide to stop, because by the time free VRAM has dropped the +// other job has already failed its allocation, which is exactly the outcome +// yielding exists to prevent. +func (p probe) freeVRAM() int64 { + total := readInt(filepath.Join(p.drmDev, "mem_info_vram_total")) + used := readInt(filepath.Join(p.drmDev, "mem_info_vram_used")) + if total <= 0 { + return 0 + } + if free := total - used; free > 0 { + return free + } + return 0 +} + +func readInt(path string) int64 { + b, err := os.ReadFile(path) + if err != nil { + return 0 + } + n, err := strconv.ParseInt(strings.TrimSpace(string(b)), 10, 64) + if err != nil { + return 0 + } + return n +} + +// readComm names the contender for the log. The log is the instrument for the +// open question in Vikunja #488: whether a process can want this card without +// ever registering on the KFD, which a Vulkan or video-decode job would. +func readComm(pid int) string { + b, err := os.ReadFile(filepath.Join("/proc", strconv.Itoa(pid), "comm")) + if err != nil { + return "?" + } + return strings.TrimSpace(string(b)) +} diff --git a/cmd/mavgpud/runner.go b/cmd/mavgpud/runner.go new file mode 100644 index 0000000..39ce341 --- /dev/null +++ b/cmd/mavgpud/runner.go @@ -0,0 +1,132 @@ +package main + +import ( + "context" + "log" + "net/http" + "os/exec" + "sync" + "syscall" + "time" +) + +// runner owns one llama-server process. Owning it is the point of the daemon: +// the workstation cannot keep a 7-14B resident, because that holds 16GB against +// the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that +// stays up is this, which costs no VRAM, and the model comes and goes under it. +type runner struct { + bin string + args []string + // ready is llama-server's own /health, which answers "is a model loaded". + // Loading a 7-14B takes tens of seconds, so started is not ready. + readyURL string + + mu sync.Mutex + cmd *exec.Cmd + ready bool + http *http.Client +} + +func newRunner(bin string, args []string, readyURL string) *runner { + return &runner{ + bin: bin, args: args, readyURL: readyURL, + http: &http.Client{Timeout: 2 * time.Second}, + } +} + +// pid is the child's, or 0. The GPU probe needs it to tell our own model apart +// from a contender. +func (r *runner) pid() int { + r.mu.Lock() + defer r.mu.Unlock() + if r.cmd == nil || r.cmd.Process == nil { + return 0 + } + return r.cmd.Process.Pid +} + +func (r *runner) running() bool { return r.pid() != 0 } + +// isReady reports the cached readiness. The supervisor loop refreshes it; the +// health handler only reads, so answering /health never costs a round trip. +func (r *runner) isReady() bool { + r.mu.Lock() + defer r.mu.Unlock() + return r.ready +} + +// start launches llama-server. It returns as soon as the process exists, not +// when the model is loaded. +func (r *runner) start() error { + r.mu.Lock() + defer r.mu.Unlock() + if r.cmd != nil { + return nil + } + cmd := exec.Command(r.bin, r.args...) + // Own process group, so stop kills anything llama-server spawned rather + // than leaving it holding VRAM after we have declared the card yielded. + cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true} + if err := cmd.Start(); err != nil { + return err + } + r.cmd, r.ready = cmd, false + log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid) + go func() { + err := cmd.Wait() + r.mu.Lock() + r.cmd, r.ready = nil, false + r.mu.Unlock() + log.Printf("mavgpud: llama-server exited: %v", err) + }() + return nil +} + +// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so +// it unmaps cleanly, SIGKILL after the grace window. Returning before the +// process is gone would let the supervisor report a free card while 14GB is +// still mapped, which is the one lie that would make yielding useless. +func (r *runner) stop(grace time.Duration) { + r.mu.Lock() + cmd := r.cmd + r.ready = false + r.mu.Unlock() + if cmd == nil || cmd.Process == nil { + return + } + pgid := -cmd.Process.Pid + _ = syscall.Kill(pgid, syscall.SIGTERM) + deadline := time.Now().Add(grace) + for time.Now().Before(deadline) { + if !r.running() { + return + } + time.Sleep(100 * time.Millisecond) + } + log.Printf("mavgpud: llama-server did not exit in %s, killing", grace) + _ = syscall.Kill(pgid, syscall.SIGKILL) +} + +// refreshReady asks llama-server whether the model is loaded. Called once per +// supervisor tick, never per request. +func (r *runner) refreshReady(ctx context.Context) { + if !r.running() { + return + } + ok := false + req, err := http.NewRequestWithContext(ctx, http.MethodGet, r.readyURL, nil) + if err == nil { + resp, err := r.http.Do(req) + if err == nil { + ok = resp.StatusCode == http.StatusOK + resp.Body.Close() + } + } + r.mu.Lock() + was := r.ready + r.ready = ok + r.mu.Unlock() + if ok && !was { + log.Printf("mavgpud: model ready") + } +}