mavgpud: read the card from sysfs and own llama-server's lifecycle (V-488)

The workstation cannot keep a 7-14B resident: it would hold 16GB against the
owner's CPT runs, Correx and the manga-recap pipeline. So the process that
stays up costs no VRAM and the model comes and goes under it.

Contention is detected by presence on the KFD, not by a VRAM threshold. A ROCm
process registers under /sys/class/kfd/kfd/proc when it initialises HIP, well
before it allocates, so we see a contender during its startup instead of after
it has already lost an allocation race. rocm-smi is not installed on that box
and a per-second subprocess would get tuned down until useless, so this reads
sysfs and forks nothing.

Free VRAM is read only to decide whether to start. It is never a reason to
stop: by the time free VRAM has dropped, the other job has already failed.
This commit is contained in:
2026-08-02 18:49:01 +04:00
committed by kami
parent 94d553570d
commit ab42db2b87
2 changed files with 249 additions and 0 deletions
+117
View File
@@ -0,0 +1,117 @@
package main
import (
"os"
"path/filepath"
"strconv"
"strings"
)
// The card is an AMD 7900 GRE with 16GB, driven by amdgpu and ROCm. Everything
// here reads sysfs and forks nothing: rocm-smi is not even installed on the
// workstation, and a poll that costs a subprocess every second is a poll that
// gets tuned down until it is useless.
// gpuProc — one process holding the compute engine.
type gpuProc struct {
PID int
Comm string
VRAM int64 // bytes, as the kernel accounts them to this process
}
// probe reads the two sysfs trees the supervisor decides from.
//
// kfdRoot is /sys/class/kfd/kfd/proc, one directory per ROCm process. The
// directory appears when the process initialises HIP, which is well before it
// allocates anything large. That is the whole reason this works: the job that
// is about to want the card announces itself while it is still starting up,
// so we see the contender rather than only the winner of an allocation race.
//
// drmDev is /sys/class/drm/cardN/device, which reports total and used VRAM for
// the card as a whole.
type probe struct {
kfdRoot string
drmDev string
}
// foreign lists every ROCm process that is not ours. selfPID is the supervisor's
// llama-server child, or 0 when it is not running.
//
// An unreadable kfd tree returns no processes and no error. That is deliberate
// and it is the safe direction only because startVRAM also has to agree before
// anything launches: a supervisor that cannot see the KFD never sees free VRAM
// either, because the CPT run holding the card shows up in the drm totals.
func (p probe) foreign(selfPID int) []gpuProc {
entries, err := os.ReadDir(p.kfdRoot)
if err != nil {
return nil
}
var out []gpuProc
for _, e := range entries {
pid, err := strconv.Atoi(e.Name())
if err != nil || pid == selfPID {
continue
}
out = append(out, gpuProc{
PID: pid,
Comm: readComm(pid),
VRAM: p.procVRAM(e.Name()),
})
}
return out
}
// procVRAM sums the per-node vram_* files under one process directory. The
// suffix is the KFD topology node id (vram_35881 on this card), so it is
// globbed rather than named, and a machine with two cards sums both.
func (p probe) procVRAM(pid string) int64 {
matches, err := filepath.Glob(filepath.Join(p.kfdRoot, pid, "vram_*"))
if err != nil {
return 0
}
var total int64
for _, m := range matches {
total += readInt(m)
}
return total
}
// freeVRAM reports the bytes the card has left. Used only to decide whether to
// start: a shortfall here means llama-server would refuse to load anyway. It is
// never used to decide to stop, because by the time free VRAM has dropped the
// other job has already failed its allocation, which is exactly the outcome
// yielding exists to prevent.
func (p probe) freeVRAM() int64 {
total := readInt(filepath.Join(p.drmDev, "mem_info_vram_total"))
used := readInt(filepath.Join(p.drmDev, "mem_info_vram_used"))
if total <= 0 {
return 0
}
if free := total - used; free > 0 {
return free
}
return 0
}
func readInt(path string) int64 {
b, err := os.ReadFile(path)
if err != nil {
return 0
}
n, err := strconv.ParseInt(strings.TrimSpace(string(b)), 10, 64)
if err != nil {
return 0
}
return n
}
// readComm names the contender for the log. The log is the instrument for the
// open question in Vikunja #488: whether a process can want this card without
// ever registering on the KFD, which a Vulkan or video-decode job would.
func readComm(pid int) string {
b, err := os.ReadFile(filepath.Join("/proc", strconv.Itoa(pid), "comm"))
if err != nil {
return "?"
}
return strings.TrimSpace(string(b))
}
+132
View File
@@ -0,0 +1,132 @@
package main
import (
"context"
"log"
"net/http"
"os/exec"
"sync"
"syscall"
"time"
)
// runner owns one llama-server process. Owning it is the point of the daemon:
// the workstation cannot keep a 7-14B resident, because that holds 16GB against
// the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that
// stays up is this, which costs no VRAM, and the model comes and goes under it.
type runner struct {
bin string
args []string
// ready is llama-server's own /health, which answers "is a model loaded".
// Loading a 7-14B takes tens of seconds, so started is not ready.
readyURL string
mu sync.Mutex
cmd *exec.Cmd
ready bool
http *http.Client
}
func newRunner(bin string, args []string, readyURL string) *runner {
return &runner{
bin: bin, args: args, readyURL: readyURL,
http: &http.Client{Timeout: 2 * time.Second},
}
}
// pid is the child's, or 0. The GPU probe needs it to tell our own model apart
// from a contender.
func (r *runner) pid() int {
r.mu.Lock()
defer r.mu.Unlock()
if r.cmd == nil || r.cmd.Process == nil {
return 0
}
return r.cmd.Process.Pid
}
func (r *runner) running() bool { return r.pid() != 0 }
// isReady reports the cached readiness. The supervisor loop refreshes it; the
// health handler only reads, so answering /health never costs a round trip.
func (r *runner) isReady() bool {
r.mu.Lock()
defer r.mu.Unlock()
return r.ready
}
// start launches llama-server. It returns as soon as the process exists, not
// when the model is loaded.
func (r *runner) start() error {
r.mu.Lock()
defer r.mu.Unlock()
if r.cmd != nil {
return nil
}
cmd := exec.Command(r.bin, r.args...)
// Own process group, so stop kills anything llama-server spawned rather
// than leaving it holding VRAM after we have declared the card yielded.
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
if err := cmd.Start(); err != nil {
return err
}
r.cmd, r.ready = cmd, false
log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid)
go func() {
err := cmd.Wait()
r.mu.Lock()
r.cmd, r.ready = nil, false
r.mu.Unlock()
log.Printf("mavgpud: llama-server exited: %v", err)
}()
return nil
}
// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so
// it unmaps cleanly, SIGKILL after the grace window. Returning before the
// process is gone would let the supervisor report a free card while 14GB is
// still mapped, which is the one lie that would make yielding useless.
func (r *runner) stop(grace time.Duration) {
r.mu.Lock()
cmd := r.cmd
r.ready = false
r.mu.Unlock()
if cmd == nil || cmd.Process == nil {
return
}
pgid := -cmd.Process.Pid
_ = syscall.Kill(pgid, syscall.SIGTERM)
deadline := time.Now().Add(grace)
for time.Now().Before(deadline) {
if !r.running() {
return
}
time.Sleep(100 * time.Millisecond)
}
log.Printf("mavgpud: llama-server did not exit in %s, killing", grace)
_ = syscall.Kill(pgid, syscall.SIGKILL)
}
// refreshReady asks llama-server whether the model is loaded. Called once per
// supervisor tick, never per request.
func (r *runner) refreshReady(ctx context.Context) {
if !r.running() {
return
}
ok := false
req, err := http.NewRequestWithContext(ctx, http.MethodGet, r.readyURL, nil)
if err == nil {
resp, err := r.http.Do(req)
if err == nil {
ok = resp.StatusCode == http.StatusOK
resp.Body.Close()
}
}
r.mu.Lock()
was := r.ready
r.ready = ok
r.mu.Unlock()
if ok && !was {
log.Printf("mavgpud: model ready")
}
}