b975716759
CW2 is a ROCm process, so it registers on the KFD like any contender. Running it as its own systemd unit made mavgpud yield llama-server to it every few seconds. The gemma-4-12b arm was down for eight minutes on 2026-08-09 and routing had silently fallen back to the resident model. So mavgpud takes an `stt` block and runs the transcriber itself. `foreign` now excludes every child rather than one pid, which is the fix. Yielding is all or nothing, because a job that wants the card wants all of it. Idle unloading stays llama-server's alone: CW2 holds 1.6GB and unloading it would only send the next voice turn to the homesrv floor. Maven still talks to the transcriber directly on 8081. There is no proxy, because with no idle timer there is nothing for one to measure. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013ptwopxyo3Z2kwFckHkLvN
127 lines
3.9 KiB
Go
127 lines
3.9 KiB
Go
package main
|
|
|
|
import (
|
|
"os"
|
|
"path/filepath"
|
|
"strconv"
|
|
"strings"
|
|
)
|
|
|
|
// The card is an AMD 7900 GRE with 16GB, driven by amdgpu and ROCm. Everything
|
|
// here reads sysfs and forks nothing: rocm-smi is not even installed on the
|
|
// workstation, and a poll that costs a subprocess every second is a poll that
|
|
// gets tuned down until it is useless.
|
|
|
|
// gpuProc — one process holding the compute engine.
|
|
type gpuProc struct {
|
|
PID int
|
|
Comm string
|
|
VRAM int64 // bytes, as the kernel accounts them to this process
|
|
}
|
|
|
|
// probe reads the two sysfs trees the supervisor decides from.
|
|
//
|
|
// kfdRoot is /sys/class/kfd/kfd/proc, one directory per ROCm process. The
|
|
// directory appears when the process initialises HIP, which is well before it
|
|
// allocates anything large. That is the whole reason this works: the job that
|
|
// is about to want the card announces itself while it is still starting up,
|
|
// so we see the contender rather than only the winner of an allocation race.
|
|
//
|
|
// drmDev is /sys/class/drm/cardN/device, which reports total and used VRAM for
|
|
// the card as a whole.
|
|
type probe struct {
|
|
kfdRoot string
|
|
drmDev string
|
|
}
|
|
|
|
// foreign lists every ROCm process that is not ours. self holds the pids of the
|
|
// supervisor's own children, and a child that is not running contributes 0.
|
|
//
|
|
// There is more than one child since 09-08-2026. CW2 registers on the KFD like
|
|
// any ROCm job, so a supervisor that excluded only llama-server would read its
|
|
// own transcriber as a contender, yield the card to it, and never keep a model
|
|
// loaded again.
|
|
//
|
|
// An unreadable kfd tree returns no processes and no error. That is deliberate
|
|
// and it is the safe direction only because startVRAM also has to agree before
|
|
// anything launches: a supervisor that cannot see the KFD never sees free VRAM
|
|
// either, because the CPT run holding the card shows up in the drm totals.
|
|
func (p probe) foreign(self ...int) []gpuProc {
|
|
entries, err := os.ReadDir(p.kfdRoot)
|
|
if err != nil {
|
|
return nil
|
|
}
|
|
mine := make(map[int]bool, len(self))
|
|
for _, pid := range self {
|
|
mine[pid] = true
|
|
}
|
|
var out []gpuProc
|
|
for _, e := range entries {
|
|
pid, err := strconv.Atoi(e.Name())
|
|
if err != nil || mine[pid] {
|
|
continue
|
|
}
|
|
out = append(out, gpuProc{
|
|
PID: pid,
|
|
Comm: readComm(pid),
|
|
VRAM: p.procVRAM(e.Name()),
|
|
})
|
|
}
|
|
return out
|
|
}
|
|
|
|
// procVRAM sums the per-node vram_* files under one process directory. The
|
|
// suffix is the KFD topology node id (vram_35881 on this card), so it is
|
|
// globbed rather than named, and a machine with two cards sums both.
|
|
func (p probe) procVRAM(pid string) int64 {
|
|
matches, err := filepath.Glob(filepath.Join(p.kfdRoot, pid, "vram_*"))
|
|
if err != nil {
|
|
return 0
|
|
}
|
|
var total int64
|
|
for _, m := range matches {
|
|
total += readInt(m)
|
|
}
|
|
return total
|
|
}
|
|
|
|
// freeVRAM reports the bytes the card has left. Used only to decide whether to
|
|
// start: a shortfall here means llama-server would refuse to load anyway. It is
|
|
// never used to decide to stop, because by the time free VRAM has dropped the
|
|
// other job has already failed its allocation, which is exactly the outcome
|
|
// yielding exists to prevent.
|
|
func (p probe) freeVRAM() int64 {
|
|
total := readInt(filepath.Join(p.drmDev, "mem_info_vram_total"))
|
|
used := readInt(filepath.Join(p.drmDev, "mem_info_vram_used"))
|
|
if total <= 0 {
|
|
return 0
|
|
}
|
|
if free := total - used; free > 0 {
|
|
return free
|
|
}
|
|
return 0
|
|
}
|
|
|
|
func readInt(path string) int64 {
|
|
b, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return 0
|
|
}
|
|
n, err := strconv.ParseInt(strings.TrimSpace(string(b)), 10, 64)
|
|
if err != nil {
|
|
return 0
|
|
}
|
|
return n
|
|
}
|
|
|
|
// readComm names the contender for the log. The log is the instrument for the
|
|
// open question in Vikunja #488: whether a process can want this card without
|
|
// ever registering on the KFD, which a Vulkan or video-decode job would.
|
|
func readComm(pid int) string {
|
|
b, err := os.ReadFile(filepath.Join("/proc", strconv.Itoa(pid), "comm"))
|
|
if err != nil {
|
|
return "?"
|
|
}
|
|
return strings.TrimSpace(string(b))
|
|
}
|