mavgpud: read the card from sysfs and own llama-server's lifecycle (V-488)
The workstation cannot keep a 7-14B resident: it would hold 16GB against the owner's CPT runs, Correx and the manga-recap pipeline. So the process that stays up costs no VRAM and the model comes and goes under it. Contention is detected by presence on the KFD, not by a VRAM threshold. A ROCm process registers under /sys/class/kfd/kfd/proc when it initialises HIP, well before it allocates, so we see a contender during its startup instead of after it has already lost an allocation race. rocm-smi is not installed on that box and a per-second subprocess would get tuned down until useless, so this reads sysfs and forks nothing. Free VRAM is read only to decide whether to start. It is never a reason to stop: by the time free VRAM has dropped, the other job has already failed.
This commit is contained in:
@@ -0,0 +1,132 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"net/http"
|
||||
"os/exec"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// runner owns one llama-server process. Owning it is the point of the daemon:
|
||||
// the workstation cannot keep a 7-14B resident, because that holds 16GB against
|
||||
// the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that
|
||||
// stays up is this, which costs no VRAM, and the model comes and goes under it.
|
||||
type runner struct {
|
||||
bin string
|
||||
args []string
|
||||
// ready is llama-server's own /health, which answers "is a model loaded".
|
||||
// Loading a 7-14B takes tens of seconds, so started is not ready.
|
||||
readyURL string
|
||||
|
||||
mu sync.Mutex
|
||||
cmd *exec.Cmd
|
||||
ready bool
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
func newRunner(bin string, args []string, readyURL string) *runner {
|
||||
return &runner{
|
||||
bin: bin, args: args, readyURL: readyURL,
|
||||
http: &http.Client{Timeout: 2 * time.Second},
|
||||
}
|
||||
}
|
||||
|
||||
// pid is the child's, or 0. The GPU probe needs it to tell our own model apart
|
||||
// from a contender.
|
||||
func (r *runner) pid() int {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
if r.cmd == nil || r.cmd.Process == nil {
|
||||
return 0
|
||||
}
|
||||
return r.cmd.Process.Pid
|
||||
}
|
||||
|
||||
func (r *runner) running() bool { return r.pid() != 0 }
|
||||
|
||||
// isReady reports the cached readiness. The supervisor loop refreshes it; the
|
||||
// health handler only reads, so answering /health never costs a round trip.
|
||||
func (r *runner) isReady() bool {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
return r.ready
|
||||
}
|
||||
|
||||
// start launches llama-server. It returns as soon as the process exists, not
|
||||
// when the model is loaded.
|
||||
func (r *runner) start() error {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
if r.cmd != nil {
|
||||
return nil
|
||||
}
|
||||
cmd := exec.Command(r.bin, r.args...)
|
||||
// Own process group, so stop kills anything llama-server spawned rather
|
||||
// than leaving it holding VRAM after we have declared the card yielded.
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
|
||||
if err := cmd.Start(); err != nil {
|
||||
return err
|
||||
}
|
||||
r.cmd, r.ready = cmd, false
|
||||
log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid)
|
||||
go func() {
|
||||
err := cmd.Wait()
|
||||
r.mu.Lock()
|
||||
r.cmd, r.ready = nil, false
|
||||
r.mu.Unlock()
|
||||
log.Printf("mavgpud: llama-server exited: %v", err)
|
||||
}()
|
||||
return nil
|
||||
}
|
||||
|
||||
// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so
|
||||
// it unmaps cleanly, SIGKILL after the grace window. Returning before the
|
||||
// process is gone would let the supervisor report a free card while 14GB is
|
||||
// still mapped, which is the one lie that would make yielding useless.
|
||||
func (r *runner) stop(grace time.Duration) {
|
||||
r.mu.Lock()
|
||||
cmd := r.cmd
|
||||
r.ready = false
|
||||
r.mu.Unlock()
|
||||
if cmd == nil || cmd.Process == nil {
|
||||
return
|
||||
}
|
||||
pgid := -cmd.Process.Pid
|
||||
_ = syscall.Kill(pgid, syscall.SIGTERM)
|
||||
deadline := time.Now().Add(grace)
|
||||
for time.Now().Before(deadline) {
|
||||
if !r.running() {
|
||||
return
|
||||
}
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
}
|
||||
log.Printf("mavgpud: llama-server did not exit in %s, killing", grace)
|
||||
_ = syscall.Kill(pgid, syscall.SIGKILL)
|
||||
}
|
||||
|
||||
// refreshReady asks llama-server whether the model is loaded. Called once per
|
||||
// supervisor tick, never per request.
|
||||
func (r *runner) refreshReady(ctx context.Context) {
|
||||
if !r.running() {
|
||||
return
|
||||
}
|
||||
ok := false
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, r.readyURL, nil)
|
||||
if err == nil {
|
||||
resp, err := r.http.Do(req)
|
||||
if err == nil {
|
||||
ok = resp.StatusCode == http.StatusOK
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
r.mu.Lock()
|
||||
was := r.ready
|
||||
r.ready = ok
|
||||
r.mu.Unlock()
|
||||
if ok && !was {
|
||||
log.Printf("mavgpud: model ready")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user