ab42db2b87
The workstation cannot keep a 7-14B resident: it would hold 16GB against the owner's CPT runs, Correx and the manga-recap pipeline. So the process that stays up costs no VRAM and the model comes and goes under it. Contention is detected by presence on the KFD, not by a VRAM threshold. A ROCm process registers under /sys/class/kfd/kfd/proc when it initialises HIP, well before it allocates, so we see a contender during its startup instead of after it has already lost an allocation race. rocm-smi is not installed on that box and a per-second subprocess would get tuned down until useless, so this reads sysfs and forks nothing. Free VRAM is read only to decide whether to start. It is never a reason to stop: by the time free VRAM has dropped, the other job has already failed.
133 lines
3.4 KiB
Go
133 lines
3.4 KiB
Go
package main
|
|
|
|
import (
|
|
"context"
|
|
"log"
|
|
"net/http"
|
|
"os/exec"
|
|
"sync"
|
|
"syscall"
|
|
"time"
|
|
)
|
|
|
|
// runner owns one llama-server process. Owning it is the point of the daemon:
|
|
// the workstation cannot keep a 7-14B resident, because that holds 16GB against
|
|
// the owner's CPT runs, Correx and the manga-recap pipeline. So the thing that
|
|
// stays up is this, which costs no VRAM, and the model comes and goes under it.
|
|
type runner struct {
|
|
bin string
|
|
args []string
|
|
// ready is llama-server's own /health, which answers "is a model loaded".
|
|
// Loading a 7-14B takes tens of seconds, so started is not ready.
|
|
readyURL string
|
|
|
|
mu sync.Mutex
|
|
cmd *exec.Cmd
|
|
ready bool
|
|
http *http.Client
|
|
}
|
|
|
|
func newRunner(bin string, args []string, readyURL string) *runner {
|
|
return &runner{
|
|
bin: bin, args: args, readyURL: readyURL,
|
|
http: &http.Client{Timeout: 2 * time.Second},
|
|
}
|
|
}
|
|
|
|
// pid is the child's, or 0. The GPU probe needs it to tell our own model apart
|
|
// from a contender.
|
|
func (r *runner) pid() int {
|
|
r.mu.Lock()
|
|
defer r.mu.Unlock()
|
|
if r.cmd == nil || r.cmd.Process == nil {
|
|
return 0
|
|
}
|
|
return r.cmd.Process.Pid
|
|
}
|
|
|
|
func (r *runner) running() bool { return r.pid() != 0 }
|
|
|
|
// isReady reports the cached readiness. The supervisor loop refreshes it; the
|
|
// health handler only reads, so answering /health never costs a round trip.
|
|
func (r *runner) isReady() bool {
|
|
r.mu.Lock()
|
|
defer r.mu.Unlock()
|
|
return r.ready
|
|
}
|
|
|
|
// start launches llama-server. It returns as soon as the process exists, not
|
|
// when the model is loaded.
|
|
func (r *runner) start() error {
|
|
r.mu.Lock()
|
|
defer r.mu.Unlock()
|
|
if r.cmd != nil {
|
|
return nil
|
|
}
|
|
cmd := exec.Command(r.bin, r.args...)
|
|
// Own process group, so stop kills anything llama-server spawned rather
|
|
// than leaving it holding VRAM after we have declared the card yielded.
|
|
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
|
|
if err := cmd.Start(); err != nil {
|
|
return err
|
|
}
|
|
r.cmd, r.ready = cmd, false
|
|
log.Printf("mavgpud: started llama-server pid=%d", cmd.Process.Pid)
|
|
go func() {
|
|
err := cmd.Wait()
|
|
r.mu.Lock()
|
|
r.cmd, r.ready = nil, false
|
|
r.mu.Unlock()
|
|
log.Printf("mavgpud: llama-server exited: %v", err)
|
|
}()
|
|
return nil
|
|
}
|
|
|
|
// stop ends llama-server and waits for the VRAM to come back. SIGTERM first so
|
|
// it unmaps cleanly, SIGKILL after the grace window. Returning before the
|
|
// process is gone would let the supervisor report a free card while 14GB is
|
|
// still mapped, which is the one lie that would make yielding useless.
|
|
func (r *runner) stop(grace time.Duration) {
|
|
r.mu.Lock()
|
|
cmd := r.cmd
|
|
r.ready = false
|
|
r.mu.Unlock()
|
|
if cmd == nil || cmd.Process == nil {
|
|
return
|
|
}
|
|
pgid := -cmd.Process.Pid
|
|
_ = syscall.Kill(pgid, syscall.SIGTERM)
|
|
deadline := time.Now().Add(grace)
|
|
for time.Now().Before(deadline) {
|
|
if !r.running() {
|
|
return
|
|
}
|
|
time.Sleep(100 * time.Millisecond)
|
|
}
|
|
log.Printf("mavgpud: llama-server did not exit in %s, killing", grace)
|
|
_ = syscall.Kill(pgid, syscall.SIGKILL)
|
|
}
|
|
|
|
// refreshReady asks llama-server whether the model is loaded. Called once per
|
|
// supervisor tick, never per request.
|
|
func (r *runner) refreshReady(ctx context.Context) {
|
|
if !r.running() {
|
|
return
|
|
}
|
|
ok := false
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, r.readyURL, nil)
|
|
if err == nil {
|
|
resp, err := r.http.Do(req)
|
|
if err == nil {
|
|
ok = resp.StatusCode == http.StatusOK
|
|
resp.Body.Close()
|
|
}
|
|
}
|
|
r.mu.Lock()
|
|
was := r.ready
|
|
r.ready = ok
|
|
r.mu.Unlock()
|
|
if ok && !was {
|
|
log.Printf("mavgpud: model ready")
|
|
}
|
|
}
|