phraser: the spawn handshake joins server.go (V-397)
Verbatim move of spawnLlamaServer, llamaArgs, defaultStartupTimeout and startLlamaProc. llmphraser.go no longer imports bufio, os, os/exec or syscall.
This commit is contained in:
@@ -1,7 +1,6 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
@@ -10,11 +9,8 @@ import (
|
||||
"io"
|
||||
"log"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/delivery"
|
||||
@@ -215,129 +211,6 @@ func loadNudgeTemplates() *NudgeTemplates {
|
||||
return nt
|
||||
}
|
||||
|
||||
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
|
||||
// address it is listening on. ctx owns the process lifetime, so it must be the
|
||||
// daemon's context, not a request's.
|
||||
func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
p, err := startLlamaProc(ctx, cfg)
|
||||
if err != nil {
|
||||
cancel()
|
||||
return nil, err
|
||||
}
|
||||
p.cancel = cancel
|
||||
return p, nil
|
||||
}
|
||||
|
||||
// llamaArgs is the command line for one resident server. It is a function and
|
||||
// not an inline literal because kill-maven.sh's orphan sweep matches against
|
||||
// this exact line, and a test pins the two together.
|
||||
func llamaArgs(cfg Config) []string {
|
||||
args := []string{
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
"--port", extractPort(cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}
|
||||
if cfg.CacheRAMMiB > 0 {
|
||||
args = append(args, "--cache-ram", fmt.Sprintf("%d", cfg.CacheRAMMiB))
|
||||
}
|
||||
return args
|
||||
}
|
||||
|
||||
// defaultStartupTimeout — the wait for llama-server's listen line when Config
|
||||
// does not set one. A cold model load off disk is the slow part.
|
||||
const defaultStartupTimeout = 60 * time.Second
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
startupTimeout := cfg.StartupTimeout
|
||||
if startupTimeout <= 0 {
|
||||
startupTimeout = defaultStartupTimeout
|
||||
}
|
||||
p := &llamaProc{}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...)
|
||||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||||
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
|
||||
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
|
||||
// eating GPU/RAM); repeated dev restarts pile up orphans until the box OOMs.
|
||||
// Setpgid isolates it in its own process group so a stray Ctrl-C on the
|
||||
// terminal group doesn't half-kill it out from under us. (Linux-only, like
|
||||
// the rest of the daemon.)
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true, Pdeathsig: syscall.SIGKILL}
|
||||
p.cmd = cmd
|
||||
|
||||
// One pipe for both streams. llama.cpp writes its buffer sizes, KV-cache
|
||||
// layout and offload lines to stderr and its request log to stdout, and
|
||||
// stdout used to go nowhere at all — so nothing about the model's memory was
|
||||
// diagnosable from a running box. Both ends land in mavend's log now.
|
||||
pr, pw, err := os.Pipe()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("llm: output pipe: %w", err)
|
||||
}
|
||||
cmd.Stdout = pw
|
||||
cmd.Stderr = pw
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
pr.Close()
|
||||
pw.Close()
|
||||
return nil, fmt.Errorf("llm: start: %w", err)
|
||||
}
|
||||
// The child holds the only other reference to the write end. Dropping ours
|
||||
// is what makes the reader see EOF when the child dies.
|
||||
pw.Close()
|
||||
|
||||
portCh := make(chan string, 1)
|
||||
errCh := make(chan error, 1)
|
||||
tail := &lineTail{}
|
||||
p.wg.Add(1)
|
||||
go func() {
|
||||
defer p.wg.Done()
|
||||
defer pr.Close()
|
||||
sc := bufio.NewScanner(pr)
|
||||
// llama.cpp prints one prompt per line and a prompt can be long.
|
||||
sc.Buffer(make([]byte, 0, 64*1024), 1024*1024)
|
||||
listening := false
|
||||
for sc.Scan() {
|
||||
line := sc.Bytes()
|
||||
log.Printf("llama: %s", line)
|
||||
if !listening {
|
||||
tail.add(string(line))
|
||||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||||
listening = true
|
||||
portCh <- string(m[1])
|
||||
close(portCh)
|
||||
}
|
||||
}
|
||||
}
|
||||
err := sc.Err()
|
||||
if err == nil {
|
||||
err = io.EOF
|
||||
}
|
||||
errCh <- err
|
||||
}()
|
||||
|
||||
fail := func(err error) (*llamaProc, error) {
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, err
|
||||
}
|
||||
select {
|
||||
case addr := <-portCh:
|
||||
p.base = addr
|
||||
return p, nil
|
||||
case err := <-errCh:
|
||||
// The tail is the whole diagnosis when the server dies during load: bare
|
||||
// "EOF" never said which layer or which allocation it choked on.
|
||||
return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String()))
|
||||
case <-ctx.Done():
|
||||
return fail(ctx.Err())
|
||||
case <-time.After(startupTimeout):
|
||||
return fail(fmt.Errorf("llm: server did not start within %s; last output: %s", startupTimeout, tail.String()))
|
||||
}
|
||||
}
|
||||
|
||||
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
||||
// the model is swapped, so callers that cache it must register an observer
|
||||
// (OnSwap) rather than keeping the string forever.
|
||||
|
||||
@@ -6,11 +6,18 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
var listenRE = regexp.MustCompile(`listening on (https?://\S+)`)
|
||||
@@ -85,3 +92,126 @@ func extractPort(listen string) string {
|
||||
}
|
||||
return port
|
||||
}
|
||||
|
||||
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
|
||||
// address it is listening on. ctx owns the process lifetime, so it must be the
|
||||
// daemon's context, not a request's.
|
||||
func spawnLlamaServer(ctx context.Context, cfg Config) (backend, error) {
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
p, err := startLlamaProc(ctx, cfg)
|
||||
if err != nil {
|
||||
cancel()
|
||||
return nil, err
|
||||
}
|
||||
p.cancel = cancel
|
||||
return p, nil
|
||||
}
|
||||
|
||||
// llamaArgs is the command line for one resident server. It is a function and
|
||||
// not an inline literal because kill-maven.sh's orphan sweep matches against
|
||||
// this exact line, and a test pins the two together.
|
||||
func llamaArgs(cfg Config) []string {
|
||||
args := []string{
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
"--port", extractPort(cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}
|
||||
if cfg.CacheRAMMiB > 0 {
|
||||
args = append(args, "--cache-ram", fmt.Sprintf("%d", cfg.CacheRAMMiB))
|
||||
}
|
||||
return args
|
||||
}
|
||||
|
||||
// defaultStartupTimeout — the wait for llama-server's listen line when Config
|
||||
// does not set one. A cold model load off disk is the slow part.
|
||||
const defaultStartupTimeout = 60 * time.Second
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
startupTimeout := cfg.StartupTimeout
|
||||
if startupTimeout <= 0 {
|
||||
startupTimeout = defaultStartupTimeout
|
||||
}
|
||||
p := &llamaProc{}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...)
|
||||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||||
// ANY means, including SIGKILL/OOM/panic where our Close() never runs. Without
|
||||
// it a hard-killed mavend orphans its llama-server (reparented to init, keeps
|
||||
// eating GPU/RAM); repeated dev restarts pile up orphans until the box OOMs.
|
||||
// Setpgid isolates it in its own process group so a stray Ctrl-C on the
|
||||
// terminal group doesn't half-kill it out from under us. (Linux-only, like
|
||||
// the rest of the daemon.)
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true, Pdeathsig: syscall.SIGKILL}
|
||||
p.cmd = cmd
|
||||
|
||||
// One pipe for both streams. llama.cpp writes its buffer sizes, KV-cache
|
||||
// layout and offload lines to stderr and its request log to stdout, and
|
||||
// stdout used to go nowhere at all — so nothing about the model's memory was
|
||||
// diagnosable from a running box. Both ends land in mavend's log now.
|
||||
pr, pw, err := os.Pipe()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("llm: output pipe: %w", err)
|
||||
}
|
||||
cmd.Stdout = pw
|
||||
cmd.Stderr = pw
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
pr.Close()
|
||||
pw.Close()
|
||||
return nil, fmt.Errorf("llm: start: %w", err)
|
||||
}
|
||||
// The child holds the only other reference to the write end. Dropping ours
|
||||
// is what makes the reader see EOF when the child dies.
|
||||
pw.Close()
|
||||
|
||||
portCh := make(chan string, 1)
|
||||
errCh := make(chan error, 1)
|
||||
tail := &lineTail{}
|
||||
p.wg.Add(1)
|
||||
go func() {
|
||||
defer p.wg.Done()
|
||||
defer pr.Close()
|
||||
sc := bufio.NewScanner(pr)
|
||||
// llama.cpp prints one prompt per line and a prompt can be long.
|
||||
sc.Buffer(make([]byte, 0, 64*1024), 1024*1024)
|
||||
listening := false
|
||||
for sc.Scan() {
|
||||
line := sc.Bytes()
|
||||
log.Printf("llama: %s", line)
|
||||
if !listening {
|
||||
tail.add(string(line))
|
||||
if m := listenRE.FindSubmatch(line); len(m) > 1 {
|
||||
listening = true
|
||||
portCh <- string(m[1])
|
||||
close(portCh)
|
||||
}
|
||||
}
|
||||
}
|
||||
err := sc.Err()
|
||||
if err == nil {
|
||||
err = io.EOF
|
||||
}
|
||||
errCh <- err
|
||||
}()
|
||||
|
||||
fail := func(err error) (*llamaProc, error) {
|
||||
_ = cmd.Process.Kill()
|
||||
_ = cmd.Wait()
|
||||
return nil, err
|
||||
}
|
||||
select {
|
||||
case addr := <-portCh:
|
||||
p.base = addr
|
||||
return p, nil
|
||||
case err := <-errCh:
|
||||
// The tail is the whole diagnosis when the server dies during load: bare
|
||||
// "EOF" never said which layer or which allocation it choked on.
|
||||
return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String()))
|
||||
case <-ctx.Done():
|
||||
return fail(ctx.Err())
|
||||
case <-time.After(startupTimeout):
|
||||
return fail(fmt.Errorf("llm: server did not start within %s; last output: %s", startupTimeout, tail.String()))
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user