phraser: forward llama-server's output to mavend's log (V-499)
mavend scraped the child's stderr for the listen line and threw every other line away, and never piped its stdout at all. Nothing about the resident model's memory was diagnosable from a running box: no buffer sizes, no KV-cache layout, no offload lines, no prompt-cache limit. Both streams now share one pipe and every line lands in mavend's log with a `llama:` prefix. The last 12 startup lines are also kept and go into the error when the server dies before it listens, because bare "EOF" never named which allocation it choked on. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -1,9 +1,11 @@
|
||||
package phraser
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
@@ -59,6 +61,19 @@ func TestExtractPort(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The prompt cache is what ate 6.8GB of the deployed server's RSS, so the cap
|
||||
// has to reach the command line, and the opt-out has to leave it off.
|
||||
func TestLlamaArgsCapsPromptCache(t *testing.T) {
|
||||
cfg := DefaultConfig("/m.gguf")
|
||||
if got := strings.Join(llamaArgs(cfg), " "); !strings.Contains(got, "--cache-ram 512") {
|
||||
t.Errorf("default args = %q, want --cache-ram 512", got)
|
||||
}
|
||||
cfg.CacheRAMMiB = 0
|
||||
if got := strings.Join(llamaArgs(cfg), " "); strings.Contains(got, "--cache-ram") {
|
||||
t.Errorf("args with the cap off = %q, want no --cache-ram flag", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartLlamaProcScrapesPortAndReaps(t *testing.T) {
|
||||
bin := fakeLlama(t, listensThenSleeps)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
@@ -86,6 +101,48 @@ func TestStartLlamaProcScrapesPortAndReaps(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// captureLog redirects the standard logger for the duration of a test and
|
||||
// returns what was written to it.
|
||||
func captureLog(t *testing.T) *bytes.Buffer {
|
||||
t.Helper()
|
||||
var buf bytes.Buffer
|
||||
old := log.Writer()
|
||||
flags := log.Flags()
|
||||
log.SetOutput(&buf)
|
||||
log.SetFlags(0)
|
||||
t.Cleanup(func() { log.SetOutput(old); log.SetFlags(flags) })
|
||||
return &buf
|
||||
}
|
||||
|
||||
// The child's buffer-size, KV-cache and offload lines are the only way to
|
||||
// account for its memory on a running box, and they used to be dropped: stderr
|
||||
// was scraped for the listen line and thrown away, stdout was never piped.
|
||||
func TestStartLlamaProcForwardsChildOutput(t *testing.T) {
|
||||
buf := captureLog(t)
|
||||
bin := fakeLlama(t, `echo "load_tensors: Vulkan0 model buffer size = 1053.34 MiB" >&2
|
||||
echo "llama_context: KV self size = 448.00 MiB"
|
||||
`+listensThenSleeps)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
p, err := startLlamaProc(ctx, testCfg(bin))
|
||||
if err != nil {
|
||||
t.Fatalf("startLlamaProc: %v", err)
|
||||
}
|
||||
p.cancel = cancel
|
||||
defer p.Close()
|
||||
|
||||
got := buf.String()
|
||||
for _, want := range []string{
|
||||
"llama: load_tensors: Vulkan0 model buffer size = 1053.34 MiB", // stderr
|
||||
"llama: llama_context: KV self size = 448.00 MiB", // stdout, previously discarded
|
||||
} {
|
||||
if !strings.Contains(got, want) {
|
||||
t.Errorf("log missing %q\nlog was:\n%s", want, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStartLlamaProcFailureArms(t *testing.T) {
|
||||
t.Run("binary missing", func(t *testing.T) {
|
||||
cfg := testCfg(filepath.Join(t.TempDir(), "does-not-exist"))
|
||||
@@ -96,13 +153,18 @@ func TestStartLlamaProcFailureArms(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("server exits without listening", func(t *testing.T) {
|
||||
// stderr closes, so the reader goroutine reports EOF on errCh.
|
||||
// stderr closes, so the reader goroutine reports EOF on errCh. The error
|
||||
// must carry the child's last words: bare "EOF" named no cause.
|
||||
captureLog(t)
|
||||
bin := fakeLlama(t, `echo "ggml_vulkan: no device" >&2
|
||||
exit 1`)
|
||||
_, err := startLlamaProc(context.Background(), testCfg(bin))
|
||||
if err == nil || !strings.Contains(err.Error(), "llm: server output") {
|
||||
t.Fatalf("err = %v, want the server-output arm", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "ggml_vulkan: no device") {
|
||||
t.Errorf("err = %v, want the child's last output in it", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("context cancelled during startup", func(t *testing.T) {
|
||||
@@ -241,15 +303,7 @@ func TestKillMavenScriptMatchesRealCommandLine(t *testing.T) {
|
||||
// startLlamaProc that breaks the sweep fails here instead of on the box.
|
||||
cfg := DefaultConfig("/opt/maven/models/llm/Qwen3-1.7B-UD-Q4_K_XL.gguf")
|
||||
cfg.NCtx, cfg.NGpuLayers = 4096, 99
|
||||
cmdline := strings.Join([]string{
|
||||
cfg.BinPath,
|
||||
"-m", cfg.ModelPath,
|
||||
"--host", "127.0.0.1",
|
||||
"--port", extractPort(cfg.Listen),
|
||||
"-c", fmt.Sprintf("%d", cfg.NCtx),
|
||||
"-ngl", fmt.Sprintf("%d", cfg.NGpuLayers),
|
||||
"--no-webui",
|
||||
}, " ")
|
||||
cmdline := cfg.BinPath + " " + strings.Join(llamaArgs(cfg), " ")
|
||||
if !pat.MatchString(cmdline) {
|
||||
t.Fatalf("kill-maven.sh pattern %q does not match %q — orphans would leak", m[1], cmdline)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user