Files
Maven/internal/phraser/eval/llmphraser_test.go
T
kami a788ca3915 Label eval runs with the model the server actually loaded (#379)
The phrasing eval printed "llm (0.8B, ...)" no matter which gguf
llama-server had loaded, so two runs of two different models came out
named the same and were easy to mix up when comparing.

It now asks llama-server over /v1/models, same as the router eval
already did. The helper moved to internal/llm so both share it, and it
now errors instead of returning a blank name when the id field is
missing — an unreachable server gets labelled "unknown-model", never a
plausible-looking guess.

Both eval paths stay opt-in behind MAVEN_LLM_URL; no server needed for
go test.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 14:25:28 +04:00

88 lines
3.0 KiB
Go

package eval
import (
"context"
"os"
"strings"
"testing"
"time"
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/phraser"
)
// TestLLMPhrasingBaseline — the resident model wording real nudges. Opt-in,
// same shape as internal/router/eval's MAVEN_LLM_URL gate, because CI has no
// model and a phrasing run costs minutes on the CPU target.
//
// llama-server -m /mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf \
// --host 127.0.0.1 --port 18099 -c 2048 -ngl 99
// MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-phrasing
//
// It reports and does not assert a quality bar. The numbers are the input to
// tuning the persona prompt; an assertion here would be the test inventing the
// bar rather than measuring against it. The one thing worth failing on is a
// harness fault — every case erroring means the run measured infrastructure.
func TestLLMPhrasingBaseline(t *testing.T) {
base := os.Getenv("MAVEN_LLM_URL")
if base == "" {
t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server (see doc comment)")
}
// A local llama-server must not go through an HTTP proxy. This box proxies
// loopback through a SOCKS bridge that answers 503, which would score every
// case as a phrasing error and read as "the model cannot phrase".
noProxyLoopback(t)
ctx := context.Background()
f, err := Load()
if err != nil {
t.Fatalf("Load: %v", err)
}
cfg := phraser.DefaultConfig("")
// Generous: an unconstrained 0.8B can spend a minute thinking before it
// writes a word, and a timeout would be scored as a model failure.
cfg.Timeout = 5 * time.Minute
p := phraser.NewLLMPhraserAt(base, cfg)
defer p.Close()
// Label the run with whatever gguf the server actually has loaded. It used
// to say "0.8B" no matter what, so two runs of two different models came
// out named the same and were easy to mix up when comparing.
model, err := llm.ModelID(ctx, base)
if err != nil {
// An unlabelled score is still a score, but say so loudly — a made-up
// name in a bake-off table is worse than no name.
t.Logf("could not read model id from %s: %v — report will say %q", base, err, llm.UnknownModel)
model = llm.UnknownModel
}
t.Logf("scoring model %s at %s", model, base)
rep, err := Score(ctx, "llm ("+model+", built-in persona)", p, f)
if err != nil {
t.Fatalf("Score: %v", err)
}
t.Log("\n" + rep.String() + "\nmessages:\n" + rep.Messages() + "\nfailures:\n" + rep.Failures())
if rep.Errors == rep.Total {
t.Errorf("all %d cases errored — harness fault, not a measurement", rep.Total)
}
}
// noProxyLoopback appends the loopback host to no_proxy before any request, so
// http.ProxyFromEnvironment (which caches the environment on first use) sees it.
func noProxyLoopback(t *testing.T) {
t.Helper()
for _, key := range []string{"no_proxy", "NO_PROXY"} {
cur := os.Getenv(key)
if strings.Contains(cur, "127.0.0.1") {
continue
}
if cur == "" {
t.Setenv(key, "127.0.0.1,localhost")
continue
}
t.Setenv(key, cur+",127.0.0.1,localhost")
}
}