a788ca3915
The phrasing eval printed "llm (0.8B, ...)" no matter which gguf llama-server had loaded, so two runs of two different models came out named the same and were easy to mix up when comparing. It now asks llama-server over /v1/models, same as the router eval already did. The helper moved to internal/llm so both share it, and it now errors instead of returning a blank name when the id field is missing — an unreachable server gets labelled "unknown-model", never a plausible-looking guess. Both eval paths stay opt-in behind MAVEN_LLM_URL; no server needed for go test. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
138 lines
5.1 KiB
Go
138 lines
5.1 KiB
Go
package eval
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"os"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/kami/maven/internal/llm"
|
|
"github.com/kami/maven/internal/router"
|
|
)
|
|
|
|
// TestLLMRouterBaseline — the other half of Vikunja #319: the resident model as
|
|
// route decider, scored on the same 76 held-out cases as the classifier, so the
|
|
// #320 flip is a comparison and not a preference.
|
|
//
|
|
// Opt-in against a running llama-server:
|
|
//
|
|
// llama-server -m /mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf \
|
|
// --host 127.0.0.1 --port 18099 -c 2048 -ngl 99
|
|
// MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-models
|
|
//
|
|
// Every report name carries the model llama-server reports over /v1/models, so
|
|
// a bake-off across checkpoints (#278, #250) produces tables you can tell
|
|
// apart. Point the variable at one server at a time.
|
|
//
|
|
// Two configurations, because "the LLM router" is ambiguous and the two numbers
|
|
// answer different questions:
|
|
//
|
|
// llm-only — the model alone. Measures the prompt + grammar contract.
|
|
// cascade+llm — what #320 would actually ship: stage-0 grammar, then the
|
|
// model, then the classifier as the failure floor.
|
|
//
|
|
// There used to be a third, "thinking off", which looked 6 points better. It is
|
|
// gone: it was measured with a hand-rolled HTTP client that quietly dropped
|
|
// repeat_penalty, so the gap was the missing penalty and not the thinking mode.
|
|
// Re-measured with everything else held equal, thinking off scores exactly the
|
|
// same, case for case — and a direct probe shows this llama-server build ignores
|
|
// enable_thinking / reasoning_budget for this model anyway, so there was nothing
|
|
// to turn off. Full write-up in ROUTING-EVAL-31-07-2026.md (Vikunja #376).
|
|
func TestLLMRouterBaseline(t *testing.T) {
|
|
base := os.Getenv("MAVEN_LLM_URL")
|
|
if base == "" {
|
|
t.Skip("MAVEN_LLM_URL unset — start llama-server and point it here (see doc comment)")
|
|
}
|
|
// A local llama-server must not go through an HTTP proxy. This box has
|
|
// http_proxy pointing at a SOCKS bridge that answers 503 for the loopback
|
|
// address, which would score every case as a route error and read as "the
|
|
// model can't route" — measure the model, not the proxy.
|
|
noProxyLoopback(t)
|
|
|
|
f, err := Load()
|
|
if err != nil {
|
|
t.Fatalf("Load: %v", err)
|
|
}
|
|
client := llm.New(base, 60*time.Second)
|
|
if err := ping(context.Background(), client); err != nil {
|
|
t.Skipf("llama-server at %s unreachable: %v", base, err)
|
|
}
|
|
|
|
ctx := context.Background()
|
|
model, err := llm.ModelID(ctx, base)
|
|
if err != nil {
|
|
// Not fatal: an unlabelled score is still a score. But say so loudly,
|
|
// because an unlabelled row in a bake-off table is worthless.
|
|
t.Logf("could not read model id from %s: %v — reports will say %q", base, err, llm.UnknownModel)
|
|
model = llm.UnknownModel
|
|
}
|
|
t.Logf("scoring model %s at %s", model, base)
|
|
lr := router.NewLLMRouter(client)
|
|
|
|
// llm-only: the LLM stage in isolation. Route returns (Decision, ok, err);
|
|
// !ok without an error would be a contract violation, so it is surfaced as
|
|
// one rather than silently scored as a miss.
|
|
llmOnly := RouterFunc(func(ctx context.Context, u string, now time.Time) (router.Decision, error) {
|
|
d, ok, err := lr.Route(ctx, u, now)
|
|
if err != nil {
|
|
return d, err
|
|
}
|
|
if !ok {
|
|
return d, fmt.Errorf("llm router declined without an error")
|
|
}
|
|
return d, nil
|
|
})
|
|
repLLM, err := Score(ctx, "llm-only ("+model+", as deployed)", llmOnly, f)
|
|
if err != nil {
|
|
t.Fatalf("Score llm-only: %v", err)
|
|
}
|
|
t.Log("\n" + repLLM.String() + repLLM.Failures())
|
|
|
|
// cascade+llm: stage-0 grammar → LLM → classifier fallback, the wiring #320
|
|
// proposes. Hash embedder for the fallback so the classifier contribution is
|
|
// the deterministic floor and any lift is attributable to the model.
|
|
repCascade, err := Score(ctx, "cascade+llm ("+model+") + hash fallback",
|
|
newBaselineRouter(t, router.NewHashEmbedder(1024), lr), f)
|
|
if err != nil {
|
|
t.Fatalf("Score cascade: %v", err)
|
|
}
|
|
t.Log("\n" + repCascade.String() + repCascade.Failures())
|
|
|
|
// Reports rather than asserts — the numbers are inputs to the #320
|
|
// decision, and an assertion here would be this test inventing the bar.
|
|
// The one thing worth failing on is a harness fault: if every single case
|
|
// errors, the run measured infrastructure, not routing, and the report
|
|
// must not be mistaken for a score.
|
|
for _, rep := range []Report{repLLM, repCascade} {
|
|
if rep.Errors == rep.Total {
|
|
t.Errorf("%s: all %d cases errored — harness fault, not a measurement", rep.Name, rep.Total)
|
|
}
|
|
}
|
|
}
|
|
|
|
func ping(ctx context.Context, c *llm.Client) error {
|
|
ctx, cancel := context.WithTimeout(ctx, 90*time.Second)
|
|
defer cancel()
|
|
_, err := c.Complete(ctx, llm.Req{User: "ping", MaxTokens: 4})
|
|
return err
|
|
}
|
|
|
|
// noProxyLoopback appends the loopback host to no_proxy before any request, so
|
|
// http.ProxyFromEnvironment (which caches the environment on first use) sees it.
|
|
func noProxyLoopback(t *testing.T) {
|
|
t.Helper()
|
|
for _, key := range []string{"no_proxy", "NO_PROXY"} {
|
|
cur := os.Getenv(key)
|
|
if strings.Contains(cur, "127.0.0.1") {
|
|
continue
|
|
}
|
|
if cur == "" {
|
|
t.Setenv(key, "127.0.0.1,localhost")
|
|
continue
|
|
}
|
|
t.Setenv(key, cur+",127.0.0.1,localhost")
|
|
}
|
|
}
|