Label eval runs with the model the server actually loaded (#379)

The phrasing eval printed "llm (0.8B, ...)" no matter which gguf
llama-server had loaded, so two runs of two different models came out
named the same and were easy to mix up when comparing.

It now asks llama-server over /v1/models, same as the router eval
already did. The helper moved to internal/llm so both share it, and it
now errors instead of returning a blank name when the id field is
missing — an unreachable server gets labelled "unknown-model", never a
plausible-looking guess.

Both eval paths stay opt-in behind MAVEN_LLM_URL; no server needed for
go test.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
This commit is contained in:
kami
2026-07-31 14:25:28 +04:00
parent 3dbf67f8f9
commit a788ca3915
4 changed files with 105 additions and 8 deletions
+3 -3
View File
@@ -61,12 +61,12 @@ func TestLLMRouterBaseline(t *testing.T) {
}
ctx := context.Background()
model, err := ModelID(ctx, base)
model, err := llm.ModelID(ctx, base)
if err != nil {
// Not fatal: an unlabelled score is still a score. But say so loudly,
// because an unlabelled row in a bake-off table is worthless.
t.Logf("could not read model id from %s: %v — reports will say %q", base, err, "unknown-model")
model = "unknown-model"
t.Logf("could not read model id from %s: %v — reports will say %q", base, err, llm.UnknownModel)
model = llm.UnknownModel
}
t.Logf("scoring model %s at %s", model, base)
lr := router.NewLLMRouter(client)
-51
View File
@@ -1,51 +0,0 @@
package eval
import (
"context"
"encoding/json"
"fmt"
"net/http"
"strings"
)
// ModelID asks llama-server which model it has loaded, so a scoring run can
// label itself. Without this a bake-off between two models produces two tables
// that look identical, and the operator has to remember which server was up.
//
// Read from the server rather than passed in on purpose: a hand-typed label
// goes stale the moment someone restarts the server with a different -m.
func ModelID(ctx context.Context, base string) (string, error) {
req, err := http.NewRequestWithContext(ctx, "GET", strings.TrimSuffix(base, "/")+"/v1/models", nil)
if err != nil {
return "", err
}
resp, err := http.DefaultClient.Do(req)
if err != nil {
return "", err
}
defer resp.Body.Close()
if resp.StatusCode != 200 {
return "", fmt.Errorf("models: status %d", resp.StatusCode)
}
var out struct {
Data []struct {
ID string `json:"id"`
} `json:"data"`
}
if err := json.NewDecoder(resp.Body).Decode(&out); err != nil {
return "", err
}
if len(out.Data) == 0 {
return "", fmt.Errorf("models: empty list")
}
return shortModelID(out.Data[0].ID), nil
}
// shortModelID trims the path and the .gguf suffix — llama-server reports the
// file name it was started with, which is too long for a table header.
func shortModelID(id string) string {
if i := strings.LastIndexAny(id, "/\\"); i >= 0 {
id = id[i+1:]
}
return strings.TrimSuffix(id, ".gguf")
}