Label eval runs with the model the server actually loaded (#379)
The phrasing eval printed "llm (0.8B, ...)" no matter which gguf llama-server had loaded, so two runs of two different models came out named the same and were easy to mix up when comparing. It now asks llama-server over /v1/models, same as the router eval already did. The helper moved to internal/llm so both share it, and it now errors instead of returning a blank name when the id field is missing — an unreachable server gets labelled "unknown-model", never a plausible-looking guess. Both eval paths stay opt-in behind MAVEN_LLM_URL; no server needed for go test. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
This commit is contained in:
@@ -0,0 +1,70 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// UnknownModel is the label to print when the server would not say what it has
|
||||
// loaded. Deliberately ugly: an honest "unknown" is fine, a plausible-looking
|
||||
// but wrong model name is the bug this whole file exists to prevent.
|
||||
const UnknownModel = "unknown-model"
|
||||
|
||||
// llama-server is local, so never send this through a proxy: this box's
|
||||
// http_proxy answers 503 for loopback, which would look like "server won't say
|
||||
// which model it has" when the server is right there and fine.
|
||||
// A Transport with no Proxy set bypasses http_proxy entirely.
|
||||
var modelHTTP = &http.Client{Timeout: 10 * time.Second, Transport: &http.Transport{}}
|
||||
|
||||
// ModelID asks llama-server which model it has loaded, so a scoring run can
|
||||
// label itself. Without this a bake-off between two models produces two tables
|
||||
// that look identical, and the operator has to remember which server was up.
|
||||
//
|
||||
// Read from the server rather than passed in on purpose: a hand-typed label
|
||||
// goes stale the moment someone restarts the server with a different -m.
|
||||
func ModelID(ctx context.Context, base string) (string, error) {
|
||||
req, err := http.NewRequestWithContext(ctx, "GET", strings.TrimSuffix(base, "/")+"/v1/models", nil)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
resp, err := modelHTTP.Do(req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != 200 {
|
||||
return "", fmt.Errorf("models: status %d", resp.StatusCode)
|
||||
}
|
||||
var out struct {
|
||||
Data []struct {
|
||||
ID string `json:"id"`
|
||||
} `json:"data"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&out); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if len(out.Data) == 0 {
|
||||
return "", fmt.Errorf("models: empty list")
|
||||
}
|
||||
short := shortModelID(out.Data[0].ID)
|
||||
if short == "" {
|
||||
// Server answered but the id field was missing or blank. Say so
|
||||
// instead of handing back an empty label that reads as a real name.
|
||||
return "", fmt.Errorf("models: no id in response")
|
||||
}
|
||||
return short, nil
|
||||
}
|
||||
|
||||
// shortModelID trims the path and the .gguf suffix — llama-server reports the
|
||||
// file name it was started with, which is too long for a table header.
|
||||
func shortModelID(id string) string {
|
||||
id = strings.TrimSpace(id)
|
||||
if i := strings.LastIndexAny(id, "/\\"); i >= 0 {
|
||||
id = id[i+1:]
|
||||
}
|
||||
return strings.TrimSpace(strings.TrimSuffix(id, ".gguf"))
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The point of these tests: a wrong-but-plausible model label is the bug, so
|
||||
// every path that cannot learn the real name must return an error instead of a
|
||||
// guess. No llama-server needed — a stub server stands in.
|
||||
func TestModelID(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
body string
|
||||
code int
|
||||
want string // "" ⇒ expect an error
|
||||
}{
|
||||
{"full path", `{"data":[{"id":"/mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf"}]}`, 200, "Qwen3.5-0.8B.Q4_K_M"},
|
||||
{"bare name", `{"data":[{"id":"LFM2.5-1.2B"}]}`, 200, "LFM2.5-1.2B"},
|
||||
{"empty list", `{"data":[]}`, 200, ""},
|
||||
{"id missing", `{"data":[{}]}`, 200, ""},
|
||||
{"id blank", `{"data":[{"id":" "}]}`, 200, ""},
|
||||
{"server error", `nope`, 500, ""},
|
||||
{"not json", `<html>`, 200, ""},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.URL.Path != "/v1/models" {
|
||||
t.Errorf("asked for %s, want /v1/models", r.URL.Path)
|
||||
}
|
||||
w.WriteHeader(c.code)
|
||||
_, _ = w.Write([]byte(c.body))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
got, err := ModelID(context.Background(), srv.URL+"/")
|
||||
if c.want == "" {
|
||||
if err == nil {
|
||||
t.Fatalf("want an error, got label %q", got)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("ModelID: %v", err)
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("got %q, want %q", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestModelIDUnreachable(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {}))
|
||||
url := srv.URL
|
||||
srv.Close() // nothing listening now
|
||||
|
||||
if got, err := ModelID(context.Background(), url); err == nil {
|
||||
t.Fatalf("want an error from a dead server, got label %q", got)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user