Merge commit '94eb92f' into overnight-jul31
# Conflicts: # Makefile
This commit is contained in:
@@ -16,7 +16,7 @@ PIPER_BIN := $(shell pwd)/deps/piper/piper
|
|||||||
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
||||||
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
||||||
|
|
||||||
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing
|
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||||
|
|
||||||
all: build
|
all: build
|
||||||
|
|
||||||
@@ -112,6 +112,21 @@ eval-recall:
|
|||||||
eval-phrasing:
|
eval-phrasing:
|
||||||
$(GO) test -v -count=1 -timeout 40m ./internal/phraser/eval/
|
$(GO) test -v -count=1 -timeout 40m ./internal/phraser/eval/
|
||||||
|
|
||||||
|
# eval-models — score ONE llama-server against the same fixture, for the
|
||||||
|
# resident-model bake-off (#278, #250). Start a server with the gguf you want,
|
||||||
|
# then:
|
||||||
|
#
|
||||||
|
# make eval-models MAVEN_LLM_URL=http://127.0.0.1:18100
|
||||||
|
#
|
||||||
|
# The report names carry the model llama-server reports, so runs from two
|
||||||
|
# checkpoints stay apart. Only the LLM test runs — the classifier baselines do
|
||||||
|
# not depend on the model and take the ONNX runtime with them.
|
||||||
|
MAVEN_LLM_URL ?= http://127.0.0.1:18099
|
||||||
|
|
||||||
|
eval-models:
|
||||||
|
MAVEN_LLM_URL="$(MAVEN_LLM_URL)" $(GO) test -v -count=1 -timeout 60m \
|
||||||
|
-run TestLLMRouterBaseline ./internal/router/eval/
|
||||||
|
|
||||||
run-stt: build-stt
|
run-stt: build-stt
|
||||||
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||||
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
|
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
|
||||||
|
|||||||
@@ -23,7 +23,11 @@ import (
|
|||||||
//
|
//
|
||||||
// llama-server -m /mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf \
|
// llama-server -m /mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf \
|
||||||
// --host 127.0.0.1 --port 18099 -c 2048 -ngl 99
|
// --host 127.0.0.1 --port 18099 -c 2048 -ngl 99
|
||||||
// MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-router
|
// MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-models
|
||||||
|
//
|
||||||
|
// Every report name carries the model llama-server reports over /v1/models, so
|
||||||
|
// a bake-off across checkpoints (#278, #250) produces tables you can tell
|
||||||
|
// apart. Point the variable at one server at a time.
|
||||||
//
|
//
|
||||||
// Three configurations, because "the LLM router" is ambiguous and the three
|
// Three configurations, because "the LLM router" is ambiguous and the three
|
||||||
// numbers answer different questions:
|
// numbers answer different questions:
|
||||||
@@ -53,6 +57,14 @@ func TestLLMRouterBaseline(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
ctx := context.Background()
|
ctx := context.Background()
|
||||||
|
model, err := ModelID(ctx, base)
|
||||||
|
if err != nil {
|
||||||
|
// Not fatal: an unlabelled score is still a score. But say so loudly,
|
||||||
|
// because an unlabelled row in a bake-off table is worthless.
|
||||||
|
t.Logf("could not read model id from %s: %v — reports will say %q", base, err, "unknown-model")
|
||||||
|
model = "unknown-model"
|
||||||
|
}
|
||||||
|
t.Logf("scoring model %s at %s", model, base)
|
||||||
lr := router.NewLLMRouter(client)
|
lr := router.NewLLMRouter(client)
|
||||||
|
|
||||||
// llm-only: the LLM stage in isolation. Route returns (Decision, ok, err);
|
// llm-only: the LLM stage in isolation. Route returns (Decision, ok, err);
|
||||||
@@ -68,7 +80,7 @@ func TestLLMRouterBaseline(t *testing.T) {
|
|||||||
}
|
}
|
||||||
return d, nil
|
return d, nil
|
||||||
})
|
})
|
||||||
repLLM, err := Score(ctx, "llm-only (0.8B, as deployed)", llmOnly, f)
|
repLLM, err := Score(ctx, "llm-only ("+model+", as deployed)", llmOnly, f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Score llm-only: %v", err)
|
t.Fatalf("Score llm-only: %v", err)
|
||||||
}
|
}
|
||||||
@@ -77,7 +89,7 @@ func TestLLMRouterBaseline(t *testing.T) {
|
|||||||
// cascade+llm: stage-0 grammar → LLM → classifier fallback, the wiring #320
|
// cascade+llm: stage-0 grammar → LLM → classifier fallback, the wiring #320
|
||||||
// proposes. Hash embedder for the fallback so the classifier contribution is
|
// proposes. Hash embedder for the fallback so the classifier contribution is
|
||||||
// the deterministic floor and any lift is attributable to the model.
|
// the deterministic floor and any lift is attributable to the model.
|
||||||
repCascade, err := Score(ctx, "cascade+llm (0.8B) + hash fallback",
|
repCascade, err := Score(ctx, "cascade+llm ("+model+") + hash fallback",
|
||||||
newBaselineRouter(t, router.NewHashEmbedder(1024), lr), f)
|
newBaselineRouter(t, router.NewHashEmbedder(1024), lr), f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Score cascade: %v", err)
|
t.Fatalf("Score cascade: %v", err)
|
||||||
@@ -96,7 +108,7 @@ func TestLLMRouterBaseline(t *testing.T) {
|
|||||||
// either way. Kept so the question stays answered instead of being
|
// either way. Kept so the question stays answered instead of being
|
||||||
// re-asked, and so internal/llm does NOT grow a chat_template_kwargs field
|
// re-asked, and so internal/llm does NOT grow a chat_template_kwargs field
|
||||||
// for a problem that does not exist.
|
// for a problem that does not exist.
|
||||||
repNoThink, err := Score(ctx, "llm-only (0.8B, thinking off) [diagnostic]",
|
repNoThink, err := Score(ctx, "llm-only ("+model+", thinking off) [diagnostic]",
|
||||||
RouterFunc(func(ctx context.Context, u string, now time.Time) (router.Decision, error) {
|
RouterFunc(func(ctx context.Context, u string, now time.Time) (router.Decision, error) {
|
||||||
d, ok, err := router.NewLLMRouter(&noThinkCompleter{base: base, http: &http.Client{Timeout: 60 * time.Second}}).Route(ctx, u, now)
|
d, ok, err := router.NewLLMRouter(&noThinkCompleter{base: base, http: &http.Client{Timeout: 60 * time.Second}}).Route(ctx, u, now)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
package eval
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"net/http"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ModelID asks llama-server which model it has loaded, so a scoring run can
|
||||||
|
// label itself. Without this a bake-off between two models produces two tables
|
||||||
|
// that look identical, and the operator has to remember which server was up.
|
||||||
|
//
|
||||||
|
// Read from the server rather than passed in on purpose: a hand-typed label
|
||||||
|
// goes stale the moment someone restarts the server with a different -m.
|
||||||
|
func ModelID(ctx context.Context, base string) (string, error) {
|
||||||
|
req, err := http.NewRequestWithContext(ctx, "GET", strings.TrimSuffix(base, "/")+"/v1/models", nil)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
resp, err := http.DefaultClient.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != 200 {
|
||||||
|
return "", fmt.Errorf("models: status %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var out struct {
|
||||||
|
Data []struct {
|
||||||
|
ID string `json:"id"`
|
||||||
|
} `json:"data"`
|
||||||
|
}
|
||||||
|
if err := json.NewDecoder(resp.Body).Decode(&out); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
if len(out.Data) == 0 {
|
||||||
|
return "", fmt.Errorf("models: empty list")
|
||||||
|
}
|
||||||
|
return shortModelID(out.Data[0].ID), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// shortModelID trims the path and the .gguf suffix — llama-server reports the
|
||||||
|
// file name it was started with, which is too long for a table header.
|
||||||
|
func shortModelID(id string) string {
|
||||||
|
if i := strings.LastIndexAny(id, "/\\"); i >= 0 {
|
||||||
|
id = id[i+1:]
|
||||||
|
}
|
||||||
|
return strings.TrimSuffix(id, ".gguf")
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user