Merge the thinking-off measurement
This commit is contained in:
@@ -1,11 +1,8 @@
|
||||
package eval
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
@@ -29,13 +26,20 @@ import (
|
||||
// a bake-off across checkpoints (#278, #250) produces tables you can tell
|
||||
// apart. Point the variable at one server at a time.
|
||||
//
|
||||
// Three configurations, because "the LLM router" is ambiguous and the three
|
||||
// numbers answer different questions:
|
||||
// Two configurations, because "the LLM router" is ambiguous and the two numbers
|
||||
// answer different questions:
|
||||
//
|
||||
// llm-only — the model alone. Measures the prompt + grammar contract.
|
||||
// cascade+llm — what #320 would actually ship: stage-0 grammar, then the
|
||||
// model, then the classifier as the failure floor.
|
||||
// llm-no-thinking — diagnostic only, not a shippable path (see below).
|
||||
// llm-only — the model alone. Measures the prompt + grammar contract.
|
||||
// cascade+llm — what #320 would actually ship: stage-0 grammar, then the
|
||||
// model, then the classifier as the failure floor.
|
||||
//
|
||||
// There used to be a third, "thinking off", which looked 6 points better. It is
|
||||
// gone: it was measured with a hand-rolled HTTP client that quietly dropped
|
||||
// repeat_penalty, so the gap was the missing penalty and not the thinking mode.
|
||||
// Re-measured with everything else held equal, thinking off scores exactly the
|
||||
// same, case for case — and a direct probe shows this llama-server build ignores
|
||||
// enable_thinking / reasoning_budget for this model anyway, so there was nothing
|
||||
// to turn off. Full write-up in ROUTING-EVAL-31-07-2026.md (Vikunja #376).
|
||||
func TestLLMRouterBaseline(t *testing.T) {
|
||||
base := os.Getenv("MAVEN_LLM_URL")
|
||||
if base == "" {
|
||||
@@ -96,104 +100,18 @@ func TestLLMRouterBaseline(t *testing.T) {
|
||||
}
|
||||
t.Log("\n" + repCascade.String() + repCascade.Failures())
|
||||
|
||||
// llm-no-thinking: same prompt and grammar with the chat template's
|
||||
// thinking mode off. Qwen3.5's template defaults thinking=1, so under a
|
||||
// grammar the constrained JSON lands in reasoning_content with content
|
||||
// empty — llm.Client's ReasoningContent fallback is what makes the router
|
||||
// work at all today, by accident rather than design.
|
||||
//
|
||||
// MEASURED 2026-07-31: this variant scores identically to as-deployed
|
||||
// (18/76, 48.7% intent-only, 2 errors, same p50). Thinking mode is a
|
||||
// non-issue under a grammar — llama.cpp constrains the same token stream
|
||||
// either way. Kept so the question stays answered instead of being
|
||||
// re-asked, and so internal/llm does NOT grow a chat_template_kwargs field
|
||||
// for a problem that does not exist.
|
||||
repNoThink, err := Score(ctx, "llm-only ("+model+", thinking off) [diagnostic]",
|
||||
RouterFunc(func(ctx context.Context, u string, now time.Time) (router.Decision, error) {
|
||||
d, ok, err := router.NewLLMRouter(&noThinkCompleter{base: base, http: &http.Client{Timeout: 60 * time.Second}}).Route(ctx, u, now)
|
||||
if err != nil {
|
||||
return d, err
|
||||
}
|
||||
if !ok {
|
||||
return d, fmt.Errorf("llm router declined without an error")
|
||||
}
|
||||
return d, nil
|
||||
}), f)
|
||||
if err != nil {
|
||||
t.Fatalf("Score no-thinking: %v", err)
|
||||
}
|
||||
t.Log("\n" + repNoThink.String() + repNoThink.Failures())
|
||||
|
||||
// Reports rather than asserts — the numbers are inputs to the #320
|
||||
// decision, and an assertion here would be this test inventing the bar.
|
||||
// The one thing worth failing on is a harness fault: if every single case
|
||||
// errors, the run measured infrastructure, not routing, and the report
|
||||
// must not be mistaken for a score.
|
||||
for _, rep := range []Report{repLLM, repCascade, repNoThink} {
|
||||
for _, rep := range []Report{repLLM, repCascade} {
|
||||
if rep.Errors == rep.Total {
|
||||
t.Errorf("%s: all %d cases errored — harness fault, not a measurement", rep.Name, rep.Total)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// noThinkCompleter — llm.Client with chat_template_kwargs.enable_thinking
|
||||
// false. A test-local copy rather than a change to internal/llm: whether the
|
||||
// daemon should send it is the open question, and answering it here by adding
|
||||
// the field would prejudge #320.
|
||||
type noThinkCompleter struct {
|
||||
base string
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
func (c *noThinkCompleter) Complete(ctx context.Context, r llm.Req) (string, error) {
|
||||
payload := map[string]any{
|
||||
"messages": []map[string]string{
|
||||
{"role": "system", "content": r.System},
|
||||
{"role": "user", "content": r.User},
|
||||
},
|
||||
"max_tokens": r.MaxTokens,
|
||||
"temperature": 0,
|
||||
"grammar": r.Grammar,
|
||||
"chat_template_kwargs": map[string]any{"enable_thinking": false},
|
||||
}
|
||||
b, err := json.Marshal(payload)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
req, err := http.NewRequestWithContext(ctx, "POST", c.base+"/v1/chat/completions", bytes.NewReader(b))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
resp, err := c.http.Do(req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != 200 {
|
||||
return "", fmt.Errorf("status %d", resp.StatusCode)
|
||||
}
|
||||
var out struct {
|
||||
Choices []struct {
|
||||
Message struct {
|
||||
Content string `json:"content"`
|
||||
ReasoningContent string `json:"reasoning_content"`
|
||||
} `json:"message"`
|
||||
} `json:"choices"`
|
||||
}
|
||||
if err := json.NewDecoder(resp.Body).Decode(&out); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if len(out.Choices) == 0 {
|
||||
return "", fmt.Errorf("no choices")
|
||||
}
|
||||
m := out.Choices[0].Message
|
||||
if m.Content != "" {
|
||||
return m.Content, nil
|
||||
}
|
||||
return m.ReasoningContent, nil
|
||||
}
|
||||
|
||||
func ping(ctx context.Context, c *llm.Client) error {
|
||||
ctx, cancel := context.WithTimeout(ctx, 90*time.Second)
|
||||
defer cancel()
|
||||
|
||||
Reference in New Issue
Block a user