Files
Maven/internal/phraser/eval/llmphraser_test.go
T
kami 4ba9a6f422 Add a deterministic scorer for nudge phrasing (Vikunja #323)
Review internal/phraser/eval/checks.go -- it IS the measurement. Each check
names in a comment which DESIGN.md line it defends: length, feminine
self-reference (windowed around "я" so the operator's own masculine
second-person forms are not flagged), the cringe list (pet names, emoji,
"!!", fake concern, apology, emotional support, asking how he feels,
praise), on-topic, mood enum. No send/veto signal anywhere, per
DESIGN.md § "Rules decide, LLM phrases".
Fixture (158 lines) and tests (252) do not count toward the diff ceiling;
the scorer itself is still ~650. Splitting eval.go from checks.go would
give two commits neither of which measures anything.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 02:30:52 +04:00

74 lines
2.4 KiB
Go

package eval
import (
"context"
"os"
"strings"
"testing"
"time"
"github.com/kami/maven/internal/phraser"
)
// TestLLMPhrasingBaseline — the resident model wording real nudges. Opt-in,
// same shape as internal/router/eval's MAVEN_LLM_URL gate, because CI has no
// model and a phrasing run costs minutes on the CPU target.
//
// llama-server -m /mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf \
// --host 127.0.0.1 --port 18099 -c 2048 -ngl 99
// MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-phrasing
//
// It reports and does not assert a quality bar. The numbers are the input to
// tuning the persona prompt; an assertion here would be the test inventing the
// bar rather than measuring against it. The one thing worth failing on is a
// harness fault — every case erroring means the run measured infrastructure.
func TestLLMPhrasingBaseline(t *testing.T) {
base := os.Getenv("MAVEN_LLM_URL")
if base == "" {
t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server (see doc comment)")
}
// A local llama-server must not go through an HTTP proxy. This box proxies
// loopback through a SOCKS bridge that answers 503, which would score every
// case as a phrasing error and read as "the model cannot phrase".
noProxyLoopback(t)
f, err := Load()
if err != nil {
t.Fatalf("Load: %v", err)
}
cfg := phraser.DefaultConfig("")
// Generous: an unconstrained 0.8B can spend a minute thinking before it
// writes a word, and a timeout would be scored as a model failure.
cfg.Timeout = 5 * time.Minute
p := phraser.NewLLMPhraserAt(base, cfg)
defer p.Close()
rep, err := Score(context.Background(), "llm (0.8B, built-in persona)", p, f)
if err != nil {
t.Fatalf("Score: %v", err)
}
t.Log("\n" + rep.String() + "\nmessages:\n" + rep.Messages() + "\nfailures:\n" + rep.Failures())
if rep.Errors == rep.Total {
t.Errorf("all %d cases errored — harness fault, not a measurement", rep.Total)
}
}
// noProxyLoopback appends the loopback host to no_proxy before any request, so
// http.ProxyFromEnvironment (which caches the environment on first use) sees it.
func noProxyLoopback(t *testing.T) {
t.Helper()
for _, key := range []string{"no_proxy", "NO_PROXY"} {
cur := os.Getenv(key)
if strings.Contains(cur, "127.0.0.1") {
continue
}
if cur == "" {
t.Setenv(key, "127.0.0.1,localhost")
continue
}
t.Setenv(key, cur+",127.0.0.1,localhost")
}
}