package eval import ( "context" "fmt" "os" "testing" "time" "github.com/kami/maven/internal/llm" "github.com/kami/maven/internal/persona" "github.com/kami/maven/internal/phraser" ) // sweepTemperatures — the dial positions worth comparing (Vikunja #402). // 0.7 is what the transport has always sent; 0.05 stands in for near-greedy, // since 0 means "use the default" to the phraser. var sweepTemperatures = []float64{0.7, 0.4, 0.2, 0.05} // sweepRuns — how many runs per position. Three, because one run of a sampled // model tells you nothing about whether a two-point difference is real. const sweepRuns = 3 // TestTalkTemperatureSweep scores the talk fixture at each temperature. // // Opt-in twice over: it needs a llama-server AND it costs roughly // len(sweepTemperatures) * sweepRuns * the baseline run time, which is upwards // of twenty minutes on the CPU floor. // // MAVEN_LLM_URL=http://127.0.0.1:18099 MAVEN_TEMP_SWEEP=1 \ // go test -v -timeout 90m -run TestTalkTemperatureSweep ./internal/phraser/eval/ // // Reports, asserts nothing. The composite is not the number to read — the task // says to watch ontopic and invented content against how flat the replies get, // and the replies are logged for exactly that reason. // // Note that only the chat/query/world paths move: the reply path is a Replier // over llm.Client, which samples greedily and does not read this dial. func TestTalkTemperatureSweep(t *testing.T) { base := os.Getenv("MAVEN_LLM_URL") if base == "" { t.Skip("MAVEN_LLM_URL unset — point it at a running llama-server") } if os.Getenv("MAVEN_TEMP_SWEEP") == "" { t.Skip("MAVEN_TEMP_SWEEP unset — this sweep costs many minutes, see the doc comment") } noProxyLoopback(t) ctx := context.Background() f, err := LoadTalk() if err != nil { t.Fatalf("LoadTalk: %v", err) } model, err := llm.ModelID(ctx, base) if err != nil { t.Fatalf("no model at %s: %v", base, err) } block := func() string { return persona.Facts{}.Block(time.Now()) } summary := fmt.Sprintf("temperature sweep, %s, %d runs each\n", model, sweepRuns) for _, temp := range sweepTemperatures { for run := 1; run <= sweepRuns; run++ { cfg := phraser.DefaultConfig("") cfg.Timeout = 5 * time.Minute cfg.ContextBlock = block cfg.Temperature = temp p := phraser.NewLLMPhraserAt(base, cfg) name := fmt.Sprintf("temp %.2f run %d", temp, run) target := Pair{Talker: p, Confirmer: phraser.NewReplier(llm.New(base, cfg.Timeout), block)} rep, err := ScoreTalk(ctx, name, target, f) p.Close() if err != nil { t.Fatalf("ScoreTalk at %.2f: %v", temp, err) } if rep.Errors == rep.Total { t.Fatalf("every case errored at %.2f — nothing was measured", temp) } t.Log("\n" + rep.String() + "\nreplies:\n" + rep.Replies() + "\nfailures:\n" + rep.Failures()) summary += fmt.Sprintf(" %-18s %2d/%2d (%.1f%%) ontopic %d/%d errors %d\n", name, rep.Passed, rep.Total, 100*rep.Accuracy(), rep.ByCheck[CheckOnTopic], rep.Total, rep.Errors) } } t.Log("\n" + summary) }