eval: the resident model never reaches Praxis either (V-517)

V-405 measured reach with the classifier only, and the LLM router is the
deployed default, so 16/30 was the floor rather than the shipped behaviour.
TestReachWithLLMRouter scores the same 30 cases with the model, gated on
MAVEN_LLM_URL like TestLLMRouterBaseline.

The open question was whether the model writes a literal Praxis capability
into the fn slot and reaches a service the classifier structurally cannot. It
does not. Praxis is 0/12 with the model alone, exactly what the classifier
alone scores, and all twelve fail the same way: local, empty fn. Nothing in the
router prompt names a Praxis capability, so there is no string for it to write.

So V-516's stage-0 grammars are the only path to Praxis, not a determinism
argument. Through the cascade the model scores 28/30 with praxis 11/12, one
point above the classifier baseline. Hexis is 10/10 either way.

Overreach is 1 in both configurations, under the 4 the harness asserts.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SoL7EBdYC5Mhz3DJd49GJy
This commit is contained in:
2026-08-05 21:03:05 +04:00
parent ed9db8dc44
commit 576dfd8b4c
3 changed files with 155 additions and 1 deletions
+78
View File
@@ -112,6 +112,84 @@ func TestLLMRouterBaseline(t *testing.T) {
}
}
// TestReachWithLLMRouter — the reach fixture scored with the resident model as
// router (Vikunja #517). V-405 measured the classifier only, and the LLM router
// is the deployed default, so 16/30 with praxis 0/12 is the floor rather than
// the shipped behaviour.
//
// The question is specific. The route grammar lets the model write any string
// into the fn slot, so it *could* emit a literal Praxis capability name and
// reach a service the classifier structurally cannot. If it does, V-516's
// stage-0 Praxis grammars are a determinism argument. If it does not, they are
// the only path.
//
// Same gate and same two configurations as TestLLMRouterBaseline, for the same
// reason: "the LLM router" alone and the cascade that actually ships answer
// different questions.
func TestReachWithLLMRouter(t *testing.T) {
base := os.Getenv("MAVEN_LLM_URL")
if base == "" {
t.Skip("MAVEN_LLM_URL unset — start llama-server and point it here (see doc comment)")
}
noProxyLoopback(t)
f, err := LoadReach()
if err != nil {
t.Fatalf("LoadReach: %v", err)
}
client := llm.New(base, 60*time.Second)
if err := ping(context.Background(), client); err != nil {
t.Skipf("llama-server at %s unreachable: %v", base, err)
}
ctx := context.Background()
model, err := llm.ModelID(ctx, base)
if err != nil {
t.Logf("could not read model id from %s: %v — reports will say %q", base, err, llm.UnknownModel)
model = llm.UnknownModel
}
t.Logf("scoring reach with model %s at %s", model, base)
lr := router.NewLLMRouter(client)
m := router.DefaultActMatcher{Fns: actFns}
llmOnly := RouterFunc(func(ctx context.Context, u string, now time.Time) (router.Decision, error) {
d, ok, err := lr.Route(ctx, u, now)
if err != nil {
return d, err
}
if !ok {
return d, fmt.Errorf("llm router declined without an error")
}
return d, nil
})
repLLM, err := ScoreReach(ctx, "reach: llm-only ("+model+")", llmOnly, m, f)
if err != nil {
t.Fatalf("ScoreReach llm-only: %v", err)
}
t.Log("\n" + repLLM.String() + repLLM.Failures())
// Hash embedder for the classifier floor, so any lift is the model's and
// not the embedder's — the same control TestLLMRouterBaseline uses.
repCascade, err := ScoreReach(ctx, "reach: cascade+llm ("+model+") + hash fallback",
newBaselineRouter(t, router.NewHashEmbedder(1024), lr), m, f)
if err != nil {
t.Fatalf("ScoreReach cascade: %v", err)
}
t.Log("\n" + repCascade.String() + repCascade.Failures())
for _, rep := range []ReachReport{repLLM, repCascade} {
if rep.Errors == rep.Total {
t.Errorf("%s: all %d cases errored — harness fault, not a measurement", rep.Name, rep.Total)
}
}
// Overreach is the one direction worth failing on, for the reason
// TestReachBaselineHash gives: he never gets asked about it.
if repCascade.Overreach > 4 {
t.Errorf("%d utterances reached a service they should not have, want <= 4:\n%s",
repCascade.Overreach, repCascade.Failures())
}
}
func ping(ctx context.Context, c *llm.Client) error {
ctx, cancel := context.WithTimeout(ctx, 90*time.Second)
defer cancel()