router/semantic: slice 22 residual non-action router — emit step, Go harness (legacy + heads modes), Python experiment
This commit is contained in:
@@ -0,0 +1,64 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// headsMain runs the deployed cascade minus the resident LLM: stage-0
|
||||
// grammars → routing heads (fine-tuned e5 copy + softmax, router_heads.onnx,
|
||||
// 0.6 decline threshold) → ONNX-embedder nearest-centroid classifier →
|
||||
// 0.55 confidence gate. This is what a production turn takes when the model
|
||||
// server is out (docs/routing.md: pickLLMRouter degrades to the classifier).
|
||||
//
|
||||
// The classifier is seeded from models/seeds like the daemon's seedClassifier,
|
||||
// embedded with the real multilingual-e5-small model rather than the block
|
||||
// hash, so this is the closest headless reproduction of the authoritative
|
||||
// router output the slice-22 report can run.
|
||||
//
|
||||
// Requires the ONNX model files and a libonnxruntime.so. Pass the library via
|
||||
// the MAVEN_ONNX_LIB environment variable, exactly as the daemon does.
|
||||
func headsMain(poolPath, outPath string) {
|
||||
lib := os.Getenv("MAVEN_ONNX_LIB")
|
||||
if lib == "" {
|
||||
fmt.Fprintln(os.Stderr, "heads mode needs MAVEN_ONNX_LIB pointing at libonnxruntime.so")
|
||||
os.Exit(2)
|
||||
}
|
||||
const (
|
||||
embedModel = "models/embedder/multilingual-e5-small/model_quantized.onnx"
|
||||
tokPath = "models/embedder/multilingual-e5-small/tokenizer.json"
|
||||
headsModel = "models/embedder/router-heads/router_heads.onnx"
|
||||
)
|
||||
emb, err := router.NewONNXEmbedder(embedModel, tokPath, lib)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "heads: embedder: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
defer emb.Close()
|
||||
|
||||
cls := router.NewClassifier(emb)
|
||||
seedClassifier(cls)
|
||||
|
||||
heads, err := router.NewRouterHeads(headsModel, tokPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "heads: %v\n", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
defer heads.Close()
|
||||
|
||||
acts := router.DefaultActMatcher{Fns: actVerbList()}
|
||||
r := router.New(router.Config{
|
||||
Grammars: router.StageZeroGrammars(acts),
|
||||
Classifier: cls,
|
||||
Extractor: router.Extractor{
|
||||
Time: router.StubDateTimeParser{},
|
||||
Acts: acts,
|
||||
Facts: router.DefaultFactParser{},
|
||||
},
|
||||
Threshold: 0.55,
|
||||
Heads: heads,
|
||||
})
|
||||
runOverPool(r, poolPath, outPath)
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// Seed loading replicated from cmd/mavend/voicewire.go (seedClassifier) and
|
||||
// internal/router/semantic/helpers_test.go, which build the same classifier
|
||||
// from models/seeds/<intent>.txt. The daemon and the eval fixture must agree
|
||||
// on the seeds; so must a measurement.
|
||||
const seedDir = "models/seeds"
|
||||
|
||||
var seedIntents = []router.Intent{
|
||||
router.IntentAct, router.IntentReminder, router.IntentFact,
|
||||
router.IntentNote, router.IntentQuery, router.IntentChat, router.IntentSystem,
|
||||
}
|
||||
|
||||
// actVerbList mirrors the eval fixture's static allowlist
|
||||
// (internal/router/semantic/helpers_test.go). The production matcher's
|
||||
// allowlist is the deployment's enabled tools; an act grammar can only catch
|
||||
// a row whose first tokens match an allowlisted verb, and the experiment's
|
||||
// act allowlist is the one the accepted slice-21 methodology used.
|
||||
func actVerbList() []string {
|
||||
return []string{
|
||||
"перезапусти", "перезагрузи", "выключи", "включи", "останови", "запусти",
|
||||
"закрой", "открой", "сделай", "поставь",
|
||||
"restart", "reboot", "stop", "start", "turn off", "turn on", "open", "close",
|
||||
}
|
||||
}
|
||||
|
||||
// buildMinimalRouter reproduces internal/router/semantic/buildMinimalRouter:
|
||||
// the daemon's grammar set, a hash-embedder classifier seeded from
|
||||
// models/seeds, and the deployed 0.55 threshold. Deterministic and
|
||||
// reproducible. The ONNX embedder and the routing heads score elsewhere;
|
||||
// this is the floor the eval fixture reports as the legacy baseline.
|
||||
func buildMinimalRouter() *router.Router {
|
||||
acts := router.DefaultActMatcher{Fns: actVerbList()}
|
||||
cls := router.NewClassifier(router.NewHashEmbedder(1024))
|
||||
seedClassifier(cls)
|
||||
return router.New(router.Config{
|
||||
Grammars: router.StageZeroGrammars(acts),
|
||||
Classifier: cls,
|
||||
Extractor: router.Extractor{
|
||||
Time: router.StubDateTimeParser{},
|
||||
Acts: acts,
|
||||
Facts: router.DefaultFactParser{},
|
||||
},
|
||||
Threshold: 0.55,
|
||||
})
|
||||
}
|
||||
|
||||
func seedClassifier(c *router.Classifier) {
|
||||
// Walk up to find models/seeds like the daemon's seedPath, so the program
|
||||
// can run from any depth of the repo tree.
|
||||
dir := seedDir
|
||||
for i := 0; i < 5; i++ {
|
||||
if st, err := os.Stat(dir); err == nil && st.IsDir() {
|
||||
break
|
||||
}
|
||||
dir = filepath.Join("..", dir)
|
||||
}
|
||||
ctx := context.Background()
|
||||
total := 0
|
||||
for _, intent := range seedIntents {
|
||||
path := filepath.Join(dir, string(intent)+".txt")
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
log.Printf("legacy: open seed %s: %v", path, err)
|
||||
continue
|
||||
}
|
||||
sc := bufio.NewScanner(f)
|
||||
lines := []string{}
|
||||
for sc.Scan() {
|
||||
line := strings.TrimSpace(sc.Text())
|
||||
if line == "" || strings.HasPrefix(line, "#") {
|
||||
continue
|
||||
}
|
||||
lines = append(lines, line)
|
||||
}
|
||||
f.Close()
|
||||
sort.Strings(lines)
|
||||
for _, line := range lines {
|
||||
if err := c.AddExample(ctx, intent, line); err != nil {
|
||||
log.Printf("legacy: seed %s %q: %v", intent, line, err)
|
||||
continue
|
||||
}
|
||||
total++
|
||||
}
|
||||
}
|
||||
log.Printf("legacy: loaded %d seed examples from %s", total, dir)
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"flag"
|
||||
"fmt"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// Legacy-baseline runner for slice 22: run the actual router cascade (stage 0
|
||||
// grammars → hash-embedder nearest-centroid classifier → 0.55 confidence gate)
|
||||
// over the frozen residual non-action dev pool and project each decision into
|
||||
// the five-way non-action semantic space.
|
||||
//
|
||||
// Projection rules (the daemon's behaviour, not just ScoreLegacy's):
|
||||
// - route error → uncertain
|
||||
// - Clarify=true (stage 3) → uncertain: the daemon asks, it does not commit
|
||||
// to a semantic bucket
|
||||
// - chat/query/fact+note/system → conversation/knowledge/memory_write/system
|
||||
// - act/reminder on a trusted non-action row → class "action" recorded
|
||||
// VERBATIM with illegal_action_prediction=true; never mapped to uncertain
|
||||
// - anything else → uncertain
|
||||
//
|
||||
// Reads /tmp/mvn-s22/pool.json (emit step) and writes /tmp/mvn-s22/legacy.json
|
||||
// with both the raw decision fields and the projected class, plus a summary
|
||||
// printout. No embedding is recomputed and no label is changed.
|
||||
|
||||
type poolRow struct {
|
||||
IDX int `json:"idx"`
|
||||
Text string `json:"text"`
|
||||
NText string `json:"n_text"`
|
||||
Route string `json:"route"`
|
||||
Tags []string `json:"tags"`
|
||||
CVFold int `json:"cv_fold"`
|
||||
SplitGroup string `json:"split_group"`
|
||||
FamilyID string `json:"family_id"`
|
||||
SourceID string `json:"source_id"`
|
||||
}
|
||||
|
||||
type legacyRow struct {
|
||||
IDX int `json:"idx"`
|
||||
Text string `json:"text"`
|
||||
Route string `json:"route"`
|
||||
Intent string `json:"intent"`
|
||||
Class string `json:"class"`
|
||||
Illegal bool `json:"illegal_action_prediction"`
|
||||
Confidence float64 `json:"confidence"`
|
||||
Stage int `json:"stage"`
|
||||
Clarify bool `json:"clarify"`
|
||||
Producer string `json:"producer"`
|
||||
Error string `json:"error,omitempty"`
|
||||
SourceID string `json:"source_id"`
|
||||
}
|
||||
|
||||
func main() {
|
||||
var mode, poolPath, outPath string
|
||||
flag.StringVar(&mode, "mode", "legacy", "baseline mode: legacy (hash classifier) or heads (ONNX cascade minus LLM)")
|
||||
flag.StringVar(&poolPath, "pool", "/tmp/mvn-s22/pool.json", "emit-step pool.json")
|
||||
flag.StringVar(&outPath, "out", "/tmp/mvn-s22/legacy.json", "output path")
|
||||
flag.Parse()
|
||||
switch mode {
|
||||
case "legacy":
|
||||
legacyMain(poolPath, outPath)
|
||||
case "heads":
|
||||
headsMain(poolPath, outPath)
|
||||
default:
|
||||
fmt.Fprintf(os.Stderr, "unknown -mode %q\n", mode)
|
||||
os.Exit(2)
|
||||
}
|
||||
}
|
||||
|
||||
func legacyMain(poolPath, outPath string) {
|
||||
runOverPool(buildMinimalRouter(), poolPath, outPath)
|
||||
}
|
||||
|
||||
func runOverPool(r *router.Router, poolPath, outPath string) {
|
||||
raw, err := os.ReadFile(poolPath)
|
||||
if err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
var rows []poolRow
|
||||
if err := json.Unmarshal(raw, &rows); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
ctx := context.Background()
|
||||
now := time.Now()
|
||||
|
||||
out := make([]legacyRow, 0, len(rows))
|
||||
classCount := map[string]int{}
|
||||
for _, pr := range rows {
|
||||
d, err := r.Route(ctx, router.NormalizedInput{Text: pr.Text}, now)
|
||||
lr := legacyRow{
|
||||
IDX: pr.IDX,
|
||||
Text: pr.Text,
|
||||
Route: pr.Route,
|
||||
SourceID: pr.SourceID,
|
||||
}
|
||||
if err != nil {
|
||||
lr.Class = "uncertain"
|
||||
lr.Error = err.Error()
|
||||
} else {
|
||||
lr.Intent = string(d.Intent)
|
||||
lr.Confidence = d.Confidence
|
||||
lr.Stage = d.Stage
|
||||
lr.Clarify = d.Clarify
|
||||
lr.Producer = string(d.Producer)
|
||||
}
|
||||
lr.Class, lr.Illegal = project(d, err)
|
||||
classCount[lr.Class]++
|
||||
out = append(out, lr)
|
||||
}
|
||||
|
||||
if err := writeJSON(outPath, out); err != nil {
|
||||
fmt.Fprintln(os.Stderr, err)
|
||||
os.Exit(1)
|
||||
}
|
||||
|
||||
fmt.Printf("legacy baseline over %d residual non-action rows:\n", len(out))
|
||||
for _, c := range []string{"conversation", "knowledge", "memory_write", "system", "uncertain", "action"} {
|
||||
fmt.Printf(" %-14s %d (%.1f%%)\n", c, classCount[c], 100*float64(classCount[c])/float64(len(out)))
|
||||
}
|
||||
fmt.Printf(" illegal_action_prediction: %d\n", classCount["action"])
|
||||
// Grammar hits inside a corpus-residual population would be a
|
||||
// corpus/harness disagreement worth telling the report about: the corpus
|
||||
// marked each row not-fast-path-resolved, so a current stage-0 rule
|
||||
// resolving it means the corpus's fast-path mirror is stale or a grammar
|
||||
// landed after the corpus froze.
|
||||
gh := 0
|
||||
ghByRoute := map[string]int{}
|
||||
ghByIntent := map[string]int{}
|
||||
for _, lr := range out {
|
||||
if lr.Producer == string(router.RouteProducerGrammar) {
|
||||
gh++
|
||||
ghByRoute[lr.Route]++
|
||||
ghByIntent[lr.Intent]++
|
||||
}
|
||||
}
|
||||
fmt.Printf(" stage-0 grammar hits: %d\n", gh)
|
||||
if gh > 0 {
|
||||
fmt.Printf(" by ground-truth route: %v\n", ghByRoute)
|
||||
fmt.Printf(" by grammar intent: %v\n", ghByIntent)
|
||||
}
|
||||
}
|
||||
|
||||
// project maps the router's authoritative output into the five-way non-action
|
||||
// space, or to the "action" bucket verbatim when the router calls an act or a
|
||||
// reminder on a non-action row.
|
||||
func project(d router.Decision, err error) (string, bool) {
|
||||
if err != nil {
|
||||
return "uncertain", false
|
||||
}
|
||||
if d.Clarify {
|
||||
return "uncertain", false
|
||||
}
|
||||
switch d.Intent {
|
||||
case router.IntentChat:
|
||||
return "conversation", false
|
||||
case router.IntentQuery:
|
||||
return "knowledge", false
|
||||
case router.IntentFact, router.IntentNote:
|
||||
return "memory_write", false
|
||||
case router.IntentSystem:
|
||||
return "system", false
|
||||
case router.IntentAct, router.IntentReminder:
|
||||
return "action", true
|
||||
default:
|
||||
return "uncertain", false
|
||||
}
|
||||
}
|
||||
|
||||
func writeJSON(path string, v any) error {
|
||||
fh, err := os.Create(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer fh.Close()
|
||||
w := bufio.NewWriter(fh)
|
||||
enc := json.NewEncoder(w)
|
||||
enc.SetIndent("", " ")
|
||||
if err := enc.Encode(v); err != nil {
|
||||
return err
|
||||
}
|
||||
return w.Flush()
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Slice 22 emit: five-way residual non-action semantic router — data files
|
||||
========================================================================
|
||||
|
||||
Slice 21 accepted the deterministic execution-frame guard (docs/evals/
|
||||
2026-09-07-execution-frame-guard.md). Slice 22 returns to the coarse non-action
|
||||
router that the guard hands to: after TryFastPath misses and the guard passes,
|
||||
the remaining utterance is one of five non-action semantics — conversation,
|
||||
knowledge, memory_write, system, uncertain. Action rows never reach this
|
||||
router; they are usable only as out-of-domain probes, never in primary metrics.
|
||||
|
||||
This script only repackages the frozen dev pool for the Go legacy baseline and
|
||||
the Python experiment. It reuses slice 18's loader/filters and slice 19's
|
||||
normalizers verbatim, so the population here is the same one slices 18-21
|
||||
measured. It writes:
|
||||
|
||||
/tmp/mvn-s22/pool.json residual non-action dev rows: idx, text, n_text,
|
||||
route, tags, cv_fold, split_group, family_id,
|
||||
source_id (1652 rows)
|
||||
/tmp/mvn-s22/ood.json residual ACTION dev rows (766): same shape; OOD
|
||||
probes only, never primary metrics
|
||||
/tmp/mvn-s22/stats.json population summary (routes, families, folds)
|
||||
|
||||
idx is the row's position among dev_pool rows in dev-pool order, so the Python
|
||||
experiment can align the embedding vectors from /tmp/mvn-experiment/embeddings.json
|
||||
by index exactly as slice19.load_dev does.
|
||||
|
||||
No training happens here and no label is changed.
|
||||
|
||||
Population (verified 2026-09-08 from the frozen file):
|
||||
dev_pool 2490
|
||||
dev residual 2418 (= dev_pool minus fast_path_resolved)
|
||||
residual non-action 1652 knowledge 715 / memory_write 553 / system 184 /
|
||||
uncertain 107 / conversation 93
|
||||
residual action 766 (OOD probes only)
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
sys.path.insert(0, HERE)
|
||||
|
||||
import slice18_sparse # noqa: E402 (normalize_match_text, load_data, filters)
|
||||
|
||||
OUT_DIR = "/tmp/mvn-s22"
|
||||
|
||||
ROUTES = ["conversation", "knowledge", "memory_write", "system", "uncertain"]
|
||||
|
||||
|
||||
def main():
|
||||
meta, examples = slice18_sparse.load_data()
|
||||
dev = slice18_sparse.filter_dev_pool(examples)
|
||||
print(f"dev pool: {len(dev)} rows "
|
||||
f"(meta declares dev_count={meta.get('dev_count')})")
|
||||
|
||||
rows = []
|
||||
for i, e in enumerate(dev):
|
||||
if not e["fast_path_resolved"]:
|
||||
rows.append({
|
||||
"idx": i,
|
||||
"text": e["text"],
|
||||
"n_text": slice18_sparse.normalize_match_text(e["text"]),
|
||||
"route": e["route"],
|
||||
"tags": sorted(set(e.get("tags", []))),
|
||||
"cv_fold": e["cv_fold"],
|
||||
"split_group": e["split_group"],
|
||||
"family_id": e["family_id"],
|
||||
"source_id": e["source_id"],
|
||||
})
|
||||
|
||||
na = [r for r in rows if r["route"] != "action"]
|
||||
ood = [r for r in rows if r["route"] == "action"]
|
||||
print(f"residual rows: {len(rows)} non-action: {len(na)} action(OOD): {len(ood)}")
|
||||
|
||||
by_route = {}
|
||||
for r in na:
|
||||
by_route[r["route"]] = by_route.get(r["route"], 0) + 1
|
||||
print("routes:", by_route)
|
||||
assert sum(by_route.values()) == len(na)
|
||||
assert set(ROUTES) == set(by_route), "route set must be the five-way"
|
||||
|
||||
by_family = {}
|
||||
for r in na:
|
||||
by_family[r["family_id"]] = by_family.get(r["family_id"], 0) + 1
|
||||
by_fold = {}
|
||||
for r in na:
|
||||
by_fold[r["cv_fold"]] = by_fold.get(r["cv_fold"], 0) + 1
|
||||
print(f"family_ids: {len(by_family)} split_groups: {len(set(r['split_group'] for r in na))}")
|
||||
print("folds:", by_fold)
|
||||
|
||||
os.makedirs(OUT_DIR, exist_ok=True)
|
||||
with open(os.path.join(OUT_DIR, "pool.json"), "w") as f:
|
||||
json.dump(na, f, ensure_ascii=False, indent=1)
|
||||
with open(os.path.join(OUT_DIR, "ood.json"), "w") as f:
|
||||
json.dump(ood, f, ensure_ascii=False, indent=1)
|
||||
with open(os.path.join(OUT_DIR, "stats.json"), "w") as f:
|
||||
json.dump({
|
||||
"dev_count": len(dev),
|
||||
"residual_count": len(rows),
|
||||
"non_action_count": len(na),
|
||||
"action_ood_count": len(ood),
|
||||
"routes": by_route,
|
||||
"family_ids": len(by_family),
|
||||
"folds": by_fold,
|
||||
"top_family": dict(sorted(by_family.items(), key=lambda kv: -kv[1])[:15]),
|
||||
}, f, ensure_ascii=False, indent=1)
|
||||
print(f"wrote {OUT_DIR}/{{pool,ood,stats}}.json")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,560 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Slice 22: five-way residual non-action semantic router (experiment)
|
||||
==================================================================
|
||||
|
||||
After TryFastPath misses and the ExecutionFrameGuard passes, the residual
|
||||
utterance is one of five non-action semantics: conversation, knowledge,
|
||||
memory_write, system, uncertain. This measures whether the deployed e5-small
|
||||
embeddings (384-d, query-prefixed, mean-pooled, L2, frozen) fed to a linear
|
||||
softmax head suffice, and how they compare to the legacy router, to floors,
|
||||
and to the deployed routing heads.
|
||||
|
||||
Population: the frozen dev-pool residual non-action rows (1652; the pool
|
||||
written by slice22_emit.py). Action rows (766) are out-of-domain probes only.
|
||||
|
||||
Metrics written to /tmp/mvn-s22/results.json:
|
||||
§1 population
|
||||
§2 legacy baseline (legacy.json / legacy_heads.json): acc, macro-F1,
|
||||
per-class P/R/F1, confusion, illegal_action_prediction count
|
||||
§3 e5-linear primary head: C grid, grouped CV OOF, per-fold P/R/F1 +
|
||||
variance + composition
|
||||
§4 floors: majority, centroid (cosine nearest-mean), sparse word+char
|
||||
TF-IDF logistic (slice18 builder), all grouped CV
|
||||
§5 route-family (family_id) leave-family-out
|
||||
§6 knowledge vs memory_write: matched pairs (water/homelab/task) ordering
|
||||
§7 uncertain as an explicit class: P/R/F1 + top confusions
|
||||
§8 OOF confidence: max-softmax correct/wrong, ECE, log-loss, Brier,
|
||||
coverage/accuracy/macro-F1 abstention curves (no threshold chosen)
|
||||
§9 action OOD probes: fold models applied to the 766 action rows
|
||||
§10 artifact cost: head params, serialized bytes, incremental head latency
|
||||
|
||||
No corpus label is changed. No frozen-holdout rows are inspected.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
import numpy as np
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
sys.path.insert(0, HERE)
|
||||
|
||||
import slice18_sparse # noqa: E402
|
||||
import slice19_main # noqa: E402
|
||||
|
||||
EMB_PATH = "/tmp/mvn-experiment/embeddings.json"
|
||||
OUT_DIR = "/tmp/mvn-s22"
|
||||
CLASSES = ["conversation", "knowledge", "memory_write", "system", "uncertain"]
|
||||
CLASS_PREFIX = ["conversation", "knowledge", "memory_write", "system", "uncertain"]
|
||||
C_GRID = [0.1, 1.0, 10.0]
|
||||
|
||||
# Route-family holdouts the report calls out by name (slice-22 brief): every
|
||||
# family that is not part of the shared subject inventory on either side.
|
||||
HOLDOUT_GROUPS = {
|
||||
"capability": ["knowledge:capability-ha", "knowledge:capability-tool"],
|
||||
"world": ["knowledge:world-def", "knowledge:world-explain"],
|
||||
"calendar": ["knowledge:calendar", "knowledge:calendar-time", "knowledge:calendar-next"],
|
||||
"recall": ["knowledge:recall-fact", "knowledge:recall-note", "knowledge:recall-possessive"],
|
||||
"fact": ["fact:meal", "fact:water", "fact:sleep", "fact:shower", "fact:break", "fact:pills", "fact:exercise"],
|
||||
"note": ["note:idea", "note:homelab", "note:task"],
|
||||
"remember": ["free:remember"],
|
||||
"system": None, # all system:*
|
||||
"conversation": None,
|
||||
"uncertain": None,
|
||||
}
|
||||
|
||||
|
||||
def load_pool_and_embeds():
|
||||
with open(os.path.join(OUT_DIR, "pool.json")) as f:
|
||||
pool = json.load(f)
|
||||
meta, examples = slice18_sparse.load_data()
|
||||
dev = slice18_sparse.filter_dev_pool(examples)
|
||||
by_idx = {e["dev_idx"]: e for e in dev} if "dev_idx" in dev[0] else None
|
||||
# pool rows carry idx = position among dev_pool rows in dev order
|
||||
emb_by_idx = {i: np.asarray(e["embedding"], dtype=np.float64)
|
||||
for i, e in enumerate(dev)}
|
||||
for r in pool:
|
||||
r["emb"] = emb_by_idx[r["idx"]]
|
||||
r["y"] = r["route"]
|
||||
return pool, meta
|
||||
|
||||
|
||||
def oof_proba_grouped(X, y, folds, C=1.0):
|
||||
"""Grouped OOF probability matrix (n×5, class order CLASSES)."""
|
||||
y_idx = np.array([CLASSES.index(c) for c in y])
|
||||
folds = np.asarray(folds)
|
||||
proba = np.zeros((len(y_idx), len(CLASSES)))
|
||||
for te_fold in sorted(set(folds.tolist())):
|
||||
tr = folds != te_fold
|
||||
te = folds == te_fold
|
||||
clf = slice18_sparse.LogisticRegression(
|
||||
C=C, max_iter=2000, solver="lbfgs", random_state=42)
|
||||
clf.fit(X[tr], y_idx[tr])
|
||||
proba[te] = clf.predict_proba(X[te])
|
||||
return proba
|
||||
|
||||
|
||||
def cls_metrics(yt, yp):
|
||||
import sklearn.metrics as m
|
||||
yt = np.asarray(yt)
|
||||
yp = np.asarray(yp)
|
||||
if yt.dtype != np.int64 and yt.dtype != np.int32:
|
||||
yt = np.array([CLASSES.index(c) for c in yt])
|
||||
if yp.dtype != np.int64 and yp.dtype != np.int32:
|
||||
yp = np.array([CLASSES.index(c) for c in yp])
|
||||
labels = list(range(len(CLASSES)))
|
||||
n = len(yt)
|
||||
acc = m.accuracy_score(yt, yp)
|
||||
macro = m.f1_score(yt, yp, average="macro", labels=labels, zero_division=0)
|
||||
pr, rc, f1, sup = m.precision_recall_fscore_support(
|
||||
yt, yp, labels=labels, zero_division=0)
|
||||
per = {c: {"p": float(pr[i]), "r": float(rc[i]), "f1": float(f1[i]), "n": int(sup[i])}
|
||||
for i, c in enumerate(CLASSES)}
|
||||
conf = m.confusion_matrix(yt, yp, labels=labels).tolist()
|
||||
return {"n": n, "acc": acc, "macro_f1": macro, "per_class": per, "confusion": conf}
|
||||
|
||||
|
||||
def fold_report(yt, proba, folds, true_y):
|
||||
out = {}
|
||||
folds_arr = np.asarray(folds)
|
||||
comp = {}
|
||||
for f in sorted(set(folds_arr.tolist())):
|
||||
mask = folds_arr == f
|
||||
yt_f = [CLASSES.index(y) for y in true_y[mask]]
|
||||
comp[f] = {c: int((np.array(true_y[mask]) == c).sum()) for c in CLASSES}
|
||||
per_fold = {}
|
||||
for f in sorted(set(folds_arr.tolist())):
|
||||
mask = folds_arr == f
|
||||
yp = proba[mask].argmax(1).tolist()
|
||||
m = cls_metrics([yt[i] for i in np.where(mask)[0].tolist()], yp)
|
||||
per_fold[f] = {"acc": m["acc"], "macro_f1": m["macro_f1"]}
|
||||
out["composition"] = comp
|
||||
out["per_fold"] = per_fold
|
||||
accs = [v["acc"] for v in per_fold.values()]
|
||||
macros = [v["macro_f1"] for v in per_fold.values()]
|
||||
out["acc_mean"] = float(np.mean(accs))
|
||||
out["acc_std"] = float(np.std(accs))
|
||||
out["macro_f1_mean"] = float(np.mean(macros))
|
||||
out["macro_f1_std"] = float(np.std(macros))
|
||||
return out
|
||||
|
||||
|
||||
def ece(yt, proba, n_bins=15):
|
||||
conf = proba.max(1)
|
||||
pred = proba.argmax(1)
|
||||
acc = (pred == yt).astype(float)
|
||||
bins = np.linspace(0, 1, n_bins + 1)
|
||||
tot = 0.0
|
||||
details = []
|
||||
counts = 0
|
||||
for i in range(n_bins):
|
||||
lo, hi = bins[i], bins[i + 1]
|
||||
m = (conf >= lo) & (conf < hi) if i < n_bins - 1 else conf >= lo
|
||||
if m.sum() == 0:
|
||||
continue
|
||||
acc_m = acc[m].mean()
|
||||
conf_m = conf[m].mean()
|
||||
w = m.sum() / len(conf)
|
||||
tot += w * abs(acc_m - conf_m)
|
||||
counts += int(m.sum())
|
||||
details.append({"bin": i, "lo": lo, "hi": hi, "conf": float(conf_m),
|
||||
"acc": float(acc_m), "n": int(m.sum())})
|
||||
return {"ece": float(tot), "n_bins": n_bins, "counted": counts, "bins": details}
|
||||
|
||||
|
||||
def main():
|
||||
pool, meta = load_pool_and_embeds()
|
||||
pool.sort(key=lambda r: r["idx"])
|
||||
print(f"pool: {len(pool)} rows")
|
||||
|
||||
from sklearn.metrics import brier_score_loss, log_loss
|
||||
|
||||
report = {"population": {}, "legacy": {}, "e5_linear": {}, "floors": {},
|
||||
"family_holdouts": {}, "kmw": {}, "uncertain": {}, "confidence": {},
|
||||
"ood": {}, "artifact": {}}
|
||||
|
||||
# ── §1 population ──────────────────────────────────────────────────────
|
||||
cnt = {}
|
||||
for r in pool:
|
||||
cnt[r["y"]] = cnt.get(r["y"], 0) + 1
|
||||
report["population"] = {
|
||||
"n": len(pool),
|
||||
"routes": cnt,
|
||||
"family_ids": len(set(r["family_id"] for r in pool)),
|
||||
"split_groups": len(set(r["split_group"] for r in pool)),
|
||||
"folds": {str(f): int(sum(1 for r in pool if r["cv_fold"] == f)) for f in sorted(set(r["cv_fold"] for r in pool))},
|
||||
"corpus": {k: v for k, v in meta.items() if k in
|
||||
("dev_count", "residual_count", "fast_path_count",
|
||||
"dimension", "embedder_id", "input_template", "pooling", "normalization")},
|
||||
}
|
||||
print("\n§1 population:", report["population"])
|
||||
|
||||
X = np.vstack([r["emb"] for r in pool])
|
||||
y = np.array([r["y"] for r in pool])
|
||||
folds = np.array([r["cv_fold"] for r in pool])
|
||||
yt = np.array([CLASSES.index(c) for c in y])
|
||||
|
||||
# ── §2 legacy baselines ────────────────────────────────────────────────
|
||||
import collections
|
||||
for tag, fname in [("hash", "legacy.json"), ("heads", "legacy_heads.json")]:
|
||||
path = os.path.join(OUT_DIR, fname)
|
||||
if not os.path.exists(path):
|
||||
continue
|
||||
leg = json.load(open(path))
|
||||
leg_by_idx = {r["idx"]: r for r in leg}
|
||||
yp_leg = []
|
||||
illegal = []
|
||||
for r in pool:
|
||||
lr = leg_by_idx[r["idx"]]
|
||||
if lr["illegal_action_prediction"]:
|
||||
illegal.append(lr)
|
||||
yp_leg.append("action")
|
||||
else:
|
||||
yp_leg.append(lr["class"])
|
||||
yp_leg = np.array(yp_leg)
|
||||
# five-way: an 'action' prediction is an error (outside the label set)
|
||||
yp5 = np.array([("uncertain" if p == "action" else p) for p in yp_leg])
|
||||
m = cls_metrics(y, yp5)
|
||||
m["illegal_action_prediction"] = len(illegal)
|
||||
m["illegal_cases"] = [{"idx": i["idx"], "text": i["text"], "route": i["route"],
|
||||
"intent": i["intent"], "producer": i["producer"],
|
||||
"confidence": i["confidence"]} for i in illegal]
|
||||
# per-cell confusion also shows 'action' column
|
||||
conf_counts = collections.Counter(zip(y, yp_leg))
|
||||
m["confusion_with_action"] = {f"{a}->{b}": int(c) for (a, b), c in conf_counts.items()}
|
||||
report["legacy"][tag] = m
|
||||
print(f"\n§2 legacy ({tag}) acc={m['acc']:.4f} macroF1={m['macro_f1']:.4f} "
|
||||
f"illegal={len(illegal)}")
|
||||
for c in CLASSES:
|
||||
p = m["per_class"][c]
|
||||
print(f" {c:<14} P={p['p']:.3f} R={p['r']:.3f} F1={p['f1']:.3f} n={p['n']}")
|
||||
|
||||
# grammar-pure residual: rows not resolved by any current stage-0 grammar
|
||||
if os.path.exists(os.path.join(OUT_DIR, "legacy.json")):
|
||||
leg = json.load(open(os.path.join(OUT_DIR, "legacy.json")))
|
||||
gh = {r["idx"] for r in leg if r["producer"] == "grammar"}
|
||||
gp_mask = np.array([r["idx"] not in gh for r in pool])
|
||||
report["grammar_drift"] = {
|
||||
"grammar_hits_in_pool": len(gh),
|
||||
"grammar_pure_n": int(gp_mask.sum()),
|
||||
}
|
||||
|
||||
# ── §3 e5-linear primary head ─────────────────────────────────────────
|
||||
print("\n§3 e5-linear")
|
||||
bestC, bestMac = 1.0, -1.0
|
||||
grid = {}
|
||||
for C in C_GRID:
|
||||
p = oof_proba_grouped(X, y, folds, C=C)
|
||||
mp = cls_metrics(y, p.argmax(1).tolist())
|
||||
grid[float(C)] = {"acc": mp["acc"], "macro_f1": mp["macro_f1"]}
|
||||
print(f" C={C} acc={mp['acc']:.4f} macroF1={mp['macro_f1']:.4f}")
|
||||
if mp["macro_f1"] > bestMac:
|
||||
bestMac, bestC = mp["macro_f1"], C
|
||||
print(f" -> best C={bestC}")
|
||||
p_best = oof_proba_grouped(X, y, folds, C=bestC)
|
||||
m_best = cls_metrics(y, p_best.argmax(1).tolist())
|
||||
m_best["C"] = bestC
|
||||
m_best["C_grid"] = grid
|
||||
m_best["folds"] = fold_report(yt, p_best, folds, y)
|
||||
report["e5_linear"] = m_best
|
||||
for f, v in m_best["folds"]["per_fold"].items():
|
||||
print(f" fold {f}: acc={v['acc']:.4f} macroF1={v['macro_f1']:.4f}")
|
||||
print(f" fold acc mean={m_best['folds']['acc_mean']:.4f} "
|
||||
f"std={m_best['folds']['acc_std']:.4f}; "
|
||||
f"macroF1 mean={m_best['folds']['macro_f1_mean']:.4f} "
|
||||
f"std={m_best['folds']['macro_f1_std']:.4f}")
|
||||
for c in CLASSES:
|
||||
p_ = m_best["per_class"][c]
|
||||
print(f" {c:<14} P={p_['p']:.3f} R={p_['r']:.3f} F1={p_['f1']:.3f} n={p_['n']}")
|
||||
|
||||
# grammar-pure sensitivity for the primary head
|
||||
if "grammar_drift" in report:
|
||||
mp_gp = cls_metrics(y[gp_mask], p_best[gp_mask].argmax(1).tolist())
|
||||
report["e5_linear"]["grammar_pure"] = {
|
||||
"acc": mp_gp["acc"], "macro_f1": mp_gp["macro_f1"], "n": int(gp_mask.sum())}
|
||||
|
||||
# ── §4 floors ─────────────────────────────────────────────────────────
|
||||
print("\n§4 floors")
|
||||
# majority floor
|
||||
maj = CLASSES.index("knowledge")
|
||||
ym = np.full(len(y), maj)
|
||||
mm = cls_metrics(y, ym)
|
||||
report["floors"]["majority"] = {"acc": mm["acc"], "macro_f1": mm["macro_f1"],
|
||||
"per_class": mm["per_class"]}
|
||||
print(f" majority (predict {CLASSES[maj]}): acc={mm['acc']:.4f} macroF1={mm['macro_f1']:.4f}")
|
||||
|
||||
# centroid floor: cosine to per-class mean of the training folds' embeddings
|
||||
cf_proba = np.zeros((len(yt), len(CLASSES)))
|
||||
folds_arr = np.asarray(folds)
|
||||
for te_fold in sorted(set(folds_arr.tolist())):
|
||||
tr = folds_arr != te_fold
|
||||
te = folds_arr == te_fold
|
||||
centroids = []
|
||||
for c in CLASSES:
|
||||
idxs = np.where(tr & (y == c))[0]
|
||||
ctr = X[idxs].mean(axis=0)
|
||||
ctr = ctr / np.linalg.norm(ctr)
|
||||
centroids.append(ctr)
|
||||
Cm = np.vstack(centroids)
|
||||
sims = X[te] @ Cm.T
|
||||
cf_proba[te] = sims
|
||||
yc = cf_proba.argmax(1)
|
||||
# accuracy + macroF1 with the same 5-way
|
||||
mc = cls_metrics(y, yc.tolist())
|
||||
report["floors"]["centroid"] = {"acc": mc["acc"], "macro_f1": mc["macro_f1"],
|
||||
"per_class": mc["per_class"]}
|
||||
print(f" centroid cosine: acc={mc['acc']:.4f} macroF1={mc['macro_f1']:.4f}")
|
||||
|
||||
# sparse word+char logistic (slice18 builder, grouped CV, five-way)
|
||||
texts = [r["n_text"] for r in pool]
|
||||
Xs, _vec = slice18_sparse.build_features(texts, "both")
|
||||
psp = np.zeros((len(yt), len(CLASSES)))
|
||||
for te_fold in sorted(set(folds_arr.tolist())):
|
||||
tr = folds_arr != te_fold
|
||||
te = folds_arr == te_fold
|
||||
clf = slice18_sparse.LogisticRegression(
|
||||
C=1.0, max_iter=2000, solver="lbfgs", random_state=42)
|
||||
clf.fit(Xs[tr], yt[tr])
|
||||
psp[te] = clf.predict_proba(Xs[te])
|
||||
msp = cls_metrics(y, psp.argmax(1).tolist())
|
||||
report["floors"]["sparse_word_char"] = {
|
||||
"acc": msp["acc"], "macro_f1": msp["macro_f1"], "per_class": msp["per_class"],
|
||||
"vocab": slice18_sparse.vocab_size(_vec)}
|
||||
print(f" sparse both: acc={msp['acc']:.4f} macroF1={msp['macro_f1']:.4f} "
|
||||
f"vocab={report['floors']['sparse_word_char']['vocab']}")
|
||||
|
||||
# ── §5 route-family holdouts ─────────────────────────────────────────
|
||||
print("\n§5 route-family holdouts")
|
||||
fam = np.array([r["family_id"] for r in pool])
|
||||
holdouts = {}
|
||||
all_fams = sorted(set(fam.tolist()))
|
||||
for grp, fams in HOLDOUT_GROUPS.items():
|
||||
if fams is None:
|
||||
fams = [f for f in all_fams if f.startswith(grp + ":")]
|
||||
mask = np.isin(fam, fams)
|
||||
if mask.sum() == 0:
|
||||
continue
|
||||
tr = ~mask
|
||||
clf = slice18_sparse.LogisticRegression(
|
||||
C=bestC, max_iter=2000, solver="lbfgs", random_state=42)
|
||||
clf.fit(X[tr], yt[tr])
|
||||
ypgrp = clf.predict(X[mask])
|
||||
m = cls_metrics([CLASSES.index(c) for c in y[mask]], ypgrp.tolist())
|
||||
m["families"] = fams
|
||||
m["rows"] = int(mask.sum())
|
||||
holdouts[grp] = {"acc": m["acc"], "macro_f1": m["macro_f1"], "n": int(mask.sum()),
|
||||
"per_class": m["per_class"]}
|
||||
print(f" {grp:<14} n={m['rows']} acc={m['acc']:.4f} macroF1={m['macro_f1']:.4f}")
|
||||
# full leave-one-family-out summary
|
||||
lofo_accs = []
|
||||
lofo_f1s = []
|
||||
for f in all_fams:
|
||||
mask = fam == f
|
||||
tr = ~mask
|
||||
clf = slice18_sparse.LogisticRegression(
|
||||
C=bestC, max_iter=2000, solver="lbfgs", random_state=42)
|
||||
clf.fit(X[tr], yt[tr])
|
||||
ypf = clf.predict(X[mask])
|
||||
m = cls_metrics([CLASSES.index(c) for c in y[mask]], ypf.tolist())
|
||||
lofo_accs.append(m["acc"])
|
||||
lofo_f1s.append(m["macro_f1"])
|
||||
holdouts["_all_49_lo_"] = {"n_families": len(all_fams),
|
||||
"acc_mean": float(np.mean(lofo_accs)),
|
||||
"macro_f1_mean": float(np.mean(lofo_f1s))}
|
||||
report["family_holdouts"] = holdouts
|
||||
print(f" leave-one-family-out over {len(all_fams)} families: "
|
||||
f"acc mean={np.mean(lofo_accs):.4f} macroF1 mean={np.mean(lofo_f1s):.4f}")
|
||||
|
||||
# ── §6 knowledge vs memory_write ─────────────────────────────────────
|
||||
print("\n§6 knowledge vs memory_write")
|
||||
# reuse e5-linear OOF: does the model put the higher probability on the
|
||||
# right side (memory_write for a write, knowledge for a recall)?
|
||||
conf_km = np.zeros((2, 2))
|
||||
pk = p_best[:, CLASSES.index("knowledge")]
|
||||
pmw = p_best[:, CLASSES.index("memory_write")]
|
||||
for i in range(len(yt)):
|
||||
t = y[i]
|
||||
if t == "knowledge":
|
||||
conf_km[0, 1 if pmw[i] > pk[i] else 0] += 1
|
||||
elif t == "memory_write":
|
||||
conf_km[1, 1 if pmw[i] >= pk[i] else 0] += 1
|
||||
report["kmw"] = {"confusion_p_ordered": conf_km.tolist()}
|
||||
|
||||
# matched pairs with shared subject lexemes, corpus-justified
|
||||
def build_pairs(subject, fam_k, fam_mw):
|
||||
kr = [r for r in pool if r["family_id"] in fam_k]
|
||||
mr = [r for r in pool if r["family_id"] in fam_mw]
|
||||
pairs = []
|
||||
for mw in mr:
|
||||
for k in kr:
|
||||
if subject in mw["n_text"] and subject in k["n_text"]:
|
||||
pairs.append((mw["idx"], k["idx"], mw["n_text"], k["n_text"]))
|
||||
return pairs
|
||||
|
||||
sets = {
|
||||
"water": build_pairs("вод", ["knowledge:recall-fact"], ["fact:water"]),
|
||||
"homelab": build_pairs("dns", ["knowledge:homelab-status"], ["note:homelab"])
|
||||
+ build_pairs("сервер", ["knowledge:homelab-status"], ["note:homelab"])
|
||||
+ build_pairs("vlan", ["knowledge:homelab-status"], ["note:homelab"]),
|
||||
"task": build_pairs("задач", ["knowledge:task-check", "knowledge:deadline"],
|
||||
["note:task"]),
|
||||
}
|
||||
idx_of = {r["idx"]: i for i, r in enumerate(pool)}
|
||||
pair_rep = {}
|
||||
for name, pairs in sets.items():
|
||||
if not pairs:
|
||||
continue
|
||||
ok = 0
|
||||
margins = []
|
||||
bad = []
|
||||
for mi, ki, mx, kx in pairs:
|
||||
mi_i, ki_i = idx_of[mi], idx_of[ki]
|
||||
# MW row should get a higher memory_write probability than the K row
|
||||
mk = (pmw[mi_i] + 0.0)
|
||||
if pmw[mi_i] > pmw[ki_i]:
|
||||
ok += 1
|
||||
else:
|
||||
bad.append((mx[:46], round(float(pmw[mi_i]), 3), kx[:46], round(float(pmw[ki_i]), 3)))
|
||||
margins.append(pmw[mi_i] - pmw[ki_i])
|
||||
pair_rep[name] = {
|
||||
"pairs": len(pairs),
|
||||
"mw_over_k_order_acc": ok / len(pairs),
|
||||
"mean_margin": float(np.mean(margins)),
|
||||
"reversed_examples": bad[:6],
|
||||
}
|
||||
print(f" {name}: pairs={len(pairs)} order_acc={ok/len(pairs):.3f} "
|
||||
f"mean_margin={np.mean(margins):+.3f}")
|
||||
report["kmw"]["matched_pairs"] = pair_rep
|
||||
|
||||
# ── §7 uncertain as explicit class ────────────────────────────────────
|
||||
print("\n§7 uncertain")
|
||||
up = m_best["per_class"]["uncertain"]
|
||||
uc = m_best["confusion"][CLASSES.index("uncertain")]
|
||||
report["uncertain"] = {
|
||||
"per_class": up,
|
||||
"row_from_uncertain": {CLASSES[j]: int(uc[j]) for j in range(5)},
|
||||
"row_to_uncertain": {CLASSES[j]: int(m_best["confusion"][j][CLASSES.index("uncertain")])
|
||||
for j in range(5)},
|
||||
}
|
||||
print(f" uncertain n={up['n']} P={up['p']:.3f} R={up['r']:.3f} F1={up['f1']:.3f}")
|
||||
print(" wrong-→label pulled from uncertain:", report["uncertain"]["row_from_uncertain"])
|
||||
print(" →uncertain pulled from:", report["uncertain"]["row_to_uncertain"])
|
||||
|
||||
# ── §8 OOF confidence / calibration / abstention ─────────────────────
|
||||
print("\n§8 confidence / calibration")
|
||||
conf = p_best.max(1)
|
||||
right = (p_best.argmax(1) == yt)
|
||||
cer = {
|
||||
"correct_conf_mean": float(conf[right].mean()),
|
||||
"correct_conf_median": float(np.median(conf[right])),
|
||||
"wrong_conf_mean": float(conf[~right].mean()),
|
||||
"wrong_conf_median": float(np.median(conf[~right])),
|
||||
"ece": ece(yt, p_best)["ece"],
|
||||
"ece_bins": ece(yt, p_best)["bins"],
|
||||
"log_loss": float(log_loss(yt, p_best, labels=[0, 1, 2, 3, 4])),
|
||||
}
|
||||
# Brier is label-set specific: one-vs-rest mean
|
||||
briers = []
|
||||
for i in range(5):
|
||||
briers.append(brier_score_loss((yt == i).astype(int), p_best[:, i]))
|
||||
cer["brier_macro"] = float(np.mean(briers))
|
||||
report["confidence"] = cer
|
||||
print(f" right conf mean={cer['correct_conf_mean']:.3f} "
|
||||
f"wrong conf mean={cer['wrong_conf_mean']:.3f} ECE={cer['ece']:.4f}")
|
||||
print(f" log_loss={cer['log_loss']:.4f} brier_macro={cer['brier_macro']:.4f}")
|
||||
|
||||
thr_grid = np.linspace(0.10, 0.98, 45)
|
||||
abst = []
|
||||
for t in thr_grid:
|
||||
cov = (conf >= t).mean()
|
||||
if cov == 0:
|
||||
continue
|
||||
keep = conf >= t
|
||||
yt_k = yt[keep]
|
||||
yp_k = p_best[keep].argmax(1)
|
||||
mk_ = cls_metrics(yt_k.tolist(), yp_k.tolist())
|
||||
abst.append({"threshold": round(float(t), 3), "coverage": float(cov),
|
||||
"accuracy": mk_["acc"], "macro_f1": mk_["macro_f1"]})
|
||||
report["confidence"]["abstention_curve"] = abst
|
||||
print(" threshold | coverage | accuracy | macroF1 (first 6/45 + knee)")
|
||||
for row in abst[::9]:
|
||||
print(f" {row['threshold']:.2f} | {row['coverage']:.3f} | "
|
||||
f"{row['accuracy']:.3f} | {row['macro_f1']:.3f}")
|
||||
|
||||
# ── §9 action OOD probes ──────────────────────────────────────────────
|
||||
print("\n§9 action OOD")
|
||||
ood_rows = [r for r in json.load(open(os.path.join(OUT_DIR, "ood.json")))]
|
||||
emb_by_idx = {i: np.asarray(e["embedding"], dtype=np.float64)
|
||||
for i, e in enumerate(slice18_sparse.filter_dev_pool(
|
||||
slice18_sparse.load_data()[1]))}
|
||||
Xo = np.vstack([emb_by_idx[r["idx"]] for r in ood_rows])
|
||||
fold_models = []
|
||||
for te_fold in sorted(set(folds_arr.tolist())):
|
||||
tr = folds_arr != te_fold
|
||||
clf = slice18_sparse.LogisticRegression(
|
||||
C=bestC, max_iter=2000, solver="lbfgs", random_state=42)
|
||||
clf.fit(X[tr], yt[tr])
|
||||
fold_models.append(clf)
|
||||
# OOD rows are not in folds; use the full-train model to keep it simple and
|
||||
# comparable to the non-action in-fold behaviour
|
||||
po = np.zeros((len(Xo), 5))
|
||||
for clf in fold_models:
|
||||
po += clf.predict_proba(Xo)
|
||||
po /= len(fold_models)
|
||||
ood_top = int(np.argmax(po.mean(0)))
|
||||
ood_conf = po.max(1)
|
||||
ood_pred = po.argmax(1)
|
||||
top_dist = {CLASSES[i]: int((ood_pred == i).sum()) for i in range(5)}
|
||||
confident_na = int((ood_conf > 0.9).sum())
|
||||
report["ood"] = {
|
||||
"n": len(ood_rows),
|
||||
"top_class": CLASSES[int(ood_top)],
|
||||
"top_class_dist": top_dist,
|
||||
"conf_gt_0.9": confident_na,
|
||||
"conf_gt_0.9_frac": float(confident_na / len(ood_rows)),
|
||||
"conf_mean": float(ood_conf.mean()),
|
||||
"conf_median": float(np.median(ood_conf)),
|
||||
}
|
||||
print(f" action OOD n={len(ood_rows)}: most-confident class={report['ood']['top_class']} "
|
||||
f"dist={top_dist}")
|
||||
print(f" conf>0.9: {confident_na} ({confident_na/len(ood_rows):.3f}) "
|
||||
f"conf mean={report['ood']['conf_mean']:.3f}")
|
||||
|
||||
# ── §10 artifact cost ────────────────────────────────────────────────
|
||||
print("\n§10 artifact")
|
||||
n_params = len(CLASSES) * X.shape[1] + len(CLASSES)
|
||||
fp32 = n_params * 4
|
||||
report["artifact"] = {
|
||||
"e5_dim": X.shape[1],
|
||||
"head_params": n_params,
|
||||
"head_fp32_bytes": fp32,
|
||||
"head_fp32_kib": fp32 / 1024,
|
||||
"head_int8_bytes": n_params,
|
||||
}
|
||||
# incremental latency of the linear head over a batch of 1 (µs)
|
||||
clf = slice18_sparse.LogisticRegression(C=bestC, max_iter=2000, solver="lbfgs", random_state=42)
|
||||
clf.fit(X, yt)
|
||||
x1 = X[:1]
|
||||
for _ in range(50):
|
||||
clf.predict_proba(x1)
|
||||
lat = []
|
||||
for _ in range(2000):
|
||||
t0 = time.perf_counter_ns()
|
||||
clf.predict_proba(x1)
|
||||
lat.append((time.perf_counter_ns() - t0) / 1e3)
|
||||
lat = np.array(lat)
|
||||
report["artifact"]["head_latency_us_mean"] = float(lat.mean())
|
||||
report["artifact"]["head_latency_us_p50"] = float(np.median(lat))
|
||||
print(f" head params={n_params} fp32={fp32/1024:.2f}KiB "
|
||||
f"lat mean={lat.mean():.2f}us p50={np.median(lat):.2f}us")
|
||||
|
||||
with open(os.path.join(OUT_DIR, "results.json"), "w") as f:
|
||||
json.dump(report, f, ensure_ascii=False, indent=1, default=float)
|
||||
print(f"\nwrote {OUT_DIR}/results.json")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user