Files
Maven/cmd/semantic-router-experiment/diagnostic.py
T
claude 774c0660b0 router/semantic: slice 16 diagnostic — action/non-action boundary analysis
Experiments:
- Residual-only OOF safety metrics
- False-action decomposition by semantic family (7 families, 24 split groups)
- Paired e5 geometry (gap=0.010 between action and capability question)
- Binary action probe (PR-AUC=0.707, ROC-AUC=0.863)
- Cost-sensitive linear classification
- Expanded action-threshold sweep (7 thresholds)
- Voice-like punctuation stress evaluation
- e5 + tiny structural features
- Four-hypothesis comparison table

Verdict: e5 representation is the primary bottleneck. Action seeds and
capability questions are nearly indistinguishable in embedding space.
2026-09-07 14:02:00 +04:00

1138 lines
49 KiB
Python

#!/usr/bin/env python3
"""
Slice 16 Diagnostic: Action/Non-Action Boundary Analysis
========================================================
Determines exactly why the action/non-action boundary still fails,
before any nonlinear model or production gate.
"""
import json
import re
import sys
import warnings
from collections import defaultdict
from pathlib import Path
import numpy as np
from sklearn.exceptions import ConvergenceWarning
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import (
accuracy_score,
average_precision_score,
brier_score_loss,
confusion_matrix,
f1_score,
precision_recall_curve,
precision_recall_fscore_support,
roc_auc_score,
)
from sklearn.preprocessing import LabelEncoder
warnings.filterwarnings("ignore", category=ConvergenceWarning)
EMBEDDING_PATH = "/tmp/mvn-experiment/embeddings.json"
ROUTES = ["action", "conversation", "knowledge", "memory_write", "system", "uncertain"]
ROUTE_IDX = {r: i for i, r in enumerate(ROUTES)}
C_VALUES = [0.01, 0.1, 1.0, 10.0, 100.0]
ACTION_THRESHOLDS = [0.50, 0.60, 0.70, 0.80, 0.85, 0.90, 0.95]
COST_MULTIPLIERS = [1, 2, 4, 8, 16]
# ─── Data Loading ───────────────────────────────────────────────────────────
def load_data():
with open(EMBEDDING_PATH) as f:
data = json.load(f)
meta = data["meta"]
examples = data["examples"]
return meta, examples
def filter_dev_pool(examples):
return [e for e in examples if e["dev_pool"]]
def filter_residual(examples):
return [e for e in examples if not e["fast_path_resolved"]]
def extract_Xy(examples):
X = np.array([e["embedding"] for e in examples])
y = np.array([e["route"] for e in examples])
return X, y
def get_fold_groups(examples):
return np.array([e["cv_fold"] for e in examples])
# ─── Helpers ─────────────────────────────────────────────────────────────────
def pct(v, d=1):
return f"{100*v:.{d}f}%"
def ff(v, d=3):
return f"{v:.{d}f}"
# ─── Section 1: Residual Safety Metrics ─────────────────────────────────────
def run_six_way_cv(X, y, fold_ids, examples, C_values):
"""Run 6-way grouped CV and return OOF predictions for best C."""
unique_folds = sorted(set(fold_ids))
results_by_C = {}
for C in C_values:
oof_rows = []
fold_metrics = []
for test_fold in unique_folds:
train_mask = fold_ids != test_fold
test_mask = fold_ids == test_fold
X_train, y_train = X[train_mask], y[train_mask]
X_test, y_test = X[test_mask], y[test_mask]
model = LogisticRegression(C=C, max_iter=2000, solver="lbfgs", random_state=42)
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
y_proba = model.predict_proba(X_test)
classes = model.classes_
acc = accuracy_score(y_test, y_pred)
macro_f1 = f1_score(y_test, y_pred, average="macro", zero_division=0)
false_action = sum(1 for t, p in zip(y_test, y_pred) if t != "action" and p == "action")
fold_metrics.append({
"fold": test_fold, "accuracy": acc, "macro_f1": macro_f1,
"false_action": false_action, "test_size": len(X_test),
})
test_indices = np.where(test_mask)[0]
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
meta = examples[test_indices[i]]
proba_dict = {cls: float(y_proba[i][j]) for j, cls in enumerate(classes)}
oof_rows.append({
"source_id": meta["source_id"],
"fold": test_fold,
"true": true,
"predicted": pred,
"correct": true == pred,
"proba": proba_dict,
"text": meta["text"],
"tags": meta.get("tags", []),
"split_group": meta.get("split_group", ""),
"family_id": meta.get("family_id", ""),
})
mean_f1 = np.mean([m["macro_f1"] for m in fold_metrics])
results_by_C[C] = {"mean_f1": mean_f1, "fold_metrics": fold_metrics, "oof": oof_rows}
best_C = max(results_by_C, key=lambda c: results_by_C[c]["mean_f1"])
return best_C, results_by_C
def compute_action_metrics(oof_rows):
"""Compute action-specific safety metrics from OOF predictions."""
y_true = np.array([r["true"] for r in oof_rows])
y_pred = np.array([r["predicted"] for r in oof_rows])
action_tp = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p == "action")
action_fp = sum(1 for t, p in zip(y_true, y_pred) if t != "action" and p == "action")
action_fn = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p != "action")
action_precision = action_tp / max(action_tp + action_fp, 1)
action_recall = action_tp / max(action_tp + action_fn, 1)
false_action = action_fp
false_action_rate = false_action / max(len(y_true), 1)
# Uncertain F1
unc_tp = sum(1 for t, p in zip(y_true, y_pred) if t == "uncertain" and p == "uncertain")
unc_fp = sum(1 for t, p in zip(y_true, y_pred) if t != "uncertain" and p == "uncertain")
unc_fn = sum(1 for t, p in zip(y_true, y_pred) if t == "uncertain" and p != "uncertain")
unc_prec = unc_tp / max(unc_tp + unc_fp, 1)
unc_rec = unc_tp / max(unc_tp + unc_fn, 1)
unc_f1 = 2 * unc_prec * unc_rec / max(unc_prec + unc_rec, 1e-9)
return {
"action_precision": action_precision,
"action_recall": action_recall,
"false_action_count": false_action,
"false_action_rate": false_action_rate,
"uncertain_f1": unc_f1,
"total": len(y_true),
}
# ─── Section 2: False-Action Decomposition ─────────────────────────────────
# Semantic family classification from source_id prefix and tags
def classify_semantic_family(row):
sid = row["source_id"]
tags = row.get("tags", [])
text = row["text"]
if "capability_question" in tags:
return "capability_question"
if "question" in tags:
return "question"
if "negation" in tags:
return "negation"
if "reported_speech" in tags:
return "reported_speech"
if "quotation" in tags:
return "quotation"
if "hypothetical" in tags:
return "hypothetical"
# Classify by source_id prefix and text patterns
if sid.startswith("mw-note-task"):
return "memory_write"
if sid.startswith("mw-free-remember"):
return "memory_write"
if sid.startswith("mw-fact"):
return "memory_write"
if sid.startswith("mw-note-homelab"):
return "memory_write"
if sid.startswith("mw-note-idea"):
return "memory_write"
if sid.startswith("kq-cap"):
return "capability_question"
if sid.startswith("kq-world"):
return "knowledge_general"
if sid.startswith("kq-homelab"):
return "knowledge_general"
if sid.startswith("kq-task"):
return "knowledge_general"
if sid.startswith("kq-cal"):
return "knowledge_general"
if sid.startswith("kq-deadline"):
return "knowledge_general"
if sid.startswith("sys-"):
return "system"
if sid.startswith("conv-"):
return "conversation"
if sid.startswith("unc-"):
return "uncertain"
# Text-based fallback
q_markers = ["что такое", "кто такой", "что значит", "как дела"]
if any(m in text.lower() for m in q_markers):
return "knowledge_general"
if "?" in text:
return "question"
return "other"
def decompose_false_actions(oof_rows):
"""Decompose false actions by semantic family and split group."""
false_actions = [r for r in oof_rows if r["true"] != "action" and r["predicted"] == "action"]
by_family = defaultdict(list)
by_split_group = defaultdict(list)
by_fold = defaultdict(list)
for r in false_actions:
family = classify_semantic_family(r)
by_family[family].append(r)
by_split_group[r["split_group"]].append(r)
by_fold[r["fold"]].append(r)
return {
"total": len(false_actions),
"by_family": dict(by_family),
"by_split_group": dict(by_split_group),
"by_fold": dict(by_fold),
"details": false_actions,
}
# ─── Section 3: E5 Geometry ────────────────────────────────────────────────
def cosine_sim(a, b):
a = np.array(a, dtype=np.float64)
b = np.array(b, dtype=np.float64)
na = np.linalg.norm(a)
nb = np.linalg.norm(b)
if na == 0 or nb == 0:
return 0.0
return float(np.dot(a, b) / (na * nb))
def inspect_e5_geometry(examples):
"""For every action seed, compare embedding similarity to:
- positive action variants (same group, action route)
- nearest capability question (by embedding distance)
- nearest other negative contrast (knowledge/uncertain, by embedding distance)
"""
emb_lookup = {e["source_id"]: np.array(e["embedding"]) for e in examples}
# Action seeds: corpus_factory_v2 action examples without contrastive tags
action_seeds = [e for e in examples
if e["route"] == "action"
and e["source"] == "corpus_factory_v2"
and e.get("dev_pool", False)
and not any(t in e.get("tags", []) for t in
["negation", "question", "reported_speech",
"quotation", "hypothetical", "capability_question"])]
# Pool of capability questions (dev pool)
cap_qs = [e for e in examples
if "capability_question" in e.get("tags", [])
and e.get("dev_pool", False)]
# Pool of other negatives (knowledge/uncertain routes, dev pool, not capability_question)
other_neg = [e for e in examples
if e["route"] in ("knowledge", "uncertain")
and "capability_question" not in e.get("tags", [])
and e.get("dev_pool", False)
and e["route"] != "action"]
# Group action seeds by split_group for positive pairs
by_group = defaultdict(list)
for e in examples:
by_group[e["split_group"]].append(e)
positive_pairs = []
capability_pairs = []
other_negative_pairs = []
for seed in action_seeds:
seed_emb = emb_lookup.get(seed["source_id"])
if seed_emb is None:
continue
# Positive: same-group action variants
groupmates = by_group.get(seed["split_group"], [])
for c in groupmates:
if c["source_id"] == seed["source_id"] or c["route"] != "action":
continue
c_emb = emb_lookup.get(c["source_id"])
if c_emb is None:
continue
sim = cosine_sim(seed_emb, c_emb)
positive_pairs.append({
"seed_id": seed["source_id"], "contrast_id": c["source_id"],
"similarity": sim, "text": c["text"],
})
# Nearest capability question
best_cap, best_cap_sim = None, -1.0
for cap in cap_qs:
cap_emb = emb_lookup.get(cap["source_id"])
if cap_emb is None:
continue
sim = cosine_sim(seed_emb, cap_emb)
if sim > best_cap_sim:
best_cap_sim = sim
best_cap = cap
if best_cap:
capability_pairs.append({
"seed_id": seed["source_id"], "contrast_id": best_cap["source_id"],
"similarity": best_cap_sim, "text": best_cap["text"],
})
# Nearest other negative
best_neg, best_neg_sim = None, -1.0
for neg in other_neg:
neg_emb = emb_lookup.get(neg["source_id"])
if neg_emb is None:
continue
sim = cosine_sim(seed_emb, neg_emb)
if sim > best_neg_sim:
best_neg_sim = sim
best_neg = neg
if best_neg:
other_negative_pairs.append({
"seed_id": seed["source_id"], "contrast_id": best_neg["source_id"],
"similarity": best_neg_sim, "text": best_neg["text"],
})
return {
"positive": positive_pairs,
"capability": capability_pairs,
"other_negative": other_negative_pairs,
}
# ─── Section 4: Binary Action Probe ────────────────────────────────────────
def run_binary_action_probe(X, y, fold_ids, C_values):
"""Binary action vs not-action logistic probe with grouped CV."""
y_binary = np.array(["action" if t == "action" else "not_action" for t in y])
unique_folds = sorted(set(fold_ids))
results_by_C = {}
for C in C_values:
oof_rows = []
fold_metrics = []
for test_fold in unique_folds:
train_mask = fold_ids != test_fold
test_mask = fold_ids == test_fold
X_train, y_train = X[train_mask], y_binary[train_mask]
X_test, y_test = X[test_mask], y_binary[test_mask]
model = LogisticRegression(C=C, max_iter=2000, solver="lbfgs", random_state=42)
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
y_proba = model.predict_proba(X_test)
classes = model.classes_
action_idx = list(classes).index("action")
action_proba = y_proba[:, action_idx]
acc = accuracy_score(y_test, y_pred)
false_pos = sum(1 for t, p in zip(y_test, y_pred) if t == "not_action" and p == "action")
false_neg = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p == "not_action")
# ROC-AUC and PR-AUC
y_test_binary = np.array([1 if t == "action" else 0 for t in y_test])
if len(np.unique(y_test_binary)) > 1:
roc = roc_auc_score(y_test_binary, action_proba)
pr_auc = average_precision_score(y_test_binary, action_proba)
else:
roc = 0.0
pr_auc = 0.0
prec, rec, f1, sup = precision_recall_fscore_support(
y_test, y_pred, labels=["action", "not_action"], zero_division=0
)
action_prec = prec[0]
action_rec = rec[0]
fold_metrics.append({
"fold": test_fold, "accuracy": acc, "roc_auc": roc, "pr_auc": pr_auc,
"action_precision": action_prec, "action_recall": action_rec,
"false_pos": false_pos, "false_neg": false_neg,
"test_size": len(X_test),
})
test_indices = np.where(test_mask)[0]
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
oof_rows.append({
"fold": test_fold, "true": true, "predicted": pred,
"action_proba": float(action_proba[i]),
"correct": true == pred,
})
mean_roc = np.mean([m["roc_auc"] for m in fold_metrics])
mean_pr = np.mean([m["pr_auc"] for m in fold_metrics])
total_fp = sum(m["false_pos"] for m in fold_metrics)
total_fn = sum(m["false_neg"] for m in fold_metrics)
mean_prec = np.mean([m["action_precision"] for m in fold_metrics])
mean_rec = np.mean([m["action_recall"] for m in fold_metrics])
results_by_C[C] = {
"mean_roc_auc": mean_roc, "mean_pr_auc": mean_pr,
"total_fp": total_fp, "total_fn": total_fn,
"mean_precision": mean_prec, "mean_recall": mean_rec,
"fold_metrics": fold_metrics, "oof": oof_rows,
}
best_C = max(results_by_C, key=lambda c: results_by_C[c]["mean_pr_auc"])
return best_C, results_by_C
# ─── Section 5: Cost-Sensitive Classification ──────────────────────────────
def run_cost_sensitive(X, y, fold_ids, cost_multipliers):
"""Evaluate cost-sensitive 6-way classification with grouped CV."""
unique_folds = sorted(set(fold_ids))
results = {}
for cost in cost_multipliers:
oof_rows = []
fold_metrics = []
for test_fold in unique_folds:
train_mask = fold_ids != test_fold
test_mask = fold_ids == test_fold
X_train, y_train = X[train_mask], y[train_mask]
X_test, y_test = X[test_mask], y[test_mask]
# Compute class weights: action gets cost multiplier, others get 1
classes = sorted(set(y_train))
class_weights = {}
for c in classes:
if c == "action":
class_weights[c] = cost
else:
class_weights[c] = 1.0
# Normalize so weights sum to n_classes
total_w = sum(class_weights.values())
for c in class_weights:
class_weights[c] *= len(classes) / total_w
model = LogisticRegression(
C=10.0, max_iter=2000, solver="lbfgs",
class_weight=class_weights, random_state=42,
)
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
action_tp = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p == "action")
action_fp = sum(1 for t, p in zip(y_test, y_pred) if t != "action" and p == "action")
action_fn = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p != "action")
false_action = action_fp
action_prec = action_tp / max(action_tp + action_fp, 1)
action_rec = action_tp / max(action_tp + action_fn, 1)
false_action_rate = false_action / max(len(y_test), 1)
fold_metrics.append({
"fold": test_fold, "action_precision": action_prec,
"action_recall": action_rec, "false_action": false_action,
"false_action_rate": false_action_rate, "test_size": len(X_test),
})
test_indices = np.where(test_mask)[0]
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
oof_rows.append({
"fold": test_fold, "true": true, "predicted": pred,
"correct": true == pred,
})
mean_prec = np.mean([m["action_precision"] for m in fold_metrics])
mean_rec = np.mean([m["action_recall"] for m in fold_metrics])
total_fa = sum(m["false_action"] for m in fold_metrics)
mean_fa_rate = np.mean([m["false_action_rate"] for m in fold_metrics])
results[cost] = {
"mean_precision": mean_prec, "mean_recall": mean_rec,
"total_false_action": total_fa, "mean_false_action_rate": mean_fa_rate,
}
return results
# ─── Section 6: Action Threshold Sweep ─────────────────────────────────────
def run_action_threshold_sweep(oof_rows, thresholds):
"""Evaluate action-specific threshold on the already-trained OOF predictions."""
results = []
for thr in thresholds:
action_pred = []
for r in oof_rows:
p = r["predicted"]
proba = r["proba"].get("action", 0.0)
if p == "action" and proba < thr:
sorted_routes = sorted(r["proba"].items(), key=lambda x: -x[1])
for route, _ in sorted_routes:
if route != "action":
p = route
break
action_pred.append(p)
y_true = np.array([r["true"] for r in oof_rows])
y_pred = np.array(action_pred)
action_tp = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p == "action")
action_fp = sum(1 for t, p in zip(y_true, y_pred) if t != "action" and p == "action")
action_fn = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p != "action")
false_action = action_fp
action_prec = action_tp / max(action_tp + action_fp, 1)
action_rec = action_tp / max(action_tp + action_fn, 1)
# Coverage: fraction of examples where model is confident enough
# (not demoted)
n_demoted = sum(1 for i, r in enumerate(oof_rows)
if r["predicted"] == "action" and r["proba"].get("action", 0) < thr)
coverage = (len(y_true) - n_demoted) / max(len(y_true), 1)
results.append({
"threshold": thr,
"action_precision": action_prec,
"action_recall": action_rec,
"false_action_count": false_action,
"coverage": coverage,
})
return results
# ─── Section 7: Voice-Like Stress ──────────────────────────────────────────
def apply_voice_stress(text):
"""Apply representation changes typical of STT."""
# Remove final punctuation
t = re.sub(r'[?.!,;:]+$', '', text.strip())
# Remove all punctuation (safe for Russian)
t = re.sub(r'[^\w\s]', '', t)
# Lowercase
t = t.lower()
# Collapse whitespace
t = re.sub(r'\s+', ' ', t).strip()
return t
def run_voice_stress_eval(examples, fold_ids_all, dev_indices, X_all, y_all):
"""Score voice-stressed variants without retraining."""
dev_examples = [examples[i] for i in dev_indices]
stress_pairs = []
for e in dev_examples:
original_text = e["text"]
stressed_text = apply_voice_stress(original_text)
if stressed_text != original_text:
stress_pairs.append({
"source_id": e["source_id"],
"original": original_text,
"stressed": stressed_text,
"route": e["route"],
"fold": e["cv_fold"],
"tags": e.get("tags", []),
})
# Count capability-question pairs specifically
cap_q_affected = [p for p in stress_pairs
if "capability_question" in p["tags"]]
cap_q_with_q = [p for p in cap_q_affected if "?" in p["original"]]
cap_q_comma = [p for p in cap_q_affected if "," in p["original"] and "?" not in p["original"]]
return {
"total_stress_pairs": len(stress_pairs),
"cap_q_total": len(cap_q_affected),
"cap_q_with_question_mark": len(cap_q_with_q),
"cap_q_with_comma_only": len(cap_q_comma),
"pairs": stress_pairs[:20],
}
# ─── Section 8: E5 + Structural Features ───────────────────────────────────
def is_question_shaped(text):
"""Boolean feature: does the text look like a question?"""
q_starters = [
"что ", "кто ", "как ", "где ", "когда ", "почему ", "зачем ",
"сколько ", "можно ли ", "нужно ли ", "есть ли ", "хватает ли ",
"какой ", "какая ", "какие ", "чей ", "чья ",
]
t = text.lower()
return any(t.startswith(s) for s in q_starters)
def has_trailing_question_mark(text):
return text.rstrip().endswith("?")
def has_prohibition(text):
"""Check for negation patterns (prohibition)."""
t = text.lower()
return t.startswith("не ") or t.startswith("ни ")
def extract_structural_features(examples):
"""Extract [e5_embedding ; IsQuestion ; trailing_? ; prohibition] for each example."""
structural = []
for e in examples:
text = e["text"]
features = [
1.0 if is_question_shaped(text) else 0.0,
1.0 if has_trailing_question_mark(text) else 0.0,
1.0 if has_prohibition(text) else 0.0,
]
structural.append(features)
return np.array(structural)
def run_e5_structural_experiment(X_e5, structural, y, fold_ids, C_values):
"""Run grouped CV with [e5 ; structural bits]."""
X_aug = np.hstack([X_e5, structural])
unique_folds = sorted(set(fold_ids))
results_by_C = {}
for C in C_values:
oof_rows = []
fold_metrics = []
for test_fold in unique_folds:
train_mask = fold_ids != test_fold
test_mask = fold_ids == test_fold
X_train, y_train = X_aug[train_mask], y[train_mask]
X_test, y_test = X_aug[test_mask], y[test_mask]
model = LogisticRegression(C=C, max_iter=2000, solver="lbfgs", random_state=42)
model.fit(X_train, y_train)
y_pred = model.predict(X_test)
y_proba = model.predict_proba(X_test)
classes = model.classes_
acc = accuracy_score(y_test, y_pred)
macro_f1 = f1_score(y_test, y_pred, average="macro", zero_division=0)
action_tp = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p == "action")
action_fp = sum(1 for t, p in zip(y_test, y_pred) if t != "action" and p == "action")
action_fn = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p != "action")
false_action = action_fp
action_prec = action_tp / max(action_tp + action_fp, 1)
action_rec = action_tp / max(action_tp + action_fn, 1)
false_action_rate = false_action / max(len(y_test), 1)
fold_metrics.append({
"fold": test_fold, "accuracy": acc, "macro_f1": macro_f1,
"action_precision": action_prec, "action_recall": action_rec,
"false_action": false_action, "false_action_rate": false_action_rate,
"test_size": len(X_test),
})
test_indices = np.where(test_mask)[0]
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
proba_dict = {cls: float(y_proba[i][j]) for j, cls in enumerate(classes)}
oof_rows.append({
"fold": test_fold, "true": true, "predicted": pred,
"correct": true == pred, "proba": proba_dict,
})
mean_f1 = np.mean([m["macro_f1"] for m in fold_metrics])
total_fa = sum(m["false_action"] for m in fold_metrics)
mean_fa_rate = np.mean([m["false_action_rate"] for m in fold_metrics])
results_by_C[C] = {
"mean_f1": mean_f1, "total_false_action": total_fa,
"mean_false_action_rate": mean_fa_rate,
"fold_metrics": fold_metrics, "oof": oof_rows,
}
best_C = max(results_by_C, key=lambda c: results_by_C[c]["mean_f1"])
return best_C, results_by_C
# ─── Section 9: Four-Way Comparison ────────────────────────────────────────
def compute_six_way_metrics(oof_rows):
"""Compute macro F1 for 6-way model from OOF rows."""
y_true = np.array([r["true"] for r in oof_rows])
y_pred = np.array([r["predicted"] for r in oof_rows])
return compute_action_metrics(oof_rows) | {
"macro_f1": float(f1_score(y_true, y_pred, average="macro", zero_division=0)),
}
# ─── Report Generation ──────────────────────────────────────────────────────
def main():
print("Loading data...")
meta, examples = load_data()
dev_examples = filter_dev_pool(examples)
dev_residual = filter_residual(dev_examples)
X_dev, y_dev = extract_Xy(dev_examples)
fold_ids_dev = get_fold_groups(dev_examples)
X_res, y_res = extract_Xy(dev_residual)
fold_ids_res = get_fold_groups(dev_residual)
print(f"Dev pool: {len(dev_examples)} examples, residual: {len(dev_residual)}")
lines = []
lines.append("# Slice 16 Diagnostic: Action/Non-Action Boundary Analysis")
lines.append("")
# ─── 0. Slice 15 hashes ────────────────────────────────────────────────
lines.append("## 0. Slice 15 Commit Hashes")
lines.append("")
lines.append("```text")
lines.append("semantic seed type: 07bfcea")
lines.append("merge-corpus tool: 6397ea1")
lines.append("e5 embedding cache: f8ec77d")
lines.append("sklearn experiment: 9df2239")
lines.append("corpus-factory: 4bb7555")
lines.append("expanded corpus: 87411e5")
lines.append("contract tests: ad3f2d6")
lines.append("experiment reports: cbac8b9")
lines.append("")
lines.append(f"development corpus v2 hash: {meta.get('dataset_hash', 'b27fd48f478ca477')}")
lines.append(f"original frozen holdout hash: ad297fbdbbea704b (byte-identical, uninspected)")
lines.append("```")
lines.append("")
# ─── 1. Residual Safety Metrics ────────────────────────────────────────
lines.append("## 1. Router-Residual OOF Safety Metrics")
lines.append("")
best_C_six, six_way_results = run_six_way_cv(X_res, y_res, fold_ids_res, dev_residual, C_VALUES)
six_oof = six_way_results[best_C_six]["oof"]
safety = compute_action_metrics(six_oof)
lines.append("```text")
lines.append(f"action precision: {ff(safety['action_precision'])}")
lines.append(f"action recall: {ff(safety['action_recall'])}")
lines.append(f"false-action count: {safety['false_action_count']}")
lines.append(f"false-action rate: {pct(safety['false_action_rate'])}")
lines.append(f"uncertain F1: {ff(safety['uncertain_f1'])}")
lines.append(f"total examples: {safety['total']}")
lines.append(f"best C: {best_C_six}")
lines.append("```")
lines.append("")
# ─── 2. False-Action Decomposition ─────────────────────────────────────
lines.append("## 2. False-Action Decomposition by Semantic Family")
lines.append("")
decomposed = decompose_false_actions(six_oof)
lines.append(f"Total false actions: {decomposed['total']}")
lines.append("")
lines.append("### By semantic family")
lines.append("")
lines.append(f"{'family':<25} {'count':>6} {'rate':>8}")
lines.append("-" * 40)
for family, rows in sorted(decomposed["by_family"].items(), key=lambda x: -len(x[1])):
rate = len(rows) / max(decomposed["total"], 1)
lines.append(f"{family:<25} {len(rows):>6} {pct(rate):>8}")
lines.append("")
lines.append("### By held-out split group (top 20)")
lines.append("")
lines.append(f"{'split_group':<30} {'count':>6}")
lines.append("-" * 37)
for sg, rows in sorted(decomposed["by_split_group"].items(), key=lambda x: -len(x[1]))[:20]:
lines.append(f"{sg:<30} {len(rows):>6}")
lines.append("")
lines.append("### By fold")
lines.append("")
lines.append(f"{'fold':>5} {'count':>6}")
lines.append("-" * 12)
for fold, rows in sorted(decomposed["by_fold"].items()):
lines.append(f"{fold:>5} {len(rows):>6}")
lines.append("")
# Diagnosis
lines.append("### Diagnosis")
lines.append("")
n_families = len(decomposed["by_family"])
n_groups = len(decomposed["by_split_group"])
max_family = max(decomposed["by_family"].items(), key=lambda x: len(x[1]))
fold_counts = [len(v) for v in decomposed["by_fold"].values()]
fold_cv = np.std(fold_counts) / max(np.mean(fold_counts), 1e-9)
lines.append(f"- Distinct semantic families contributing false actions: {n_families}")
lines.append(f"- Distinct held-out split groups: {n_groups}")
lines.append(f"- Largest single family: {max_family[0]} ({len(max_family[1])} false actions)")
lines.append(f"- Fold false-action count CV (std/mean): {fold_cv:.2f}")
if fold_cv > 0.5:
lines.append("- High fold variance suggests bad CV folds are a significant contributor")
elif n_families <= 3:
lines.append("- Few families suggests concentrated template failures")
else:
lines.append("- Broad distribution across families suggests systematic action/non-action overlap")
lines.append("")
# ─── 3. E5 Geometry ───────────────────────────────────────────────────
lines.append("## 3. Paired E5 Geometry")
lines.append("")
geometry = inspect_e5_geometry(examples)
if geometry["positive"]:
pos_sims = [p["similarity"] for p in geometry["positive"]]
lines.append(f"cosine(action seed, positive action):")
lines.append(f" mean={ff(np.mean(pos_sims))} std={ff(np.std(pos_sims))} "
f"min={ff(np.min(pos_sims))} max={ff(np.max(pos_sims))} n={len(pos_sims)}")
else:
lines.append("No positive action pairs found.")
if geometry["capability"]:
cap_sims = [p["similarity"] for p in geometry["capability"]]
lines.append(f"cosine(action seed, capability question):")
lines.append(f" mean={ff(np.mean(cap_sims))} std={ff(np.std(cap_sims))} "
f"min={ff(np.min(cap_sims))} max={ff(np.max(cap_sims))} n={len(cap_sims)}")
else:
lines.append("No capability question pairs found.")
if geometry["other_negative"]:
neg_sims = [p["similarity"] for p in geometry["other_negative"]]
lines.append(f"cosine(action seed, other negative contrast):")
lines.append(f" mean={ff(np.mean(neg_sims))} std={ff(np.std(neg_sims))} "
f"min={ff(np.min(neg_sims))} max={ff(np.max(neg_sims))} n={len(neg_sims)}")
else:
lines.append("No other negative pairs found.")
lines.append("")
# Overlap assessment
if geometry["positive"] and geometry["capability"]:
pos_mean = np.mean([p["similarity"] for p in geometry["positive"]])
cap_mean = np.mean([p["similarity"] for p in geometry["capability"]])
gap = pos_mean - cap_mean
lines.append(f"Separation gap (positive - capability): {ff(gap)}")
if gap < 0.05:
lines.append("**WARNING**: e5 maps action seeds and capability questions nearly on top of each other.")
lines.append("The representation itself may not preserve useful separation for this boundary.")
elif gap < 0.15:
lines.append("Moderate separation. e5 preserves some signal but the boundary is narrow.")
else:
lines.append("Good separation. e5 preserves useful distance between action and capability question.")
lines.append("")
# ─── 4. Binary Action Probe ───────────────────────────────────────────
lines.append("## 4. Binary Action Probe (Linear Logistic)")
lines.append("")
best_C_bin, bin_results = run_binary_action_probe(X_dev, y_dev, fold_ids_dev, C_VALUES)
bin_res = bin_results[best_C_bin]
lines.append(f"```text")
lines.append(f"ROC-AUC: {ff(bin_res['mean_roc_auc'])}")
lines.append(f"PR-AUC: {ff(bin_res['mean_pr_auc'])}")
lines.append(f"precision: {ff(bin_res['mean_precision'])}")
lines.append(f"recall: {ff(bin_res['mean_recall'])}")
lines.append(f"false-positive: {bin_res['total_fp']}")
lines.append(f"false-negative: {bin_res['total_fn']}")
lines.append(f"total: {sum(m['test_size'] for m in bin_res['fold_metrics'])}")
lines.append(f"best C: {best_C_bin}")
lines.append(f"```")
lines.append("")
# Per-fold
lines.append("Per-fold:")
for m in bin_res["fold_metrics"]:
lines.append(f" Fold {m['fold']}: ROC={ff(m['roc_auc'])} PR={ff(m['pr_auc'])} "
f"P={ff(m['action_precision'])} R={ff(m['action_recall'])} "
f"FP={m['false_pos']} FN={m['false_neg']} n={m['test_size']}")
lines.append("")
# ─── 5. Cost-Sensitive Linear ──────────────────────────────────────────
lines.append("## 5. Cost-Sensitive Linear Action Classification")
lines.append("")
cost_results = run_cost_sensitive(X_dev, y_dev, fold_ids_dev, COST_MULTIPLIERS)
lines.append(f"{'cost':>5} {'action_P':>10} {'action_R':>10} {'FA count':>10} {'FA rate':>10}")
lines.append("-" * 46)
for cost in COST_MULTIPLIERS:
r = cost_results[cost]
lines.append(f"{cost:>5} {ff(r['mean_precision']):>10} {ff(r['mean_recall']):>10} "
f"{r['total_false_action']:>10} {pct(r['mean_false_action_rate']):>10}")
lines.append("")
# ─── 6. Action-Threshold Sweep ────────────────────────────────────────
lines.append("## 6. Expanded Action-Threshold Sweep (30 action groups)")
lines.append("")
threshold_results = run_action_threshold_sweep(six_oof, ACTION_THRESHOLDS)
lines.append(f"{'threshold':>10} {'action_P':>10} {'action_R':>10} {'FA count':>10} {'coverage':>10}")
lines.append("-" * 50)
for t in threshold_results:
lines.append(f"{t['threshold']:>10.2f} {ff(t['action_precision']):>10} {ff(t['action_recall']):>10} "
f"{t['false_action_count']:>10} {pct(t['coverage']):>10}")
lines.append("")
# ─── 7. Voice-Like Stress ─────────────────────────────────────────────
lines.append("## 7. Voice-Like Punctuation Stress Evaluation")
lines.append("")
# Build dev index map
dev_source_ids = {e["source_id"] for e in dev_examples}
dev_indices = [i for i, e in enumerate(examples) if e["source_id"] in dev_source_ids]
stress = run_voice_stress_eval(examples, fold_ids_dev, dev_indices, X_dev, y_dev)
lines.append(f"Total stress-testable pairs: {stress['total_stress_pairs']}")
lines.append("")
lines.append("Sample stress pairs:")
for p in stress["pairs"][:10]:
lines.append(f" {p['source_id']}:")
lines.append(f" original: \"{p['original']}\"")
lines.append(f" stressed: \"{p['stressed']}\"")
lines.append(f" route: {p['route']}")
lines.append("")
lines.append("Impact assessment:")
lines.append(" Removal of punctuation changes:")
lines.append(" - trailing '?' removal eliminates the strongest question signal")
lines.append(" - lowercase normalization removes proper-noun casing cues")
lines.append(" - whitespace collapse has minimal effect on e5 (subword tokenizer)")
lines.append(" Production voice punctuation is unreliable; the model must not depend on it.")
lines.append("")
lines.append(f" Capability-question pairs total in stress set: {stress['cap_q_total']}")
lines.append(f" Capability-question pairs with trailing '?': {stress['cap_q_with_question_mark']}")
lines.append(f" Capability-question pairs with comma only: {stress['cap_q_with_comma_only']}")
lines.append("")
# Show sample affected capability questions
cap_q_samples = [p for p in stress["pairs"] if "capability_question" in p.get("tags", [])]
if cap_q_samples:
lines.append(" Sample affected capability questions:")
for p in cap_q_samples[:5]:
lines.append(f" \"{p['original']}\"\"{p['stressed']}\"")
lines.append("")
# ─── 8. E5 + Structural Features ──────────────────────────────────────
lines.append("## 8. E5 + Tiny Structural Features")
lines.append("")
structural_dev = extract_structural_features(dev_examples)
best_C_struct, struct_results = run_e5_structural_experiment(
X_dev, structural_dev, y_dev, fold_ids_dev, C_VALUES
)
struct_res = struct_results[best_C_struct]
lines.append(f"```text")
lines.append(f"Features: [e5(384) ; IsQuestion(1) ; trailing_?(1) ; prohibition(1)] = 387 dims")
lines.append(f"Macro F1: {ff(struct_res['mean_f1'])}")
lines.append(f"False-action rate: {pct(struct_res['mean_false_action_rate'])}")
lines.append(f"False-action count: {struct_res['total_false_action']}")
lines.append(f"best C: {best_C_struct}")
lines.append(f"```")
lines.append("")
# Compare against pure e5
best_C_e5, e5_results = run_six_way_cv(X_dev, y_dev, fold_ids_dev, dev_examples, C_VALUES)
e5_res = e5_results[best_C_e5]
lines.append("Comparison with pure e5:")
lines.append(f" pure e5: macro_f1={ff(e5_res['mean_f1'])} FA_rate={pct(np.mean([m['false_action']/max(m['test_size'],1) for m in e5_res['fold_metrics']]))}")
fa_rates_struct = [m["false_action_rate"] for m in struct_res["fold_metrics"]]
lines.append(f" e5 + structural bits: macro_f1={ff(struct_res['mean_f1'])} FA_rate={pct(np.mean(fa_rates_struct))}")
lines.append("")
# ─── 9. Four-Way Comparison ───────────────────────────────────────────
lines.append("## 9. Four-Hypothesis Comparison Table")
lines.append("")
# Six-way e5 linear (residual)
six_metrics = compute_six_way_metrics(six_oof)
# Six-way + action threshold (best threshold)
best_thr = max(threshold_results, key=lambda t: t["action_precision"] if t["false_action_count"] <= 50 else 0)
if best_thr["false_action_count"] > 50:
best_thr = max(threshold_results, key=lambda t: t["action_precision"] * t["action_recall"])
# Binary linear probe
bin_metrics = {
"action_precision": bin_res["mean_precision"],
"action_recall": bin_res["mean_recall"],
"false_action_rate": bin_res["total_fp"] / max(sum(m["test_size"] for m in bin_res["fold_metrics"]), 1),
"macro_f1": 0.0, # binary doesn't have macro F1 in the 6-way sense
}
# E5 + structural
struct_fa_rates = [m["false_action_rate"] for m in struct_res["fold_metrics"]]
struct_metrics = {
"action_precision": np.mean([m["action_precision"] for m in struct_res["fold_metrics"]]),
"action_recall": np.mean([m["action_recall"] for m in struct_res["fold_metrics"]]),
"false_action_rate": np.mean(struct_fa_rates),
"macro_f1": struct_res["mean_f1"],
}
lines.append(f"| {'experiment':<35} | {'action_P':>10} | {'action_R':>10} | {'FA rate':>10} | {'macro F1':>10} |")
lines.append(f"| {'-'*35} | {'-'*10} | {'-'*10} | {'-'*10} | {'-'*10} |")
lines.append(f"| {'six-way e5 linear (residual)':<35} | {ff(six_metrics['action_precision']):>10} | {ff(six_metrics['action_recall']):>10} | {pct(six_metrics['false_action_rate']):>10} | {ff(six_metrics['macro_f1']):>10} |")
lines.append(f"| {'six-way + action threshold':<35} | {ff(best_thr['action_precision']):>10} | {ff(best_thr['action_recall']):>10} | {pct(best_thr['false_action_count']/max(safety['total'],1)):>10} | {'':>10} |")
lines.append(f"| {'binary linear action probe':<35} | {ff(bin_metrics['action_precision']):>10} | {ff(bin_metrics['action_recall']):>10} | {pct(bin_metrics['false_action_rate']):>10} | {'':>10} |")
lines.append(f"| {'e5 + tiny structural features':<35} | {ff(struct_metrics['action_precision']):>10} | {ff(struct_metrics['action_recall']):>10} | {pct(struct_metrics['false_action_rate']):>10} | {ff(struct_metrics['macro_f1']):>10} |")
lines.append("")
# ─── 10. Conclusion ────────────────────────────────────────────────────
lines.append("## 10. Conservative Interpretation")
lines.append("")
# Decision logic
if bin_res["mean_pr_auc"] > 0.90:
lines.append("### Binary linear action probe works well (PR-AUC > 0.90)")
lines.append("")
lines.append("The e5 representation is probably adequate for the action/non-action boundary.")
lines.append("The six-way softmax formulation is likely the problem: competition between")
lines.append("six classes creates false actions that a dedicated binary gate would not.")
lines.append("")
lines.append("A two-stage architecture becomes plausible:")
lines.append("```text")
lines.append("e5")
lines.append(" ├─ executable-action gate (binary)")
lines.append(" └─ coarse semantic route head (5-way or 6-way)")
lines.append("```")
elif struct_res["mean_f1"] > e5_res["mean_f1"] + 0.02:
lines.append("### Tiny structural bits fix it")
lines.append("")
lines.append("e5 loses a small amount of syntax/pragmatics that deterministic machinery")
lines.append("can supply cheaply. Prefer this over adding an MLP.")
elif bin_res["mean_pr_auc"] > 0.80:
lines.append("### Binary probe works but not dramatically better")
lines.append("")
lines.append("A nonlinear probe may help, but the improvement ceiling is moderate.")
lines.append("Consider whether the cost of a two-stage architecture is justified.")
else:
lines.append("### Action/capability-question embeddings are nearly indistinguishable")
lines.append("")
lines.append("The representation itself is suspect for this boundary.")
lines.append("Do not claim the boundary is merely nonlinear.")
lines.append("A fundamentally different representation or encoder may be needed.")
lines.append("")
# E5 geometry verdict
if geometry["positive"] and geometry["capability"]:
pos_mean = np.mean([p["similarity"] for p in geometry["positive"]])
cap_mean = np.mean([p["similarity"] for p in geometry["capability"]])
gap = pos_mean - cap_mean
if gap < 0.05:
lines.append("**E5 geometry verdict**: action seeds and capability questions are nearly")
lines.append("indistinguishable in embedding space. The representation intentionally")
lines.append("maps pragmatically different but semantically similar sentences close together.")
lines.append("This is a fundamental limitation of the frozen e5 representation for this boundary.")
else:
lines.append(f"**E5 geometry verdict**: moderate separation ({ff(gap)}) exists between")
lines.append("action seeds and capability questions. The representation preserves some signal.")
lines.append("")
# Softmax formulation verdict
lines.append(f"**Softmax formulation verdict**: the six-way head produces {safety['false_action_count']} false actions")
lines.append(f"at {pct(safety['false_action_rate'])} rate. The binary probe produces {bin_res['total_fp']} false actions.")
ratio = bin_res["total_fp"] / max(safety["false_action_count"], 1)
lines.append(f"The binary probe has {ratio:.1f}x the false-action count of the six-way head.")
if ratio < 0.8:
lines.append("This confirms the six-way softmax competition is a significant contributor.")
elif ratio > 1.2:
lines.append("Surprisingly, the binary probe does not reduce false actions. The issue is deeper than softmax competition.")
else:
lines.append("The binary probe and six-way head produce comparable false-action counts.")
lines.append("")
# ─── 11. Commit hash ──────────────────────────────────────────────────
lines.append("## 11. Commit hash for diagnostic tooling")
lines.append("")
lines.append("(to be filled after commit)")
lines.append("")
report = "\n".join(lines)
report_path = "/home/kami/apps/Maven/docs/evals/2026-09-07-slice16-diagnostic.md"
with open(report_path, "w") as f:
f.write(report)
print(f"\nReport written to {report_path}")
# Print summary
print("\n" + "=" * 60)
print(" SLICE 16 DIAGNOSTIC SUMMARY")
print("=" * 60)
print(f" Six-way (residual): P={ff(safety['action_precision'])} R={ff(safety['action_recall'])} FA={safety['false_action_count']} ({pct(safety['false_action_rate'])})")
print(f" Binary probe: P={ff(bin_res['mean_precision'])} R={ff(bin_res['mean_recall'])} FP={bin_res['total_fp']} PR-AUC={ff(bin_res['mean_pr_auc'])}")
print(f" E5+structural: F1={ff(struct_res['mean_f1'])} FA={struct_res['total_false_action']}")
pos_mean = ff(np.mean([p['similarity'] for p in geometry['positive']])) if geometry['positive'] else 'N/A'
cap_mean = ff(np.mean([p['similarity'] for p in geometry['capability']])) if geometry['capability'] else 'N/A'
print(f" E5 geometry: pos={pos_mean} cap={cap_mean}")
print(f" False-action families: {n_families}")
if __name__ == "__main__":
main()