774c0660b0
Experiments: - Residual-only OOF safety metrics - False-action decomposition by semantic family (7 families, 24 split groups) - Paired e5 geometry (gap=0.010 between action and capability question) - Binary action probe (PR-AUC=0.707, ROC-AUC=0.863) - Cost-sensitive linear classification - Expanded action-threshold sweep (7 thresholds) - Voice-like punctuation stress evaluation - e5 + tiny structural features - Four-hypothesis comparison table Verdict: e5 representation is the primary bottleneck. Action seeds and capability questions are nearly indistinguishable in embedding space.
1138 lines
49 KiB
Python
1138 lines
49 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Slice 16 Diagnostic: Action/Non-Action Boundary Analysis
|
|
========================================================
|
|
|
|
Determines exactly why the action/non-action boundary still fails,
|
|
before any nonlinear model or production gate.
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
import sys
|
|
import warnings
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
import numpy as np
|
|
from sklearn.exceptions import ConvergenceWarning
|
|
from sklearn.linear_model import LogisticRegression
|
|
from sklearn.metrics import (
|
|
accuracy_score,
|
|
average_precision_score,
|
|
brier_score_loss,
|
|
confusion_matrix,
|
|
f1_score,
|
|
precision_recall_curve,
|
|
precision_recall_fscore_support,
|
|
roc_auc_score,
|
|
)
|
|
from sklearn.preprocessing import LabelEncoder
|
|
|
|
warnings.filterwarnings("ignore", category=ConvergenceWarning)
|
|
|
|
EMBEDDING_PATH = "/tmp/mvn-experiment/embeddings.json"
|
|
|
|
ROUTES = ["action", "conversation", "knowledge", "memory_write", "system", "uncertain"]
|
|
ROUTE_IDX = {r: i for i, r in enumerate(ROUTES)}
|
|
|
|
C_VALUES = [0.01, 0.1, 1.0, 10.0, 100.0]
|
|
ACTION_THRESHOLDS = [0.50, 0.60, 0.70, 0.80, 0.85, 0.90, 0.95]
|
|
COST_MULTIPLIERS = [1, 2, 4, 8, 16]
|
|
|
|
|
|
# ─── Data Loading ───────────────────────────────────────────────────────────
|
|
|
|
def load_data():
|
|
with open(EMBEDDING_PATH) as f:
|
|
data = json.load(f)
|
|
meta = data["meta"]
|
|
examples = data["examples"]
|
|
return meta, examples
|
|
|
|
|
|
def filter_dev_pool(examples):
|
|
return [e for e in examples if e["dev_pool"]]
|
|
|
|
|
|
def filter_residual(examples):
|
|
return [e for e in examples if not e["fast_path_resolved"]]
|
|
|
|
|
|
def extract_Xy(examples):
|
|
X = np.array([e["embedding"] for e in examples])
|
|
y = np.array([e["route"] for e in examples])
|
|
return X, y
|
|
|
|
|
|
def get_fold_groups(examples):
|
|
return np.array([e["cv_fold"] for e in examples])
|
|
|
|
|
|
# ─── Helpers ─────────────────────────────────────────────────────────────────
|
|
|
|
def pct(v, d=1):
|
|
return f"{100*v:.{d}f}%"
|
|
|
|
|
|
def ff(v, d=3):
|
|
return f"{v:.{d}f}"
|
|
|
|
|
|
# ─── Section 1: Residual Safety Metrics ─────────────────────────────────────
|
|
|
|
def run_six_way_cv(X, y, fold_ids, examples, C_values):
|
|
"""Run 6-way grouped CV and return OOF predictions for best C."""
|
|
unique_folds = sorted(set(fold_ids))
|
|
results_by_C = {}
|
|
|
|
for C in C_values:
|
|
oof_rows = []
|
|
fold_metrics = []
|
|
|
|
for test_fold in unique_folds:
|
|
train_mask = fold_ids != test_fold
|
|
test_mask = fold_ids == test_fold
|
|
|
|
X_train, y_train = X[train_mask], y[train_mask]
|
|
X_test, y_test = X[test_mask], y[test_mask]
|
|
|
|
model = LogisticRegression(C=C, max_iter=2000, solver="lbfgs", random_state=42)
|
|
model.fit(X_train, y_train)
|
|
|
|
y_pred = model.predict(X_test)
|
|
y_proba = model.predict_proba(X_test)
|
|
classes = model.classes_
|
|
|
|
acc = accuracy_score(y_test, y_pred)
|
|
macro_f1 = f1_score(y_test, y_pred, average="macro", zero_division=0)
|
|
false_action = sum(1 for t, p in zip(y_test, y_pred) if t != "action" and p == "action")
|
|
|
|
fold_metrics.append({
|
|
"fold": test_fold, "accuracy": acc, "macro_f1": macro_f1,
|
|
"false_action": false_action, "test_size": len(X_test),
|
|
})
|
|
|
|
test_indices = np.where(test_mask)[0]
|
|
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
|
|
meta = examples[test_indices[i]]
|
|
proba_dict = {cls: float(y_proba[i][j]) for j, cls in enumerate(classes)}
|
|
oof_rows.append({
|
|
"source_id": meta["source_id"],
|
|
"fold": test_fold,
|
|
"true": true,
|
|
"predicted": pred,
|
|
"correct": true == pred,
|
|
"proba": proba_dict,
|
|
"text": meta["text"],
|
|
"tags": meta.get("tags", []),
|
|
"split_group": meta.get("split_group", ""),
|
|
"family_id": meta.get("family_id", ""),
|
|
})
|
|
|
|
mean_f1 = np.mean([m["macro_f1"] for m in fold_metrics])
|
|
results_by_C[C] = {"mean_f1": mean_f1, "fold_metrics": fold_metrics, "oof": oof_rows}
|
|
|
|
best_C = max(results_by_C, key=lambda c: results_by_C[c]["mean_f1"])
|
|
return best_C, results_by_C
|
|
|
|
|
|
def compute_action_metrics(oof_rows):
|
|
"""Compute action-specific safety metrics from OOF predictions."""
|
|
y_true = np.array([r["true"] for r in oof_rows])
|
|
y_pred = np.array([r["predicted"] for r in oof_rows])
|
|
|
|
action_tp = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p == "action")
|
|
action_fp = sum(1 for t, p in zip(y_true, y_pred) if t != "action" and p == "action")
|
|
action_fn = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p != "action")
|
|
|
|
action_precision = action_tp / max(action_tp + action_fp, 1)
|
|
action_recall = action_tp / max(action_tp + action_fn, 1)
|
|
false_action = action_fp
|
|
false_action_rate = false_action / max(len(y_true), 1)
|
|
|
|
# Uncertain F1
|
|
unc_tp = sum(1 for t, p in zip(y_true, y_pred) if t == "uncertain" and p == "uncertain")
|
|
unc_fp = sum(1 for t, p in zip(y_true, y_pred) if t != "uncertain" and p == "uncertain")
|
|
unc_fn = sum(1 for t, p in zip(y_true, y_pred) if t == "uncertain" and p != "uncertain")
|
|
unc_prec = unc_tp / max(unc_tp + unc_fp, 1)
|
|
unc_rec = unc_tp / max(unc_tp + unc_fn, 1)
|
|
unc_f1 = 2 * unc_prec * unc_rec / max(unc_prec + unc_rec, 1e-9)
|
|
|
|
return {
|
|
"action_precision": action_precision,
|
|
"action_recall": action_recall,
|
|
"false_action_count": false_action,
|
|
"false_action_rate": false_action_rate,
|
|
"uncertain_f1": unc_f1,
|
|
"total": len(y_true),
|
|
}
|
|
|
|
|
|
# ─── Section 2: False-Action Decomposition ─────────────────────────────────
|
|
|
|
# Semantic family classification from source_id prefix and tags
|
|
def classify_semantic_family(row):
|
|
sid = row["source_id"]
|
|
tags = row.get("tags", [])
|
|
text = row["text"]
|
|
|
|
if "capability_question" in tags:
|
|
return "capability_question"
|
|
if "question" in tags:
|
|
return "question"
|
|
if "negation" in tags:
|
|
return "negation"
|
|
if "reported_speech" in tags:
|
|
return "reported_speech"
|
|
if "quotation" in tags:
|
|
return "quotation"
|
|
if "hypothetical" in tags:
|
|
return "hypothetical"
|
|
|
|
# Classify by source_id prefix and text patterns
|
|
if sid.startswith("mw-note-task"):
|
|
return "memory_write"
|
|
if sid.startswith("mw-free-remember"):
|
|
return "memory_write"
|
|
if sid.startswith("mw-fact"):
|
|
return "memory_write"
|
|
if sid.startswith("mw-note-homelab"):
|
|
return "memory_write"
|
|
if sid.startswith("mw-note-idea"):
|
|
return "memory_write"
|
|
if sid.startswith("kq-cap"):
|
|
return "capability_question"
|
|
if sid.startswith("kq-world"):
|
|
return "knowledge_general"
|
|
if sid.startswith("kq-homelab"):
|
|
return "knowledge_general"
|
|
if sid.startswith("kq-task"):
|
|
return "knowledge_general"
|
|
if sid.startswith("kq-cal"):
|
|
return "knowledge_general"
|
|
if sid.startswith("kq-deadline"):
|
|
return "knowledge_general"
|
|
if sid.startswith("sys-"):
|
|
return "system"
|
|
if sid.startswith("conv-"):
|
|
return "conversation"
|
|
if sid.startswith("unc-"):
|
|
return "uncertain"
|
|
|
|
# Text-based fallback
|
|
q_markers = ["что такое", "кто такой", "что значит", "как дела"]
|
|
if any(m in text.lower() for m in q_markers):
|
|
return "knowledge_general"
|
|
if "?" in text:
|
|
return "question"
|
|
|
|
return "other"
|
|
|
|
|
|
def decompose_false_actions(oof_rows):
|
|
"""Decompose false actions by semantic family and split group."""
|
|
false_actions = [r for r in oof_rows if r["true"] != "action" and r["predicted"] == "action"]
|
|
|
|
by_family = defaultdict(list)
|
|
by_split_group = defaultdict(list)
|
|
by_fold = defaultdict(list)
|
|
|
|
for r in false_actions:
|
|
family = classify_semantic_family(r)
|
|
by_family[family].append(r)
|
|
by_split_group[r["split_group"]].append(r)
|
|
by_fold[r["fold"]].append(r)
|
|
|
|
return {
|
|
"total": len(false_actions),
|
|
"by_family": dict(by_family),
|
|
"by_split_group": dict(by_split_group),
|
|
"by_fold": dict(by_fold),
|
|
"details": false_actions,
|
|
}
|
|
|
|
|
|
# ─── Section 3: E5 Geometry ────────────────────────────────────────────────
|
|
|
|
def cosine_sim(a, b):
|
|
a = np.array(a, dtype=np.float64)
|
|
b = np.array(b, dtype=np.float64)
|
|
na = np.linalg.norm(a)
|
|
nb = np.linalg.norm(b)
|
|
if na == 0 or nb == 0:
|
|
return 0.0
|
|
return float(np.dot(a, b) / (na * nb))
|
|
|
|
|
|
def inspect_e5_geometry(examples):
|
|
"""For every action seed, compare embedding similarity to:
|
|
- positive action variants (same group, action route)
|
|
- nearest capability question (by embedding distance)
|
|
- nearest other negative contrast (knowledge/uncertain, by embedding distance)
|
|
"""
|
|
emb_lookup = {e["source_id"]: np.array(e["embedding"]) for e in examples}
|
|
|
|
# Action seeds: corpus_factory_v2 action examples without contrastive tags
|
|
action_seeds = [e for e in examples
|
|
if e["route"] == "action"
|
|
and e["source"] == "corpus_factory_v2"
|
|
and e.get("dev_pool", False)
|
|
and not any(t in e.get("tags", []) for t in
|
|
["negation", "question", "reported_speech",
|
|
"quotation", "hypothetical", "capability_question"])]
|
|
|
|
# Pool of capability questions (dev pool)
|
|
cap_qs = [e for e in examples
|
|
if "capability_question" in e.get("tags", [])
|
|
and e.get("dev_pool", False)]
|
|
|
|
# Pool of other negatives (knowledge/uncertain routes, dev pool, not capability_question)
|
|
other_neg = [e for e in examples
|
|
if e["route"] in ("knowledge", "uncertain")
|
|
and "capability_question" not in e.get("tags", [])
|
|
and e.get("dev_pool", False)
|
|
and e["route"] != "action"]
|
|
|
|
# Group action seeds by split_group for positive pairs
|
|
by_group = defaultdict(list)
|
|
for e in examples:
|
|
by_group[e["split_group"]].append(e)
|
|
|
|
positive_pairs = []
|
|
capability_pairs = []
|
|
other_negative_pairs = []
|
|
|
|
for seed in action_seeds:
|
|
seed_emb = emb_lookup.get(seed["source_id"])
|
|
if seed_emb is None:
|
|
continue
|
|
|
|
# Positive: same-group action variants
|
|
groupmates = by_group.get(seed["split_group"], [])
|
|
for c in groupmates:
|
|
if c["source_id"] == seed["source_id"] or c["route"] != "action":
|
|
continue
|
|
c_emb = emb_lookup.get(c["source_id"])
|
|
if c_emb is None:
|
|
continue
|
|
sim = cosine_sim(seed_emb, c_emb)
|
|
positive_pairs.append({
|
|
"seed_id": seed["source_id"], "contrast_id": c["source_id"],
|
|
"similarity": sim, "text": c["text"],
|
|
})
|
|
|
|
# Nearest capability question
|
|
best_cap, best_cap_sim = None, -1.0
|
|
for cap in cap_qs:
|
|
cap_emb = emb_lookup.get(cap["source_id"])
|
|
if cap_emb is None:
|
|
continue
|
|
sim = cosine_sim(seed_emb, cap_emb)
|
|
if sim > best_cap_sim:
|
|
best_cap_sim = sim
|
|
best_cap = cap
|
|
if best_cap:
|
|
capability_pairs.append({
|
|
"seed_id": seed["source_id"], "contrast_id": best_cap["source_id"],
|
|
"similarity": best_cap_sim, "text": best_cap["text"],
|
|
})
|
|
|
|
# Nearest other negative
|
|
best_neg, best_neg_sim = None, -1.0
|
|
for neg in other_neg:
|
|
neg_emb = emb_lookup.get(neg["source_id"])
|
|
if neg_emb is None:
|
|
continue
|
|
sim = cosine_sim(seed_emb, neg_emb)
|
|
if sim > best_neg_sim:
|
|
best_neg_sim = sim
|
|
best_neg = neg
|
|
if best_neg:
|
|
other_negative_pairs.append({
|
|
"seed_id": seed["source_id"], "contrast_id": best_neg["source_id"],
|
|
"similarity": best_neg_sim, "text": best_neg["text"],
|
|
})
|
|
|
|
return {
|
|
"positive": positive_pairs,
|
|
"capability": capability_pairs,
|
|
"other_negative": other_negative_pairs,
|
|
}
|
|
|
|
|
|
# ─── Section 4: Binary Action Probe ────────────────────────────────────────
|
|
|
|
def run_binary_action_probe(X, y, fold_ids, C_values):
|
|
"""Binary action vs not-action logistic probe with grouped CV."""
|
|
y_binary = np.array(["action" if t == "action" else "not_action" for t in y])
|
|
|
|
unique_folds = sorted(set(fold_ids))
|
|
results_by_C = {}
|
|
|
|
for C in C_values:
|
|
oof_rows = []
|
|
fold_metrics = []
|
|
|
|
for test_fold in unique_folds:
|
|
train_mask = fold_ids != test_fold
|
|
test_mask = fold_ids == test_fold
|
|
|
|
X_train, y_train = X[train_mask], y_binary[train_mask]
|
|
X_test, y_test = X[test_mask], y_binary[test_mask]
|
|
|
|
model = LogisticRegression(C=C, max_iter=2000, solver="lbfgs", random_state=42)
|
|
model.fit(X_train, y_train)
|
|
|
|
y_pred = model.predict(X_test)
|
|
y_proba = model.predict_proba(X_test)
|
|
classes = model.classes_
|
|
|
|
action_idx = list(classes).index("action")
|
|
action_proba = y_proba[:, action_idx]
|
|
|
|
acc = accuracy_score(y_test, y_pred)
|
|
false_pos = sum(1 for t, p in zip(y_test, y_pred) if t == "not_action" and p == "action")
|
|
false_neg = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p == "not_action")
|
|
|
|
# ROC-AUC and PR-AUC
|
|
y_test_binary = np.array([1 if t == "action" else 0 for t in y_test])
|
|
if len(np.unique(y_test_binary)) > 1:
|
|
roc = roc_auc_score(y_test_binary, action_proba)
|
|
pr_auc = average_precision_score(y_test_binary, action_proba)
|
|
else:
|
|
roc = 0.0
|
|
pr_auc = 0.0
|
|
|
|
prec, rec, f1, sup = precision_recall_fscore_support(
|
|
y_test, y_pred, labels=["action", "not_action"], zero_division=0
|
|
)
|
|
action_prec = prec[0]
|
|
action_rec = rec[0]
|
|
|
|
fold_metrics.append({
|
|
"fold": test_fold, "accuracy": acc, "roc_auc": roc, "pr_auc": pr_auc,
|
|
"action_precision": action_prec, "action_recall": action_rec,
|
|
"false_pos": false_pos, "false_neg": false_neg,
|
|
"test_size": len(X_test),
|
|
})
|
|
|
|
test_indices = np.where(test_mask)[0]
|
|
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
|
|
oof_rows.append({
|
|
"fold": test_fold, "true": true, "predicted": pred,
|
|
"action_proba": float(action_proba[i]),
|
|
"correct": true == pred,
|
|
})
|
|
|
|
mean_roc = np.mean([m["roc_auc"] for m in fold_metrics])
|
|
mean_pr = np.mean([m["pr_auc"] for m in fold_metrics])
|
|
total_fp = sum(m["false_pos"] for m in fold_metrics)
|
|
total_fn = sum(m["false_neg"] for m in fold_metrics)
|
|
mean_prec = np.mean([m["action_precision"] for m in fold_metrics])
|
|
mean_rec = np.mean([m["action_recall"] for m in fold_metrics])
|
|
|
|
results_by_C[C] = {
|
|
"mean_roc_auc": mean_roc, "mean_pr_auc": mean_pr,
|
|
"total_fp": total_fp, "total_fn": total_fn,
|
|
"mean_precision": mean_prec, "mean_recall": mean_rec,
|
|
"fold_metrics": fold_metrics, "oof": oof_rows,
|
|
}
|
|
|
|
best_C = max(results_by_C, key=lambda c: results_by_C[c]["mean_pr_auc"])
|
|
return best_C, results_by_C
|
|
|
|
|
|
# ─── Section 5: Cost-Sensitive Classification ──────────────────────────────
|
|
|
|
def run_cost_sensitive(X, y, fold_ids, cost_multipliers):
|
|
"""Evaluate cost-sensitive 6-way classification with grouped CV."""
|
|
unique_folds = sorted(set(fold_ids))
|
|
results = {}
|
|
|
|
for cost in cost_multipliers:
|
|
oof_rows = []
|
|
fold_metrics = []
|
|
|
|
for test_fold in unique_folds:
|
|
train_mask = fold_ids != test_fold
|
|
test_mask = fold_ids == test_fold
|
|
|
|
X_train, y_train = X[train_mask], y[train_mask]
|
|
X_test, y_test = X[test_mask], y[test_mask]
|
|
|
|
# Compute class weights: action gets cost multiplier, others get 1
|
|
classes = sorted(set(y_train))
|
|
class_weights = {}
|
|
for c in classes:
|
|
if c == "action":
|
|
class_weights[c] = cost
|
|
else:
|
|
class_weights[c] = 1.0
|
|
# Normalize so weights sum to n_classes
|
|
total_w = sum(class_weights.values())
|
|
for c in class_weights:
|
|
class_weights[c] *= len(classes) / total_w
|
|
|
|
model = LogisticRegression(
|
|
C=10.0, max_iter=2000, solver="lbfgs",
|
|
class_weight=class_weights, random_state=42,
|
|
)
|
|
model.fit(X_train, y_train)
|
|
|
|
y_pred = model.predict(X_test)
|
|
|
|
action_tp = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p == "action")
|
|
action_fp = sum(1 for t, p in zip(y_test, y_pred) if t != "action" and p == "action")
|
|
action_fn = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p != "action")
|
|
false_action = action_fp
|
|
|
|
action_prec = action_tp / max(action_tp + action_fp, 1)
|
|
action_rec = action_tp / max(action_tp + action_fn, 1)
|
|
false_action_rate = false_action / max(len(y_test), 1)
|
|
|
|
fold_metrics.append({
|
|
"fold": test_fold, "action_precision": action_prec,
|
|
"action_recall": action_rec, "false_action": false_action,
|
|
"false_action_rate": false_action_rate, "test_size": len(X_test),
|
|
})
|
|
|
|
test_indices = np.where(test_mask)[0]
|
|
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
|
|
oof_rows.append({
|
|
"fold": test_fold, "true": true, "predicted": pred,
|
|
"correct": true == pred,
|
|
})
|
|
|
|
mean_prec = np.mean([m["action_precision"] for m in fold_metrics])
|
|
mean_rec = np.mean([m["action_recall"] for m in fold_metrics])
|
|
total_fa = sum(m["false_action"] for m in fold_metrics)
|
|
mean_fa_rate = np.mean([m["false_action_rate"] for m in fold_metrics])
|
|
|
|
results[cost] = {
|
|
"mean_precision": mean_prec, "mean_recall": mean_rec,
|
|
"total_false_action": total_fa, "mean_false_action_rate": mean_fa_rate,
|
|
}
|
|
|
|
return results
|
|
|
|
|
|
# ─── Section 6: Action Threshold Sweep ─────────────────────────────────────
|
|
|
|
def run_action_threshold_sweep(oof_rows, thresholds):
|
|
"""Evaluate action-specific threshold on the already-trained OOF predictions."""
|
|
results = []
|
|
|
|
for thr in thresholds:
|
|
action_pred = []
|
|
for r in oof_rows:
|
|
p = r["predicted"]
|
|
proba = r["proba"].get("action", 0.0)
|
|
if p == "action" and proba < thr:
|
|
sorted_routes = sorted(r["proba"].items(), key=lambda x: -x[1])
|
|
for route, _ in sorted_routes:
|
|
if route != "action":
|
|
p = route
|
|
break
|
|
action_pred.append(p)
|
|
|
|
y_true = np.array([r["true"] for r in oof_rows])
|
|
y_pred = np.array(action_pred)
|
|
|
|
action_tp = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p == "action")
|
|
action_fp = sum(1 for t, p in zip(y_true, y_pred) if t != "action" and p == "action")
|
|
action_fn = sum(1 for t, p in zip(y_true, y_pred) if t == "action" and p != "action")
|
|
false_action = action_fp
|
|
|
|
action_prec = action_tp / max(action_tp + action_fp, 1)
|
|
action_rec = action_tp / max(action_tp + action_fn, 1)
|
|
|
|
# Coverage: fraction of examples where model is confident enough
|
|
# (not demoted)
|
|
n_demoted = sum(1 for i, r in enumerate(oof_rows)
|
|
if r["predicted"] == "action" and r["proba"].get("action", 0) < thr)
|
|
coverage = (len(y_true) - n_demoted) / max(len(y_true), 1)
|
|
|
|
results.append({
|
|
"threshold": thr,
|
|
"action_precision": action_prec,
|
|
"action_recall": action_rec,
|
|
"false_action_count": false_action,
|
|
"coverage": coverage,
|
|
})
|
|
|
|
return results
|
|
|
|
|
|
# ─── Section 7: Voice-Like Stress ──────────────────────────────────────────
|
|
|
|
def apply_voice_stress(text):
|
|
"""Apply representation changes typical of STT."""
|
|
# Remove final punctuation
|
|
t = re.sub(r'[?.!,;:]+$', '', text.strip())
|
|
# Remove all punctuation (safe for Russian)
|
|
t = re.sub(r'[^\w\s]', '', t)
|
|
# Lowercase
|
|
t = t.lower()
|
|
# Collapse whitespace
|
|
t = re.sub(r'\s+', ' ', t).strip()
|
|
return t
|
|
|
|
|
|
def run_voice_stress_eval(examples, fold_ids_all, dev_indices, X_all, y_all):
|
|
"""Score voice-stressed variants without retraining."""
|
|
dev_examples = [examples[i] for i in dev_indices]
|
|
|
|
stress_pairs = []
|
|
for e in dev_examples:
|
|
original_text = e["text"]
|
|
stressed_text = apply_voice_stress(original_text)
|
|
if stressed_text != original_text:
|
|
stress_pairs.append({
|
|
"source_id": e["source_id"],
|
|
"original": original_text,
|
|
"stressed": stressed_text,
|
|
"route": e["route"],
|
|
"fold": e["cv_fold"],
|
|
"tags": e.get("tags", []),
|
|
})
|
|
|
|
# Count capability-question pairs specifically
|
|
cap_q_affected = [p for p in stress_pairs
|
|
if "capability_question" in p["tags"]]
|
|
cap_q_with_q = [p for p in cap_q_affected if "?" in p["original"]]
|
|
cap_q_comma = [p for p in cap_q_affected if "," in p["original"] and "?" not in p["original"]]
|
|
|
|
return {
|
|
"total_stress_pairs": len(stress_pairs),
|
|
"cap_q_total": len(cap_q_affected),
|
|
"cap_q_with_question_mark": len(cap_q_with_q),
|
|
"cap_q_with_comma_only": len(cap_q_comma),
|
|
"pairs": stress_pairs[:20],
|
|
}
|
|
|
|
|
|
# ─── Section 8: E5 + Structural Features ───────────────────────────────────
|
|
|
|
def is_question_shaped(text):
|
|
"""Boolean feature: does the text look like a question?"""
|
|
q_starters = [
|
|
"что ", "кто ", "как ", "где ", "когда ", "почему ", "зачем ",
|
|
"сколько ", "можно ли ", "нужно ли ", "есть ли ", "хватает ли ",
|
|
"какой ", "какая ", "какие ", "чей ", "чья ",
|
|
]
|
|
t = text.lower()
|
|
return any(t.startswith(s) for s in q_starters)
|
|
|
|
|
|
def has_trailing_question_mark(text):
|
|
return text.rstrip().endswith("?")
|
|
|
|
|
|
def has_prohibition(text):
|
|
"""Check for negation patterns (prohibition)."""
|
|
t = text.lower()
|
|
return t.startswith("не ") or t.startswith("ни ")
|
|
|
|
|
|
def extract_structural_features(examples):
|
|
"""Extract [e5_embedding ; IsQuestion ; trailing_? ; prohibition] for each example."""
|
|
structural = []
|
|
for e in examples:
|
|
text = e["text"]
|
|
features = [
|
|
1.0 if is_question_shaped(text) else 0.0,
|
|
1.0 if has_trailing_question_mark(text) else 0.0,
|
|
1.0 if has_prohibition(text) else 0.0,
|
|
]
|
|
structural.append(features)
|
|
return np.array(structural)
|
|
|
|
|
|
def run_e5_structural_experiment(X_e5, structural, y, fold_ids, C_values):
|
|
"""Run grouped CV with [e5 ; structural bits]."""
|
|
X_aug = np.hstack([X_e5, structural])
|
|
|
|
unique_folds = sorted(set(fold_ids))
|
|
results_by_C = {}
|
|
|
|
for C in C_values:
|
|
oof_rows = []
|
|
fold_metrics = []
|
|
|
|
for test_fold in unique_folds:
|
|
train_mask = fold_ids != test_fold
|
|
test_mask = fold_ids == test_fold
|
|
|
|
X_train, y_train = X_aug[train_mask], y[train_mask]
|
|
X_test, y_test = X_aug[test_mask], y[test_mask]
|
|
|
|
model = LogisticRegression(C=C, max_iter=2000, solver="lbfgs", random_state=42)
|
|
model.fit(X_train, y_train)
|
|
|
|
y_pred = model.predict(X_test)
|
|
y_proba = model.predict_proba(X_test)
|
|
classes = model.classes_
|
|
|
|
acc = accuracy_score(y_test, y_pred)
|
|
macro_f1 = f1_score(y_test, y_pred, average="macro", zero_division=0)
|
|
|
|
action_tp = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p == "action")
|
|
action_fp = sum(1 for t, p in zip(y_test, y_pred) if t != "action" and p == "action")
|
|
action_fn = sum(1 for t, p in zip(y_test, y_pred) if t == "action" and p != "action")
|
|
false_action = action_fp
|
|
|
|
action_prec = action_tp / max(action_tp + action_fp, 1)
|
|
action_rec = action_tp / max(action_tp + action_fn, 1)
|
|
false_action_rate = false_action / max(len(y_test), 1)
|
|
|
|
fold_metrics.append({
|
|
"fold": test_fold, "accuracy": acc, "macro_f1": macro_f1,
|
|
"action_precision": action_prec, "action_recall": action_rec,
|
|
"false_action": false_action, "false_action_rate": false_action_rate,
|
|
"test_size": len(X_test),
|
|
})
|
|
|
|
test_indices = np.where(test_mask)[0]
|
|
for i, (true, pred) in enumerate(zip(y_test, y_pred)):
|
|
proba_dict = {cls: float(y_proba[i][j]) for j, cls in enumerate(classes)}
|
|
oof_rows.append({
|
|
"fold": test_fold, "true": true, "predicted": pred,
|
|
"correct": true == pred, "proba": proba_dict,
|
|
})
|
|
|
|
mean_f1 = np.mean([m["macro_f1"] for m in fold_metrics])
|
|
total_fa = sum(m["false_action"] for m in fold_metrics)
|
|
mean_fa_rate = np.mean([m["false_action_rate"] for m in fold_metrics])
|
|
|
|
results_by_C[C] = {
|
|
"mean_f1": mean_f1, "total_false_action": total_fa,
|
|
"mean_false_action_rate": mean_fa_rate,
|
|
"fold_metrics": fold_metrics, "oof": oof_rows,
|
|
}
|
|
|
|
best_C = max(results_by_C, key=lambda c: results_by_C[c]["mean_f1"])
|
|
return best_C, results_by_C
|
|
|
|
|
|
# ─── Section 9: Four-Way Comparison ────────────────────────────────────────
|
|
|
|
def compute_six_way_metrics(oof_rows):
|
|
"""Compute macro F1 for 6-way model from OOF rows."""
|
|
y_true = np.array([r["true"] for r in oof_rows])
|
|
y_pred = np.array([r["predicted"] for r in oof_rows])
|
|
return compute_action_metrics(oof_rows) | {
|
|
"macro_f1": float(f1_score(y_true, y_pred, average="macro", zero_division=0)),
|
|
}
|
|
|
|
|
|
# ─── Report Generation ──────────────────────────────────────────────────────
|
|
|
|
def main():
|
|
print("Loading data...")
|
|
meta, examples = load_data()
|
|
dev_examples = filter_dev_pool(examples)
|
|
dev_residual = filter_residual(dev_examples)
|
|
|
|
X_dev, y_dev = extract_Xy(dev_examples)
|
|
fold_ids_dev = get_fold_groups(dev_examples)
|
|
|
|
X_res, y_res = extract_Xy(dev_residual)
|
|
fold_ids_res = get_fold_groups(dev_residual)
|
|
|
|
print(f"Dev pool: {len(dev_examples)} examples, residual: {len(dev_residual)}")
|
|
|
|
lines = []
|
|
lines.append("# Slice 16 Diagnostic: Action/Non-Action Boundary Analysis")
|
|
lines.append("")
|
|
|
|
# ─── 0. Slice 15 hashes ────────────────────────────────────────────────
|
|
lines.append("## 0. Slice 15 Commit Hashes")
|
|
lines.append("")
|
|
lines.append("```text")
|
|
lines.append("semantic seed type: 07bfcea")
|
|
lines.append("merge-corpus tool: 6397ea1")
|
|
lines.append("e5 embedding cache: f8ec77d")
|
|
lines.append("sklearn experiment: 9df2239")
|
|
lines.append("corpus-factory: 4bb7555")
|
|
lines.append("expanded corpus: 87411e5")
|
|
lines.append("contract tests: ad3f2d6")
|
|
lines.append("experiment reports: cbac8b9")
|
|
lines.append("")
|
|
lines.append(f"development corpus v2 hash: {meta.get('dataset_hash', 'b27fd48f478ca477')}")
|
|
lines.append(f"original frozen holdout hash: ad297fbdbbea704b (byte-identical, uninspected)")
|
|
lines.append("```")
|
|
lines.append("")
|
|
|
|
# ─── 1. Residual Safety Metrics ────────────────────────────────────────
|
|
lines.append("## 1. Router-Residual OOF Safety Metrics")
|
|
lines.append("")
|
|
|
|
best_C_six, six_way_results = run_six_way_cv(X_res, y_res, fold_ids_res, dev_residual, C_VALUES)
|
|
six_oof = six_way_results[best_C_six]["oof"]
|
|
safety = compute_action_metrics(six_oof)
|
|
|
|
lines.append("```text")
|
|
lines.append(f"action precision: {ff(safety['action_precision'])}")
|
|
lines.append(f"action recall: {ff(safety['action_recall'])}")
|
|
lines.append(f"false-action count: {safety['false_action_count']}")
|
|
lines.append(f"false-action rate: {pct(safety['false_action_rate'])}")
|
|
lines.append(f"uncertain F1: {ff(safety['uncertain_f1'])}")
|
|
lines.append(f"total examples: {safety['total']}")
|
|
lines.append(f"best C: {best_C_six}")
|
|
lines.append("```")
|
|
lines.append("")
|
|
|
|
# ─── 2. False-Action Decomposition ─────────────────────────────────────
|
|
lines.append("## 2. False-Action Decomposition by Semantic Family")
|
|
lines.append("")
|
|
|
|
decomposed = decompose_false_actions(six_oof)
|
|
|
|
lines.append(f"Total false actions: {decomposed['total']}")
|
|
lines.append("")
|
|
lines.append("### By semantic family")
|
|
lines.append("")
|
|
lines.append(f"{'family':<25} {'count':>6} {'rate':>8}")
|
|
lines.append("-" * 40)
|
|
for family, rows in sorted(decomposed["by_family"].items(), key=lambda x: -len(x[1])):
|
|
rate = len(rows) / max(decomposed["total"], 1)
|
|
lines.append(f"{family:<25} {len(rows):>6} {pct(rate):>8}")
|
|
lines.append("")
|
|
|
|
lines.append("### By held-out split group (top 20)")
|
|
lines.append("")
|
|
lines.append(f"{'split_group':<30} {'count':>6}")
|
|
lines.append("-" * 37)
|
|
for sg, rows in sorted(decomposed["by_split_group"].items(), key=lambda x: -len(x[1]))[:20]:
|
|
lines.append(f"{sg:<30} {len(rows):>6}")
|
|
lines.append("")
|
|
|
|
lines.append("### By fold")
|
|
lines.append("")
|
|
lines.append(f"{'fold':>5} {'count':>6}")
|
|
lines.append("-" * 12)
|
|
for fold, rows in sorted(decomposed["by_fold"].items()):
|
|
lines.append(f"{fold:>5} {len(rows):>6}")
|
|
lines.append("")
|
|
|
|
# Diagnosis
|
|
lines.append("### Diagnosis")
|
|
lines.append("")
|
|
n_families = len(decomposed["by_family"])
|
|
n_groups = len(decomposed["by_split_group"])
|
|
max_family = max(decomposed["by_family"].items(), key=lambda x: len(x[1]))
|
|
fold_counts = [len(v) for v in decomposed["by_fold"].values()]
|
|
fold_cv = np.std(fold_counts) / max(np.mean(fold_counts), 1e-9)
|
|
|
|
lines.append(f"- Distinct semantic families contributing false actions: {n_families}")
|
|
lines.append(f"- Distinct held-out split groups: {n_groups}")
|
|
lines.append(f"- Largest single family: {max_family[0]} ({len(max_family[1])} false actions)")
|
|
lines.append(f"- Fold false-action count CV (std/mean): {fold_cv:.2f}")
|
|
if fold_cv > 0.5:
|
|
lines.append("- High fold variance suggests bad CV folds are a significant contributor")
|
|
elif n_families <= 3:
|
|
lines.append("- Few families suggests concentrated template failures")
|
|
else:
|
|
lines.append("- Broad distribution across families suggests systematic action/non-action overlap")
|
|
lines.append("")
|
|
|
|
# ─── 3. E5 Geometry ───────────────────────────────────────────────────
|
|
lines.append("## 3. Paired E5 Geometry")
|
|
lines.append("")
|
|
|
|
geometry = inspect_e5_geometry(examples)
|
|
|
|
if geometry["positive"]:
|
|
pos_sims = [p["similarity"] for p in geometry["positive"]]
|
|
lines.append(f"cosine(action seed, positive action):")
|
|
lines.append(f" mean={ff(np.mean(pos_sims))} std={ff(np.std(pos_sims))} "
|
|
f"min={ff(np.min(pos_sims))} max={ff(np.max(pos_sims))} n={len(pos_sims)}")
|
|
else:
|
|
lines.append("No positive action pairs found.")
|
|
|
|
if geometry["capability"]:
|
|
cap_sims = [p["similarity"] for p in geometry["capability"]]
|
|
lines.append(f"cosine(action seed, capability question):")
|
|
lines.append(f" mean={ff(np.mean(cap_sims))} std={ff(np.std(cap_sims))} "
|
|
f"min={ff(np.min(cap_sims))} max={ff(np.max(cap_sims))} n={len(cap_sims)}")
|
|
else:
|
|
lines.append("No capability question pairs found.")
|
|
|
|
if geometry["other_negative"]:
|
|
neg_sims = [p["similarity"] for p in geometry["other_negative"]]
|
|
lines.append(f"cosine(action seed, other negative contrast):")
|
|
lines.append(f" mean={ff(np.mean(neg_sims))} std={ff(np.std(neg_sims))} "
|
|
f"min={ff(np.min(neg_sims))} max={ff(np.max(neg_sims))} n={len(neg_sims)}")
|
|
else:
|
|
lines.append("No other negative pairs found.")
|
|
|
|
lines.append("")
|
|
|
|
# Overlap assessment
|
|
if geometry["positive"] and geometry["capability"]:
|
|
pos_mean = np.mean([p["similarity"] for p in geometry["positive"]])
|
|
cap_mean = np.mean([p["similarity"] for p in geometry["capability"]])
|
|
gap = pos_mean - cap_mean
|
|
lines.append(f"Separation gap (positive - capability): {ff(gap)}")
|
|
if gap < 0.05:
|
|
lines.append("**WARNING**: e5 maps action seeds and capability questions nearly on top of each other.")
|
|
lines.append("The representation itself may not preserve useful separation for this boundary.")
|
|
elif gap < 0.15:
|
|
lines.append("Moderate separation. e5 preserves some signal but the boundary is narrow.")
|
|
else:
|
|
lines.append("Good separation. e5 preserves useful distance between action and capability question.")
|
|
lines.append("")
|
|
|
|
# ─── 4. Binary Action Probe ───────────────────────────────────────────
|
|
lines.append("## 4. Binary Action Probe (Linear Logistic)")
|
|
lines.append("")
|
|
|
|
best_C_bin, bin_results = run_binary_action_probe(X_dev, y_dev, fold_ids_dev, C_VALUES)
|
|
bin_res = bin_results[best_C_bin]
|
|
|
|
lines.append(f"```text")
|
|
lines.append(f"ROC-AUC: {ff(bin_res['mean_roc_auc'])}")
|
|
lines.append(f"PR-AUC: {ff(bin_res['mean_pr_auc'])}")
|
|
lines.append(f"precision: {ff(bin_res['mean_precision'])}")
|
|
lines.append(f"recall: {ff(bin_res['mean_recall'])}")
|
|
lines.append(f"false-positive: {bin_res['total_fp']}")
|
|
lines.append(f"false-negative: {bin_res['total_fn']}")
|
|
lines.append(f"total: {sum(m['test_size'] for m in bin_res['fold_metrics'])}")
|
|
lines.append(f"best C: {best_C_bin}")
|
|
lines.append(f"```")
|
|
lines.append("")
|
|
|
|
# Per-fold
|
|
lines.append("Per-fold:")
|
|
for m in bin_res["fold_metrics"]:
|
|
lines.append(f" Fold {m['fold']}: ROC={ff(m['roc_auc'])} PR={ff(m['pr_auc'])} "
|
|
f"P={ff(m['action_precision'])} R={ff(m['action_recall'])} "
|
|
f"FP={m['false_pos']} FN={m['false_neg']} n={m['test_size']}")
|
|
lines.append("")
|
|
|
|
# ─── 5. Cost-Sensitive Linear ──────────────────────────────────────────
|
|
lines.append("## 5. Cost-Sensitive Linear Action Classification")
|
|
lines.append("")
|
|
|
|
cost_results = run_cost_sensitive(X_dev, y_dev, fold_ids_dev, COST_MULTIPLIERS)
|
|
|
|
lines.append(f"{'cost':>5} {'action_P':>10} {'action_R':>10} {'FA count':>10} {'FA rate':>10}")
|
|
lines.append("-" * 46)
|
|
for cost in COST_MULTIPLIERS:
|
|
r = cost_results[cost]
|
|
lines.append(f"{cost:>5} {ff(r['mean_precision']):>10} {ff(r['mean_recall']):>10} "
|
|
f"{r['total_false_action']:>10} {pct(r['mean_false_action_rate']):>10}")
|
|
lines.append("")
|
|
|
|
# ─── 6. Action-Threshold Sweep ────────────────────────────────────────
|
|
lines.append("## 6. Expanded Action-Threshold Sweep (30 action groups)")
|
|
lines.append("")
|
|
|
|
threshold_results = run_action_threshold_sweep(six_oof, ACTION_THRESHOLDS)
|
|
|
|
lines.append(f"{'threshold':>10} {'action_P':>10} {'action_R':>10} {'FA count':>10} {'coverage':>10}")
|
|
lines.append("-" * 50)
|
|
for t in threshold_results:
|
|
lines.append(f"{t['threshold']:>10.2f} {ff(t['action_precision']):>10} {ff(t['action_recall']):>10} "
|
|
f"{t['false_action_count']:>10} {pct(t['coverage']):>10}")
|
|
lines.append("")
|
|
|
|
# ─── 7. Voice-Like Stress ─────────────────────────────────────────────
|
|
lines.append("## 7. Voice-Like Punctuation Stress Evaluation")
|
|
lines.append("")
|
|
|
|
# Build dev index map
|
|
dev_source_ids = {e["source_id"] for e in dev_examples}
|
|
dev_indices = [i for i, e in enumerate(examples) if e["source_id"] in dev_source_ids]
|
|
|
|
stress = run_voice_stress_eval(examples, fold_ids_dev, dev_indices, X_dev, y_dev)
|
|
|
|
lines.append(f"Total stress-testable pairs: {stress['total_stress_pairs']}")
|
|
lines.append("")
|
|
lines.append("Sample stress pairs:")
|
|
for p in stress["pairs"][:10]:
|
|
lines.append(f" {p['source_id']}:")
|
|
lines.append(f" original: \"{p['original']}\"")
|
|
lines.append(f" stressed: \"{p['stressed']}\"")
|
|
lines.append(f" route: {p['route']}")
|
|
lines.append("")
|
|
lines.append("Impact assessment:")
|
|
lines.append(" Removal of punctuation changes:")
|
|
lines.append(" - trailing '?' removal eliminates the strongest question signal")
|
|
lines.append(" - lowercase normalization removes proper-noun casing cues")
|
|
lines.append(" - whitespace collapse has minimal effect on e5 (subword tokenizer)")
|
|
lines.append(" Production voice punctuation is unreliable; the model must not depend on it.")
|
|
lines.append("")
|
|
lines.append(f" Capability-question pairs total in stress set: {stress['cap_q_total']}")
|
|
lines.append(f" Capability-question pairs with trailing '?': {stress['cap_q_with_question_mark']}")
|
|
lines.append(f" Capability-question pairs with comma only: {stress['cap_q_with_comma_only']}")
|
|
lines.append("")
|
|
|
|
# Show sample affected capability questions
|
|
cap_q_samples = [p for p in stress["pairs"] if "capability_question" in p.get("tags", [])]
|
|
if cap_q_samples:
|
|
lines.append(" Sample affected capability questions:")
|
|
for p in cap_q_samples[:5]:
|
|
lines.append(f" \"{p['original']}\" → \"{p['stressed']}\"")
|
|
lines.append("")
|
|
|
|
# ─── 8. E5 + Structural Features ──────────────────────────────────────
|
|
lines.append("## 8. E5 + Tiny Structural Features")
|
|
lines.append("")
|
|
|
|
structural_dev = extract_structural_features(dev_examples)
|
|
best_C_struct, struct_results = run_e5_structural_experiment(
|
|
X_dev, structural_dev, y_dev, fold_ids_dev, C_VALUES
|
|
)
|
|
struct_res = struct_results[best_C_struct]
|
|
|
|
lines.append(f"```text")
|
|
lines.append(f"Features: [e5(384) ; IsQuestion(1) ; trailing_?(1) ; prohibition(1)] = 387 dims")
|
|
lines.append(f"Macro F1: {ff(struct_res['mean_f1'])}")
|
|
lines.append(f"False-action rate: {pct(struct_res['mean_false_action_rate'])}")
|
|
lines.append(f"False-action count: {struct_res['total_false_action']}")
|
|
lines.append(f"best C: {best_C_struct}")
|
|
lines.append(f"```")
|
|
lines.append("")
|
|
|
|
# Compare against pure e5
|
|
best_C_e5, e5_results = run_six_way_cv(X_dev, y_dev, fold_ids_dev, dev_examples, C_VALUES)
|
|
e5_res = e5_results[best_C_e5]
|
|
|
|
lines.append("Comparison with pure e5:")
|
|
lines.append(f" pure e5: macro_f1={ff(e5_res['mean_f1'])} FA_rate={pct(np.mean([m['false_action']/max(m['test_size'],1) for m in e5_res['fold_metrics']]))}")
|
|
fa_rates_struct = [m["false_action_rate"] for m in struct_res["fold_metrics"]]
|
|
lines.append(f" e5 + structural bits: macro_f1={ff(struct_res['mean_f1'])} FA_rate={pct(np.mean(fa_rates_struct))}")
|
|
lines.append("")
|
|
|
|
# ─── 9. Four-Way Comparison ───────────────────────────────────────────
|
|
lines.append("## 9. Four-Hypothesis Comparison Table")
|
|
lines.append("")
|
|
|
|
# Six-way e5 linear (residual)
|
|
six_metrics = compute_six_way_metrics(six_oof)
|
|
|
|
# Six-way + action threshold (best threshold)
|
|
best_thr = max(threshold_results, key=lambda t: t["action_precision"] if t["false_action_count"] <= 50 else 0)
|
|
if best_thr["false_action_count"] > 50:
|
|
best_thr = max(threshold_results, key=lambda t: t["action_precision"] * t["action_recall"])
|
|
|
|
# Binary linear probe
|
|
bin_metrics = {
|
|
"action_precision": bin_res["mean_precision"],
|
|
"action_recall": bin_res["mean_recall"],
|
|
"false_action_rate": bin_res["total_fp"] / max(sum(m["test_size"] for m in bin_res["fold_metrics"]), 1),
|
|
"macro_f1": 0.0, # binary doesn't have macro F1 in the 6-way sense
|
|
}
|
|
|
|
# E5 + structural
|
|
struct_fa_rates = [m["false_action_rate"] for m in struct_res["fold_metrics"]]
|
|
struct_metrics = {
|
|
"action_precision": np.mean([m["action_precision"] for m in struct_res["fold_metrics"]]),
|
|
"action_recall": np.mean([m["action_recall"] for m in struct_res["fold_metrics"]]),
|
|
"false_action_rate": np.mean(struct_fa_rates),
|
|
"macro_f1": struct_res["mean_f1"],
|
|
}
|
|
|
|
lines.append(f"| {'experiment':<35} | {'action_P':>10} | {'action_R':>10} | {'FA rate':>10} | {'macro F1':>10} |")
|
|
lines.append(f"| {'-'*35} | {'-'*10} | {'-'*10} | {'-'*10} | {'-'*10} |")
|
|
lines.append(f"| {'six-way e5 linear (residual)':<35} | {ff(six_metrics['action_precision']):>10} | {ff(six_metrics['action_recall']):>10} | {pct(six_metrics['false_action_rate']):>10} | {ff(six_metrics['macro_f1']):>10} |")
|
|
lines.append(f"| {'six-way + action threshold':<35} | {ff(best_thr['action_precision']):>10} | {ff(best_thr['action_recall']):>10} | {pct(best_thr['false_action_count']/max(safety['total'],1)):>10} | {'—':>10} |")
|
|
lines.append(f"| {'binary linear action probe':<35} | {ff(bin_metrics['action_precision']):>10} | {ff(bin_metrics['action_recall']):>10} | {pct(bin_metrics['false_action_rate']):>10} | {'—':>10} |")
|
|
lines.append(f"| {'e5 + tiny structural features':<35} | {ff(struct_metrics['action_precision']):>10} | {ff(struct_metrics['action_recall']):>10} | {pct(struct_metrics['false_action_rate']):>10} | {ff(struct_metrics['macro_f1']):>10} |")
|
|
lines.append("")
|
|
|
|
# ─── 10. Conclusion ────────────────────────────────────────────────────
|
|
lines.append("## 10. Conservative Interpretation")
|
|
lines.append("")
|
|
|
|
# Decision logic
|
|
if bin_res["mean_pr_auc"] > 0.90:
|
|
lines.append("### Binary linear action probe works well (PR-AUC > 0.90)")
|
|
lines.append("")
|
|
lines.append("The e5 representation is probably adequate for the action/non-action boundary.")
|
|
lines.append("The six-way softmax formulation is likely the problem: competition between")
|
|
lines.append("six classes creates false actions that a dedicated binary gate would not.")
|
|
lines.append("")
|
|
lines.append("A two-stage architecture becomes plausible:")
|
|
lines.append("```text")
|
|
lines.append("e5")
|
|
lines.append(" ├─ executable-action gate (binary)")
|
|
lines.append(" └─ coarse semantic route head (5-way or 6-way)")
|
|
lines.append("```")
|
|
elif struct_res["mean_f1"] > e5_res["mean_f1"] + 0.02:
|
|
lines.append("### Tiny structural bits fix it")
|
|
lines.append("")
|
|
lines.append("e5 loses a small amount of syntax/pragmatics that deterministic machinery")
|
|
lines.append("can supply cheaply. Prefer this over adding an MLP.")
|
|
elif bin_res["mean_pr_auc"] > 0.80:
|
|
lines.append("### Binary probe works but not dramatically better")
|
|
lines.append("")
|
|
lines.append("A nonlinear probe may help, but the improvement ceiling is moderate.")
|
|
lines.append("Consider whether the cost of a two-stage architecture is justified.")
|
|
else:
|
|
lines.append("### Action/capability-question embeddings are nearly indistinguishable")
|
|
lines.append("")
|
|
lines.append("The representation itself is suspect for this boundary.")
|
|
lines.append("Do not claim the boundary is merely nonlinear.")
|
|
lines.append("A fundamentally different representation or encoder may be needed.")
|
|
lines.append("")
|
|
|
|
# E5 geometry verdict
|
|
if geometry["positive"] and geometry["capability"]:
|
|
pos_mean = np.mean([p["similarity"] for p in geometry["positive"]])
|
|
cap_mean = np.mean([p["similarity"] for p in geometry["capability"]])
|
|
gap = pos_mean - cap_mean
|
|
if gap < 0.05:
|
|
lines.append("**E5 geometry verdict**: action seeds and capability questions are nearly")
|
|
lines.append("indistinguishable in embedding space. The representation intentionally")
|
|
lines.append("maps pragmatically different but semantically similar sentences close together.")
|
|
lines.append("This is a fundamental limitation of the frozen e5 representation for this boundary.")
|
|
else:
|
|
lines.append(f"**E5 geometry verdict**: moderate separation ({ff(gap)}) exists between")
|
|
lines.append("action seeds and capability questions. The representation preserves some signal.")
|
|
lines.append("")
|
|
|
|
# Softmax formulation verdict
|
|
lines.append(f"**Softmax formulation verdict**: the six-way head produces {safety['false_action_count']} false actions")
|
|
lines.append(f"at {pct(safety['false_action_rate'])} rate. The binary probe produces {bin_res['total_fp']} false actions.")
|
|
ratio = bin_res["total_fp"] / max(safety["false_action_count"], 1)
|
|
lines.append(f"The binary probe has {ratio:.1f}x the false-action count of the six-way head.")
|
|
if ratio < 0.8:
|
|
lines.append("This confirms the six-way softmax competition is a significant contributor.")
|
|
elif ratio > 1.2:
|
|
lines.append("Surprisingly, the binary probe does not reduce false actions. The issue is deeper than softmax competition.")
|
|
else:
|
|
lines.append("The binary probe and six-way head produce comparable false-action counts.")
|
|
lines.append("")
|
|
|
|
# ─── 11. Commit hash ──────────────────────────────────────────────────
|
|
lines.append("## 11. Commit hash for diagnostic tooling")
|
|
lines.append("")
|
|
lines.append("(to be filled after commit)")
|
|
lines.append("")
|
|
|
|
report = "\n".join(lines)
|
|
|
|
report_path = "/home/kami/apps/Maven/docs/evals/2026-09-07-slice16-diagnostic.md"
|
|
with open(report_path, "w") as f:
|
|
f.write(report)
|
|
print(f"\nReport written to {report_path}")
|
|
|
|
# Print summary
|
|
print("\n" + "=" * 60)
|
|
print(" SLICE 16 DIAGNOSTIC SUMMARY")
|
|
print("=" * 60)
|
|
print(f" Six-way (residual): P={ff(safety['action_precision'])} R={ff(safety['action_recall'])} FA={safety['false_action_count']} ({pct(safety['false_action_rate'])})")
|
|
print(f" Binary probe: P={ff(bin_res['mean_precision'])} R={ff(bin_res['mean_recall'])} FP={bin_res['total_fp']} PR-AUC={ff(bin_res['mean_pr_auc'])}")
|
|
print(f" E5+structural: F1={ff(struct_res['mean_f1'])} FA={struct_res['total_false_action']}")
|
|
pos_mean = ff(np.mean([p['similarity'] for p in geometry['positive']])) if geometry['positive'] else 'N/A'
|
|
cap_mean = ff(np.mean([p['similarity'] for p in geometry['capability']])) if geometry['capability'] else 'N/A'
|
|
print(f" E5 geometry: pos={pos_mean} cap={cap_mean}")
|
|
print(f" False-action families: {n_families}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|