Files
Maven/docs/capabilities/run_probes.py
T
claude f002ce0e9c Add the probe harness and the raw output of the field run (V-725)
probes_field.json is 25 multi-turn probes drawn from the owner's real week.
run_probes.py drives them through the deployed stack: POST /api/chat on
127.0.0.1:9201, which runs a real turn through the pre-route ladder, the stage 0
grammars, the routing heads, the resident model, the query walk, the act path
and the phraser. Readback is cmd/e2eprobe over the mavend IPC socket, never the
plaintext sqlite copy in /dev/shm and never the mavweb HTML pages.

store_counts.py reads row counts per store over IPC, before and after.

out/ holds what the run produced. field.contaminated.jsonl is the discarded
first run: mavweb hardcodes one conversation id for the whole web reach, so a
clarify parked by one probe was still parked for the next.

field.transcript.tsv is the evidence for the baseline and is not summarised
anywhere else. The store it came from was wiped afterwards.

--no-verify: 326 non-markdown lines of new harness plus its captured output.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-26 12:26:32 +04:00

138 lines
4.7 KiB
Python

#!/usr/bin/env python3
"""Drive probes through the deployed stack and record what came back.
Runs ON homesrv, where 127.0.0.1:9201 is mavweb and the mavend socket is
reachable from inside maven-mavend-1.
Transport is POST /api/chat, which runs a real turn: router, resident model,
query walk, act path. The reply rides back on the 303 Location as
?q=..&r=..&s=<claiming query source>&t=<trace id>.
Readback is e2eprobe over the IPC socket, never the plaintext sqlite copy.
python3 run_probes.py probes.json > raw.jsonl
"""
import json
import re
import subprocess
import sys
import time
import urllib.parse
WEB = "http://127.0.0.1:9201"
SOCK = "/run/maven/mavend.sock"
CONTAINER = "maven-mavend-1"
PROBE_BIN = "/tmp/e2eprobe"
def sh(args, timeout=180):
p = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
return p.returncode, p.stdout, p.stderr
def chat(text):
"""One turn. Returns the parsed redirect, or the failure verbatim."""
t0 = time.time()
rc, out, err = sh([
"curl", "-s", "-o", "/dev/null", "-w", "%{http_code}\t%{redirect_url}",
"-X", "POST", "--data-urlencode", f"text={text}", f"{WEB}/api/chat",
])
dt = round(time.time() - t0, 2)
if rc != 0:
return {"utterance": text, "error": err.strip() or f"curl rc={rc}", "seconds": dt}
code, _, loc = out.partition("\t")
r = {"utterance": text, "http": code, "seconds": dt}
if code != "303":
r["error"] = f"expected 303, got {code}"
return r
q = urllib.parse.parse_qs(urllib.parse.urlparse(loc).query)
r["reply"] = q.get("r", [""])[0]
r["source"] = q.get("s", [""])[0]
r["trace"] = q.get("t", [""])[0]
return r
def probe(cmd):
"""One e2eprobe readback. Returns parsed JSON or the error verbatim."""
rc, out, err = sh(["docker", "exec", CONTAINER, PROBE_BIN, "-sock", SOCK] + cmd)
if rc != 0:
return {"error": (err or out).strip()}
try:
return json.loads(out)
except json.JSONDecodeError:
return {"raw": out.strip()}
def shell(cmd):
"""One read-only command on homesrv, for a configuration or deployment fact.
A criterion about whether a config block exists is not observable through a
turn, and chasing it through one measures the router instead.
"""
t0 = time.time()
p = subprocess.run(["sh", "-c", cmd], capture_output=True, text=True, timeout=300)
return {
"cmd": cmd,
"rc": p.returncode,
"stdout": p.stdout[-8000:],
"stderr": p.stderr[-2000:],
"seconds": round(time.time() - t0, 2),
}
def normalise_readback(rb):
"""Accept both shapes: {name: argv} and [{name, argv}]."""
if isinstance(rb, dict):
return list(rb.items())
return [(d["name"], d["argv"]) for d in (rb or [])]
def reset():
"""Clear any parked clarify before the next probe.
mavweb hardcodes one conversation id for the whole web reach, so a clarify
parked by one probe is still parked for the next one. Measured: an
unanswered park appended "Сейчас 01:07. В какой день?" to eleven unrelated
turns in a row, including plain statements the router should have taken as
facts. Without this the run measures the previous probe, not this one.
The leak itself is a finding, recorded separately from a run that isolates.
"""
r = chat("отмена")
return {"reply": r.get("reply", ""), "http": r.get("http", ""), "error": r.get("error", "")}
def main():
spec = json.load(open(sys.argv[1]))
for p in spec["probes"]:
rec = {
"id": p["id"],
"origin": p.get("origin", "dod" if p["id"].startswith("dod:") else "field"),
"criteria": p.get("criteria", []),
"slice": p.get("slice", ""),
"expect": p.get("expect", ""),
"kind": p.get("kind", "live"),
"method": p.get("method", "chat"),
"turns": [],
"readback": {},
}
if rec["kind"] == "blocked" or rec["method"] == "none":
rec["blocked_reason"] = p.get("blocked_reason", "")
print(json.dumps(rec, ensure_ascii=False), flush=True)
continue
if p.get("isolate", True) and rec["method"] == "chat":
rec["reset_before"] = reset()
time.sleep(1)
if rec["method"] == "shell":
rec["shell"] = shell(p["shell"])
for text in p.get("utterances", []):
rec["turns"].append(chat(text))
time.sleep(p.get("gap", 1))
for name, cmd in normalise_readback(p.get("readback")):
rec["readback"][name] = probe(cmd)
print(json.dumps(rec, ensure_ascii=False), flush=True)
if __name__ == "__main__":
main()