init
This commit is contained in:
@@ -0,0 +1,63 @@
|
||||
"""Phase 2.3/2.4 DONE-CHECK — per-bucket stats + eyeball sample.
|
||||
|
||||
Reports doc/char counts per cleaned bucket, the synth share (must be ≤10%), and
|
||||
prints N random cleaned docs so a human confirms they read as clean Russian with
|
||||
no homoglyph words.
|
||||
|
||||
Usage: python corpus_stats.py [--sample 5]
|
||||
"""
|
||||
import sys
|
||||
import json
|
||||
import random
|
||||
|
||||
from corpus_common import CLEAN, has_mixed_script, cyrillic_ratio, SYNTH_CAP
|
||||
|
||||
|
||||
def load(bucket_path):
|
||||
with open(bucket_path, encoding="utf-8") as f:
|
||||
return [json.loads(ln)["text"] for ln in f if ln.strip()]
|
||||
|
||||
|
||||
def main():
|
||||
n_sample = 5
|
||||
if "--sample" in sys.argv:
|
||||
n_sample = int(sys.argv[sys.argv.index("--sample") + 1])
|
||||
|
||||
files = sorted(CLEAN.glob("*.jsonl"))
|
||||
if not files:
|
||||
raise SystemExit(f"no cleaned buckets in {CLEAN}; run corpus_clean.py")
|
||||
|
||||
totals = {"docs": 0, "chars": 0}
|
||||
per = {}
|
||||
for fp in files:
|
||||
texts = load(fp)
|
||||
chars = sum(len(t) for t in texts)
|
||||
per[fp.stem] = (len(texts), chars)
|
||||
totals["docs"] += len(texts)
|
||||
totals["chars"] += chars
|
||||
|
||||
print(f"{'bucket':<14}{'docs':>10}{'chars(M)':>12}{'share%':>9}")
|
||||
for b, (d, c) in per.items():
|
||||
share = 100 * c / totals["chars"] if totals["chars"] else 0
|
||||
flag = " <-- OVER CAP" if b == "synth" and share > SYNTH_CAP * 100 else ""
|
||||
print(f"{b:<14}{d:>10}{c/1e6:>12.1f}{share:>9.1f}{flag}")
|
||||
print(f"{'TOTAL':<14}{totals['docs']:>10}{totals['chars']/1e6:>12.1f}")
|
||||
# crude token estimate (~3.5 chars/tok RU)
|
||||
print(f"~est tokens: {totals['chars']/3.5/1e6:.0f}M (target ~300M)")
|
||||
|
||||
# eyeball sample — the actual DONE-CHECK
|
||||
print(f"\n--- {n_sample} random cleaned docs (verify clean Russian) ---")
|
||||
allpaths = list(files)
|
||||
for _ in range(n_sample):
|
||||
fp = random.choice(allpaths)
|
||||
texts = load(fp)
|
||||
if not texts:
|
||||
continue
|
||||
t = random.choice(texts)
|
||||
assert not has_mixed_script(t), f"MIXED SCRIPT LEAKED in {fp.stem}: {t[:120]}"
|
||||
print(f"\n[{fp.stem}] cyr={cyrillic_ratio(t):.2f}\n{t[:400]}")
|
||||
print("\nOK — no mixed-script leaked into sample.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user