textmachine/eval/exp16/length_sweep.py

102 lines
4.6 KiB
Python

#!/usr/bin/env python3
"""exp16 PRE-TASK (D39.9, $0) — sweep finish_reason across ALL exp14 / exp14b cost ledgers.
The exp14 harness bug (exp14_common.call, lines 104-113): a response with finish_reason=length is
returned as (text, err=None, usage) whenever text is non-empty — a TRUNCATED output passes as valid.
Only content_filter and empty-text are rejected. Every ledger record stamps usage.finish_reason, so
we can sweep them post-hoc and flag any accepted truncation that fed the DET scoring.
We do NOT edit the exp14 rig. This is a read-only sweep. For any finish=length record, we report
(arm/model, id, err, completion_tokens, cost) and whether it maps to a DET-battery chunk used in
scoring. Cross-reference with reaudit_14b.py's 'truncated_looking' file heuristic.
"""
from __future__ import annotations
import json
from collections import defaultdict
from pathlib import Path
LEDGERS = [
Path("/home/ubuntu/books/gu-zhenren/exp14/costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14/judge_costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14/rank_costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14/regate_costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14/sizecurve_costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14/gemini_retest_costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14b/costs.jsonl"),
Path("/home/ubuntu/books/gu-zhenren/exp14b/judge_costs.jsonl"),
]
# DET-battery chunks that fed exp14b DET scoring (a1/a2 7.0, b1 10.1, b2 6.0, c1 17.0, c2 9.1, c3 16.0, d1 19.0, d2 7.0)
DET_CHUNKS = {"7.0", "10.1", "6.0", "17.0", "9.1", "16.0", "19.0"}
def main():
by_finish = defaultdict(int) # finish_reason -> count
by_model_finish = defaultdict(lambda: defaultdict(int))
length_records = [] # accepted-or-not length records
total = 0
for lp in LEDGERS:
if not lp.exists():
print(f"[skip] {lp} (absent)")
continue
for line in lp.read_text(encoding="utf-8").splitlines():
line = line.strip()
if not line:
continue
try:
rec = json.loads(line)
except json.JSONDecodeError:
continue
total += 1
usage = rec.get("usage", {}) or {}
fr = str(usage.get("finish_reason", "")) or "(none)"
model = rec.get("model", rec.get("arm", "?"))
by_finish[fr] += 1
by_model_finish[model][fr] += 1
if fr.lower() == "length":
length_records.append({
"ledger": lp.parent.name + "/" + lp.name,
"arm": rec.get("arm"), "model": model, "id": rec.get("id"),
"attempt": rec.get("attempt"), "err": rec.get("err"),
"completion_tokens": usage.get("completion_tokens"),
"cost": rec.get("cost"),
})
print("=" * 90)
print(f"finish_reason SWEEP — {total} ledger records across {len(LEDGERS)} ledgers")
print("=" * 90)
print("\nfinish_reason distribution (all records, incl. every attempt & judge call):")
for fr, n in sorted(by_finish.items(), key=lambda x: -x[1]):
print(f" {fr:<16} {n}")
print("\nfinish=length by model:")
any_len = False
for model, fd in sorted(by_model_finish.items()):
if fd.get("length"):
any_len = True
print(f" {model:<22} length={fd['length']} (all finish: {dict(fd)})")
if not any_len:
print(" (none)")
print("\n" + "=" * 90)
print(f"finish=length RECORDS: {len(length_records)}")
print(" err=None + non-empty text => ACCEPTED as valid by the buggy gate (truncation in the pool)")
print("=" * 90)
for r in length_records:
accepted = r["err"] in (None, "", "null")
det = " <DET-CHUNK>" if r["id"] in DET_CHUNKS else ""
tag = "ACCEPTED-TRUNCATION" if accepted else f"rejected(err={r['err']})"
print(f" {r['ledger']:<22} arm={str(r['arm']):<4} model={r['model']:<20} id={str(r['id']):<6} "
f"attempt={r['attempt']} comp_tok={r['completion_tokens']} ${r['cost']} => {tag}{det}")
json.dump({"by_finish": dict(by_finish),
"by_model_finish": {m: dict(f) for m, f in by_model_finish.items()},
"length_records": length_records},
open(Path("/home/ubuntu/books/gu-zhenren/exp16") / "length_sweep.json", "w"),
ensure_ascii=False, indent=1)
print(f"\n[written] /home/ubuntu/books/gu-zhenren/exp16/length_sweep.json")
if __name__ == "__main__":
main()