diff --git a/eval/rerun2_judge/build_blind.py b/eval/rerun2_judge/build_blind.py new file mode 100644 index 00000000..5e8e6fe0 --- /dev/null +++ b/eval/rerun2_judge/build_blind.py @@ -0,0 +1,105 @@ +#!/usr/bin/env python3 +"""D39.8 blind-read package builder for the pere-run (rerun2). + +3 editor arms (glm-5 / mistral / deepseek-pro) of 蛊真人 ch1-5, zh->ru. Per CHAPTER, the 3 versions are +shuffled to neutral labels A/B/C with a PER-CHAPTER salt (so the reader cannot accumulate "C = one model" +across chapters). Source at the top of each chapter for fidelity comparison. Auto-QA (CJK-in-ru residual) +runs before the human. Correspondence key -> a SEPARATE file, withheld from the owner until verdict (D39.8). +""" +import json, random, re +from pathlib import Path + +RUN = Path("/home/ubuntu/books/gu-zhenren/rerun2") +OUT = RUN / "blind" +OUT.mkdir(exist_ok=True) +ARMS = ["glm", "mistral", "dspro"] + +def load(): + arms = {} + for a in ARMS: + d = json.load(open(RUN / f"export-{a}.json")) + arms[a] = d["chunks"] + return arms + +def by_chapter(chunks): + ch = {} + for c in sorted(chunks, key=lambda c: (int(c["chapter"]), int(c["chunk_idx"]))): + ch.setdefault(c["chapter"], {"final": [], "source": [], "disp": []}) + ch[c["chapter"]]["final"].append(c["final_text"]) + ch[c["chapter"]]["source"].append(c["source"]) + ch[c["chapter"]]["disp"].append(c["disposition"]) + return {k: {"final": "\n\n".join(v["final"]), "source": "\n\n".join(v["source"]), + "disp": v["disp"]} for k, v in ch.items()} + +CJK = re.compile(r"[㐀-鿿豈-﫿]") +def auto_qa(text): + flags = [] + cjk = CJK.findall(text) + if cjk: + flags.append(f"CJK-in-ru residual: {len(cjk)} char(s) e.g. {''.join(cjk[:6])}") + # crude latin-word leak (>=3 latin letters run, excluding common ok tokens) + lat = re.findall(r"[A-Za-z]{3,}", text) + if lat: + flags.append(f"latin run(s): {lat[:5]}") + return flags + +def main(): + arms = load() + chap = {a: by_chapter(arms[a]) for a in ARMS} + chapters = sorted(chap["glm"].keys(), key=int) + + key = {} + packet = [] + packet.append("# Слепой пакет чтения — пере-прогон rerun2 (蛊真人, главы 1–5, zh→ru)\n") + packet.append( + "Три версии перевода каждой главы (**A / B / C**) — это три редакторских «руки» одного и того же " + "чернового перевода. Метки **перетасованы заново в каждой главе** (в гл.1 «A» и в гл.2 «A» — это, " + "скорее всего, РАЗНЫЕ руки). Соответствие меток и моделей — в отдельном файле " + "`blind_read_KEY.json`, **его не смотреть до вердикта**.\n") + packet.append( + "**Как читать (планка: ≤2 претензии на версию):** по каждой главе отметьте для A/B/C претензии " + "по ярусам — **критические** (искажение смысла / пропуск / неверный род / непереведённый " + "китайский / сломанное слово), **смысловые** (неточность, единицы времени, числа), " + "**редакторские** (стиль, ритм, канцелярит) — и назовите предпочтительную версию.\n") + packet.append("---\n") + + for ch in chapters: + rng = random.Random(f"rerun2-blind-ch{ch}") + order = ARMS[:] + rng.shuffle(order) + labels = ["A", "B", "C"] + key[ch] = {labels[i]: order[i] for i in range(3)} + src = chap[order[0]][ch]["source"] # same source across arms + packet.append(f"## Глава {ch}\n") + packet.append(f"
Китайский исходник (для сверки верности)\n\n```\n{src}\n```\n
\n") + for i, lab in enumerate(labels): + arm = order[i] + txt = chap[arm][ch]["final"] + qa = auto_qa(txt) + qa_note = f"\n> _(авто-QA: {'; '.join(qa)})_\n" if qa else "" + packet.append(f"### Глава {ch} — версия {lab}\n{qa_note}\n{txt}\n") + packet.append(f"\n**Ваши претензии по главе {ch}:**\n" + "- Версия A — критич.: … · смысл.: … · редакт.: …\n" + "- Версия B — критич.: … · смысл.: … · редакт.: …\n" + "- Версия C — критич.: … · смысл.: … · редакт.: …\n" + "- Предпочтение: …\n\n---\n") + + (OUT / "blind_read_packet.md").write_text("\n".join(packet)) + (OUT / "blind_read_KEY.json").write_text(json.dumps( + {"note": "WITHHELD from owner until verdict (D39.8). Per-chapter label->arm mapping.", + "arm_label": {"glm": "glm-5 (base)", "mistral": "mistral-large-2512", "dspro": "deepseek-v4-pro"}, + "mapping": key}, ensure_ascii=False, indent=2)) + # QA summary (mine, not owner-facing) + qa_summary = {} + for ch in chapters: + for arm in ARMS: + f = auto_qa(chap[arm][ch]["final"]) + if f: qa_summary.setdefault(arm, {})[ch] = f + (OUT / "auto_qa.json").write_text(json.dumps(qa_summary, ensure_ascii=False, indent=2)) + print("packet:", OUT / "blind_read_packet.md") + print("key (withheld):", OUT / "blind_read_KEY.json") + print("chapters:", chapters) + print("auto-QA hits by arm:", {a: list(v.keys()) for a, v in qa_summary.items()}) + +if __name__ == "__main__": + main() diff --git a/eval/rerun2_judge/judge.py b/eval/rerun2_judge/judge.py new file mode 100644 index 00000000..4cf13200 --- /dev/null +++ b/eval/rerun2_judge/judge.py @@ -0,0 +1,381 @@ +#!/usr/bin/env python3 +"""D39.7 judge rig for the pere-run (rerun2) — 3 editor arms of 蛊真人 ch1-5, zh->ru. + +Judge = gemini-3.1-pro-preview (cross-family to the zh->ru editors glm-5/mistral/deepseek-pro, +so no author==reviewer). Full windows (whole source unit + whole both translations, NO truncation). +Per-vote JSONL persistence + resume. Position-swap (both orders). Per-arm catastrophe screen (ALL +arms incl. the favorite). Identical-text floor pass (judge-noise baseline). Hard cost cap. + +Metrics NOMINATE; the owner blind read RATIFIES. Backend deterministic signals (echo/flags) are +reported ALONGSIDE, never fused (no manufactured convergence). + +Usage: + eval/.venv/bin/python eval/rerun2_judge/judge.py --probe # live-probe gemini slug ($ ~0.001) + eval/.venv/bin/python eval/rerun2_judge/judge.py --run # full rig + eval/.venv/bin/python eval/rerun2_judge/judge.py --aggregate # re-aggregate from JSONL only ($0) +""" +import os, sys, json, time, argparse, itertools +from pathlib import Path + +HERE = Path(__file__).resolve().parent +EVAL = HERE.parent +RUN = Path("/home/ubuntu/books/gu-zhenren/rerun2") +OUT = RUN / "judge" +OUT.mkdir(exist_ok=True) + +# --- keys: load eval/.env via dotenv (we never read .env ourselves; the lib does) --- +try: + from dotenv import load_dotenv + load_dotenv(EVAL / ".env") +except Exception: + pass + +from openai import OpenAI # openai 2.44.0 in eval/.venv + +# Judge providers: gemini primary; grok fallback for units gemini refuses with PROHIBITED_CONTENT +# (Gemini 3.x fail-closes on sexual content — non-configurable, D22.6; erotica judge = Grok). Grok is +# cross-family to the zh->ru editors (glm/mistral/deepseek) so no author==reviewer. +PROVIDERS = { + "gemini": dict(model="gemini-3.1-pro-preview", base="https://generativelanguage.googleapis.com/v1beta/openai", + key_env="GEMINI_API_KEY", max_tokens=20000, temperature=0.0, + price_in=2.0, price_out=12.0, reasoning="from_total"), + "grok": dict(model="grok-4.3", base="https://api.x.ai/v1", + key_env="XAI_API_KEY", max_tokens=8000, temperature=0.0, + price_in=1.25, price_out=2.50, reasoning="field"), +} +PRIMARY, FALLBACK = "gemini", "grok" +JUDGE_MODEL = PROVIDERS[PRIMARY]["model"] # display/aggregate reference +COST_CAP_USD = 8.0 # hard internal ceiling (well under the $15 experiment cap) + +ARMS = ["glm", "mistral", "dspro"] +ARM_LABEL = {"glm": "glm-5", "mistral": "mistral-large-2512", "dspro": "deepseek-v4-pro"} +REPS = 2 # repeats per (pair, unit, order) — 2 orders x 2 reps = 4 votes/pair-unit + +_clients = {} +def client(prov): + if prov not in _clients: + cfg = PROVIDERS[prov] + key = os.environ.get(cfg["key_env"]) + if not key: + sys.exit(f"FATAL: {cfg['key_env']} not set (eval/.env not loaded?)") + _clients[prov] = OpenAI(base_url=cfg["base"], api_key=key, timeout=240) + return _clients[prov] + +class Ledger: + def __init__(self, path): + self.path = path + self.total = 0.0 + if path.exists(): + for ln in path.read_text().splitlines(): + try: self.total += json.loads(ln).get("cost_usd", 0.0) + except Exception: pass + def add(self, rec): + self.total += rec.get("cost_usd", 0.0) + with open(self.path, "a") as f: + f.write(json.dumps(rec, ensure_ascii=False) + "\n") + +LEDGER = Ledger(OUT / "ledger.jsonl") + +def _reasoning_tokens(u, mode): + if mode == "from_total": # gemini: thinking only in total_tokens + pt = getattr(u, "prompt_tokens", 0) or 0 + ct = getattr(u, "completion_tokens", 0) or 0 + tt = getattr(u, "total_tokens", 0) or 0 + return max(0, tt - pt - ct) + det = getattr(u, "completion_tokens_details", None) # grok/openai: reasoning_tokens field (additive) + return (getattr(det, "reasoning_tokens", 0) if det else 0) or getattr(u, "reasoning_tokens", 0) or 0 + +def _raw_call(prov, system, user, tag): + """One provider call. Returns (text, rec). Ledgers cost. Retries transient; empty on hard fail.""" + if LEDGER.total >= COST_CAP_USD: + sys.exit(f"FATAL: cost cap ${COST_CAP_USD} reached (spent ${LEDGER.total:.4f}) — STOP") + cfg = PROVIDERS[prov] + backoff = [5, 15, 35, 60] + for attempt in range(len(backoff) + 1): + try: + r = client(prov).chat.completions.create( + model=cfg["model"], + messages=[{"role": "system", "content": system}, + {"role": "user", "content": user}], + temperature=cfg["temperature"], max_tokens=cfg["max_tokens"], + ) + ch = r.choices[0] + finish = ch.finish_reason + m = getattr(ch, "message", None) + txt = ((m.content if m is not None else None) or "").strip() + u = r.usage + pt = getattr(u, "prompt_tokens", 0) or 0 + ct = getattr(u, "completion_tokens", 0) or 0 + tt = getattr(u, "total_tokens", 0) or 0 + reasoning = _reasoning_tokens(u, cfg["reasoning"]) + out_billed = ct + reasoning + cost = pt * cfg["price_in"] / 1e6 + out_billed * cfg["price_out"] / 1e6 + rec = dict(tag=tag, judge=cfg["model"], finish=finish, prompt_tok=pt, completion_tok=ct, + reasoning_tok=reasoning, total_tok=tt, cost_usd=cost, empty=(not txt)) + LEDGER.add(rec) + if finish != "stop": + rec["ANOMALY"] = f"finish_reason={finish!r}" + print(f" [{prov} {tag}] finish={finish!r} empty={not txt}") + return txt, rec + except Exception as e: + if attempt < len(backoff): + print(f" [retry {prov} {tag}] {str(e)[:100]} — sleep {backoff[attempt]}s") + time.sleep(backoff[attempt]) + else: + print(f" [FAIL {prov} {tag}] {str(e)[:180]}") + LEDGER.add(dict(tag=tag, judge=cfg["model"], error=str(e)[:300], cost_usd=0.0)) + return "", dict(tag=tag, judge=cfg["model"], error=str(e)[:300], cost_usd=0.0) + +def _refused(rec, txt): + f = str(rec.get("finish") or "") + return (not txt) and ("content_filter" in f or "PROHIBITED" in f or "SAFETY" in f) + +def gemini_call(system, user, tag): + """Judge with gemini primary; on PROHIBITED_CONTENT refusal, fall back to grok (erotica judge, D22.6).""" + txt, rec = _raw_call(PRIMARY, system, user, tag) + if _refused(rec, txt): + print(f" [fallback->grok {tag}] gemini {rec.get('finish')}") + txt2, rec2 = _raw_call(FALLBACK, system, user, tag + "::grok") + rec2["fallback_from"] = f"gemini:{rec.get('finish')}" + return txt2, rec2 + return txt, rec + +# ---------------- data ---------------- +def load_arms(): + arms = {} + for a in ARMS: + d = json.load(open(RUN / f"export-{a}.json")) + by = {} + for c in d["chunks"]: + by[(c["chapter"], c["chunk_idx"])] = c + arms[a] = by + keys = sorted(set().union(*[set(v) for v in arms.values()]), + key=lambda k: (int(k[0]), int(k[1]))) + return arms, keys + +# ---------------- prompts ---------------- +PAIR_SYS = ( + "Ты — строгий эксперт по художественному переводу с китайского на русский (веб-новелла, жанр сянься). " + "Тебе дают КИТАЙСКИЙ исходник и ДВА русских перевода: A и B. Оцени, какой перевод лучше как " + "ХУДОЖЕСТВЕННЫЙ русский текст, по трём осям в порядке важности: (1) ВЕРНОСТЬ — смысл исходника " + "передан без искажений, пропусков и отсебятины; (2) ЕСТЕСТВЕННЫЙ ЛИТЕРАТУРНЫЙ РУССКИЙ — читается как " + "родная русская проза, без переводческого канцелярита и кальки; (3) СОГЛАСОВАННОСТЬ имён/терминов. " + "Отметь КАТАСТРОФУ у стороны, если есть: искажение смысла, пропуск предложения, неверный род персонажа, " + "непереведённый китайский в тексте, сломанная русская морфология. " + "Ответь СТРОГИМ JSON одной строкой, без пояснений вокруг:\n" + '{"winner":"A|B|tie","margin":"clear|slight","reason":"<=25 слов, с краткой цитатой-уликой>",' + '"catastrophe_A":true|false,"catastrophe_B":true|false,"catastrophe_detail":"<кратко или пусто>"}') + +CAT_SYS = ( + "Ты — строгий редактор-контролёр перевода с китайского на русский. Тебе дают КИТАЙСКИЙ исходник и ОДИН " + "русский перевод. Найди КАТАСТРОФИЧЕСКИЕ дефекты (те, из-за которых взыскательный читатель забракует " + "отрывок): искажение смысла, пропуск целого предложения/абзаца, неверный род персонажа, непереведённый " + "китайский (иероглифы) в русском тексте, сломанная русская словоформа, грубая ошибка в числах/единицах " + "времени. Мелкие стилистические придирки НЕ катастрофа. " + "Ответь СТРОГИМ JSON одной строкой:\n" + '{"catastrophe":true|false,"severity":"none|minor|major|critical","items":["<кратко с цитатой>", ...]}') + +def pair_user(src, ta, tb): + return (f"=== КИТАЙСКИЙ ИСХОДНИК ===\n{src}\n\n=== ПЕРЕВОД A ===\n{ta}\n\n=== ПЕРЕВОД B ===\n{tb}\n\n" + "Верни JSON-вердикт.") + +def cat_user(src, t): + return f"=== КИТАЙСКИЙ ИСХОДНИК ===\n{src}\n\n=== РУССКИЙ ПЕРЕВОД ===\n{t}\n\nВерни JSON." + +def parse_json(txt): + if not txt: return None + s = txt.strip() + if s.startswith("```"): + s = s.strip("`") + s = s[s.find("{"):] + i, j = s.find("{"), s.rfind("}") + if i < 0 or j < 0: return None + try: return json.loads(s[i:j+1]) + except Exception: return None + +# ---------------- resumable vote store ---------------- +def load_votes(path): + seen = {} + if path.exists(): + for ln in path.read_text().splitlines(): + try: + v = json.loads(ln) + seen[v["vote_id"]] = v + except Exception: pass + return seen + +def append_vote(path, v): + with open(path, "a") as f: + f.write(json.dumps(v, ensure_ascii=False) + "\n") + +# ---------------- passes ---------------- +def run_probe(): + print(f"=== LIVE PROBE {JUDGE_MODEL} ===") + txt, rec = gemini_call( + "Ты judge. Ответь строгим JSON.", + 'Верни ровно: {"ok":true,"lang":"ru"}', + "probe") + print("finish:", rec.get("finish"), "| cost $", round(rec.get("cost_usd", 0), 5), + "| tokens p/c/r:", rec.get("prompt_tok"), rec.get("completion_tok"), rec.get("reasoning_tok")) + print("response:", txt[:200]) + ok = rec.get("finish") == "stop" and parse_json(txt) is not None + print("PROBE", "OK" if ok else "FAILED (check slug/vendor doc per two-directions rule)") + return ok + +def run_floor(arms, keys): + """Identical-text position-bias floor: judge glm vs glm (same text), both orders. + Deviation from 'tie' = judge noise. (Full A0<->A0' independent-regen floor deferred; owner read ratifies.)""" + path = OUT / "votes_floor.jsonl" + seen = load_votes(path) + print(f"=== FLOOR pass (glm vs glm identical), {len(keys)} units x 2 orders ===") + for k in keys: + c = arms["glm"][k] + for order in ("AB", "BA"): + vid = f"floor::{k[0]}-{k[1]}::{order}" + if vid in seen: continue + txt, rec = gemini_call(PAIR_SYS, pair_user(c["source"], c["final_text"], c["final_text"]), + vid) + v = parse_json(txt) or {} + rec_v = dict(vote_id=vid, kind="floor", unit=f"{k[0]}-{k[1]}", order=order, + winner=v.get("winner"), margin=v.get("margin"), reason=v.get("reason"), + raw=txt[:400], finish=rec.get("finish"), judge=rec.get("judge")) + append_vote(path, rec_v) + print(f" floor {k[0]}-{k[1]} {order}: winner={v.get('winner')} (identical) ${LEDGER.total:.3f}") + +def run_catastrophe(arms, keys): + """Per-arm absolute catastrophe screen — ALL arms incl. favorite.""" + path = OUT / "votes_catastrophe.jsonl" + seen = load_votes(path) + print(f"=== CATASTROPHE screen, {len(keys)} units x {len(ARMS)} arms ===") + for a in ARMS: + for k in keys: + c = arms[a][k] + vid = f"cat::{a}::{k[0]}-{k[1]}" + if vid in seen: continue + txt, rec = gemini_call(CAT_SYS, cat_user(c["source"], c["final_text"]), vid) + v = parse_json(txt) or {} + rec_v = dict(vote_id=vid, kind="catastrophe", arm=a, unit=f"{k[0]}-{k[1]}", + disposition=c.get("disposition"), + catastrophe=bool(v.get("catastrophe")), severity=v.get("severity"), + items=v.get("items"), raw=txt[:500], finish=rec.get("finish"), judge=rec.get("judge")) + append_vote(path, rec_v) + flag = "CATA" if v.get("catastrophe") else "ok" + print(f" cat {a} {k[0]}-{k[1]}: {flag} ({v.get('severity')}) ${LEDGER.total:.3f}") + +def run_pairwise(arms, keys): + """Round-robin pairwise, both orders, REPS reps. Full windows. Order-normalized winner.""" + path = OUT / "votes_pairwise.jsonl" + seen = load_votes(path) + pairs = list(itertools.combinations(ARMS, 2)) + print(f"=== PAIRWISE, {len(pairs)} pairs x {len(keys)} units x 2 orders x {REPS} reps ===") + for (x, y) in pairs: + for k in keys: + cx, cy = arms[x][k], arms[y][k] + for order in ("AB", "BA"): + A, B = (x, y) if order == "AB" else (y, x) + cA, cB = arms[A][k], arms[B][k] + for rep in range(REPS): + vid = f"pair::{x}-{y}::{k[0]}-{k[1]}::{order}::r{rep}" + if vid in seen: continue + txt, rec = gemini_call(PAIR_SYS, pair_user(cA["source"], cA["final_text"], cB["final_text"]), + vid) + v = parse_json(txt) or {} + w = v.get("winner") + # normalize winner label (A/B) -> actual arm + if w == "A": win_arm = A + elif w == "B": win_arm = B + elif w == "tie": win_arm = "tie" + else: win_arm = None + rec_v = dict(vote_id=vid, kind="pairwise", pair=f"{x}-{y}", unit=f"{k[0]}-{k[1]}", + order=order, rep=rep, posA=A, posB=B, + winner_arm=win_arm, margin=v.get("margin"), reason=v.get("reason"), + cata_A=bool(v.get("catastrophe_A")), cata_B=bool(v.get("catastrophe_B")), + cata_detail=v.get("catastrophe_detail"), raw=txt[:400], + finish=rec.get("finish"), judge=rec.get("judge")) + append_vote(path, rec_v) + print(f" pair {x}-{y} {k[0]}-{k[1]} {order} r{rep}: win={win_arm} ({v.get('margin')}) ${LEDGER.total:.3f}") + +# ---------------- aggregate ---------------- +def aggregate(): + import statistics + res = {"judge": JUDGE_MODEL, "arms": {a: ARM_LABEL[a] for a in ARMS}} + # floor + fl = [json.loads(l) for l in (OUT/"votes_floor.jsonl").read_text().splitlines()] if (OUT/"votes_floor.jsonl").exists() else [] + tie = sum(1 for v in fl if v.get("winner") == "tie") + res["floor"] = dict(n=len(fl), tie=tie, tie_rate=round(tie/len(fl), 3) if fl else None, + non_tie=[{"unit": v["unit"], "order": v["order"], "winner": v["winner"]} for v in fl if v.get("winner") != "tie"]) + # catastrophe + ca = [json.loads(l) for l in (OUT/"votes_catastrophe.jsonl").read_text().splitlines()] if (OUT/"votes_catastrophe.jsonl").exists() else [] + cat_by_arm = {} + for a in ARMS: + rows = [v for v in ca if v["arm"] == a] + hits = [v for v in rows if v.get("catastrophe")] + cat_by_arm[a] = dict(n=len(rows), catastrophes=len(hits), + detail=[{"unit": v["unit"], "severity": v["severity"], "items": v.get("items"), "disposition": v.get("disposition")} for v in hits]) + res["catastrophe_screen"] = cat_by_arm + # pairwise + pw = [json.loads(l) for l in (OUT/"votes_pairwise.jsonl").read_text().splitlines()] if (OUT/"votes_pairwise.jsonl").exists() else [] + # per-arm points: win=1, tie=0.5 + points = {a: 0.0 for a in ARMS}; games = {a: 0 for a in ARMS} + pair_tally = {} + order_flip = 0; order_pairs = 0 + for v in pw: + x, y = v["pair"].split("-") + w = v["winner_arm"] + for a in (x, y): games[a] += 1 + if w == "tie": + points[x] += 0.5; points[y] += 0.5 + elif w in (x, y): + points[w] += 1.0 + pair_tally.setdefault(v["pair"], {x: 0.0, y: 0.0, "tie": 0}) + if w == "tie": pair_tally[v["pair"]]["tie"] += 1 + elif w in (x, y): pair_tally[v["pair"]][w] += 1 + # order consistency (noise proxy): per (pair,unit,rep) compare AB vs BA winner + by_cell = {} + for v in pw: + c = (v["pair"], v["unit"], v["rep"]) + by_cell.setdefault(c, {})[v["order"]] = v["winner_arm"] + for c, od in by_cell.items(): + if "AB" in od and "BA" in od: + order_pairs += 1 + if od["AB"] != od["BA"]: order_flip += 1 + res["pairwise"] = dict( + win_points={a: round(points[a], 1) for a in ARMS}, + games={a: games[a] for a in ARMS}, + win_rate={a: round(points[a]/games[a], 3) if games[a] else None for a in ARMS}, + pair_tally=pair_tally, + order_consistency=dict(cells=order_pairs, flips=order_flip, + flip_rate=round(order_flip/order_pairs, 3) if order_pairs else None), + cata_flags_relative={a: sum(1 for v in pw for side in (("A", v["posA"]), ("B", v["posB"])) + if side[1] == a and v.get(f"cata_{side[0]}")) for a in ARMS}, + ) + res["judges_used"] = {} + for v in fl + ca + pw: + j = v.get("judge") or "unknown" + res["judges_used"][j] = res["judges_used"].get(j, 0) + 1 + res["cost_usd"] = round(LEDGER.total, 4) + (OUT/"judge_results.json").write_text(json.dumps(res, ensure_ascii=False, indent=2)) + print(json.dumps(res, ensure_ascii=False, indent=2)) + return res + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--probe", action="store_true") + ap.add_argument("--run", action="store_true") + ap.add_argument("--aggregate", action="store_true") + a = ap.parse_args() + if a.probe: + sys.exit(0 if run_probe() else 1) + arms, keys = load_arms() + print(f"loaded {len(ARMS)} arms x {len(keys)} units") + if a.run: + run_floor(arms, keys) + run_catastrophe(arms, keys) + run_pairwise(arms, keys) + aggregate() + print(f"\nTOTAL judge cost: ${LEDGER.total:.4f}") + elif a.aggregate: + aggregate() + +if __name__ == "__main__": + main()