Land the rerun2 judge rig and blind-packet builder into eval: D39.7 full-window pairwise judging and D39.8 per-chapter-salted anonymization, outputs stay on the stand

This commit is contained in:
Claude (backend session) 2026-07-24 01:17:58 +03:00
parent cdf7bdd8f7
commit cd107e1281
2 changed files with 486 additions and 0 deletions

View file

@ -0,0 +1,105 @@
#!/usr/bin/env python3
"""D39.8 blind-read package builder for the pere-run (rerun2).
3 editor arms (glm-5 / mistral / deepseek-pro) of 蛊真人 ch1-5, zh->ru. Per CHAPTER, the 3 versions are
shuffled to neutral labels A/B/C with a PER-CHAPTER salt (so the reader cannot accumulate "C = one model"
across chapters). Source at the top of each chapter for fidelity comparison. Auto-QA (CJK-in-ru residual)
runs before the human. Correspondence key -> a SEPARATE file, withheld from the owner until verdict (D39.8).
"""
import json, random, re
from pathlib import Path
RUN = Path("/home/ubuntu/books/gu-zhenren/rerun2")
OUT = RUN / "blind"
OUT.mkdir(exist_ok=True)
ARMS = ["glm", "mistral", "dspro"]
def load():
arms = {}
for a in ARMS:
d = json.load(open(RUN / f"export-{a}.json"))
arms[a] = d["chunks"]
return arms
def by_chapter(chunks):
ch = {}
for c in sorted(chunks, key=lambda c: (int(c["chapter"]), int(c["chunk_idx"]))):
ch.setdefault(c["chapter"], {"final": [], "source": [], "disp": []})
ch[c["chapter"]]["final"].append(c["final_text"])
ch[c["chapter"]]["source"].append(c["source"])
ch[c["chapter"]]["disp"].append(c["disposition"])
return {k: {"final": "\n\n".join(v["final"]), "source": "\n\n".join(v["source"]),
"disp": v["disp"]} for k, v in ch.items()}
CJK = re.compile(r"[㐀-鿿豈-﫿]")
def auto_qa(text):
flags = []
cjk = CJK.findall(text)
if cjk:
flags.append(f"CJK-in-ru residual: {len(cjk)} char(s) e.g. {''.join(cjk[:6])}")
# crude latin-word leak (>=3 latin letters run, excluding common ok tokens)
lat = re.findall(r"[A-Za-z]{3,}", text)
if lat:
flags.append(f"latin run(s): {lat[:5]}")
return flags
def main():
arms = load()
chap = {a: by_chapter(arms[a]) for a in ARMS}
chapters = sorted(chap["glm"].keys(), key=int)
key = {}
packet = []
packet.append("# Слепой пакет чтения — пере-прогон rerun2 (蛊真人, главы 15, zh→ru)\n")
packet.append(
"Три версии перевода каждой главы (**A / B / C**) — это три редакторских «руки» одного и того же "
"чернового перевода. Метки **перетасованы заново в каждой главе** (в гл.1 «A» и в гл.2 «A» — это, "
"скорее всего, РАЗНЫЕ руки). Соответствие меток и моделей — в отдельном файле "
"`blind_read_KEY.json`, **его не смотреть до вердикта**.\n")
packet.append(
"**Как читать (планка: ≤2 претензии на версию):** по каждой главе отметьте для A/B/C претензии "
"по ярусам — **критические** (искажение смысла / пропуск / неверный род / непереведённый "
"китайский / сломанное слово), **смысловые** (неточность, единицы времени, числа), "
"**редакторские** (стиль, ритм, канцелярит) — и назовите предпочтительную версию.\n")
packet.append("---\n")
for ch in chapters:
rng = random.Random(f"rerun2-blind-ch{ch}")
order = ARMS[:]
rng.shuffle(order)
labels = ["A", "B", "C"]
key[ch] = {labels[i]: order[i] for i in range(3)}
src = chap[order[0]][ch]["source"] # same source across arms
packet.append(f"## Глава {ch}\n")
packet.append(f"<details><summary>Китайский исходник (для сверки верности)</summary>\n\n```\n{src}\n```\n</details>\n")
for i, lab in enumerate(labels):
arm = order[i]
txt = chap[arm][ch]["final"]
qa = auto_qa(txt)
qa_note = f"\n> _(авто-QA: {'; '.join(qa)})_\n" if qa else ""
packet.append(f"### Глава {ch} — версия {lab}\n{qa_note}\n{txt}\n")
packet.append(f"\n**Ваши претензии по главе {ch}:**\n"
"- Версия A — критич.: … · смысл.: … · редакт.: …\n"
"- Версия B — критич.: … · смысл.: … · редакт.: …\n"
"- Версия C — критич.: … · смысл.: … · редакт.: …\n"
"- Предпочтение: …\n\n---\n")
(OUT / "blind_read_packet.md").write_text("\n".join(packet))
(OUT / "blind_read_KEY.json").write_text(json.dumps(
{"note": "WITHHELD from owner until verdict (D39.8). Per-chapter label->arm mapping.",
"arm_label": {"glm": "glm-5 (base)", "mistral": "mistral-large-2512", "dspro": "deepseek-v4-pro"},
"mapping": key}, ensure_ascii=False, indent=2))
# QA summary (mine, not owner-facing)
qa_summary = {}
for ch in chapters:
for arm in ARMS:
f = auto_qa(chap[arm][ch]["final"])
if f: qa_summary.setdefault(arm, {})[ch] = f
(OUT / "auto_qa.json").write_text(json.dumps(qa_summary, ensure_ascii=False, indent=2))
print("packet:", OUT / "blind_read_packet.md")
print("key (withheld):", OUT / "blind_read_KEY.json")
print("chapters:", chapters)
print("auto-QA hits by arm:", {a: list(v.keys()) for a, v in qa_summary.items()})
if __name__ == "__main__":
main()

381
eval/rerun2_judge/judge.py Normal file
View file

@ -0,0 +1,381 @@
#!/usr/bin/env python3
"""D39.7 judge rig for the pere-run (rerun2) — 3 editor arms of 蛊真人 ch1-5, zh->ru.
Judge = gemini-3.1-pro-preview (cross-family to the zh->ru editors glm-5/mistral/deepseek-pro,
so no author==reviewer). Full windows (whole source unit + whole both translations, NO truncation).
Per-vote JSONL persistence + resume. Position-swap (both orders). Per-arm catastrophe screen (ALL
arms incl. the favorite). Identical-text floor pass (judge-noise baseline). Hard cost cap.
Metrics NOMINATE; the owner blind read RATIFIES. Backend deterministic signals (echo/flags) are
reported ALONGSIDE, never fused (no manufactured convergence).
Usage:
eval/.venv/bin/python eval/rerun2_judge/judge.py --probe # live-probe gemini slug ($ ~0.001)
eval/.venv/bin/python eval/rerun2_judge/judge.py --run # full rig
eval/.venv/bin/python eval/rerun2_judge/judge.py --aggregate # re-aggregate from JSONL only ($0)
"""
import os, sys, json, time, argparse, itertools
from pathlib import Path
HERE = Path(__file__).resolve().parent
EVAL = HERE.parent
RUN = Path("/home/ubuntu/books/gu-zhenren/rerun2")
OUT = RUN / "judge"
OUT.mkdir(exist_ok=True)
# --- keys: load eval/.env via dotenv (we never read .env ourselves; the lib does) ---
try:
from dotenv import load_dotenv
load_dotenv(EVAL / ".env")
except Exception:
pass
from openai import OpenAI # openai 2.44.0 in eval/.venv
# Judge providers: gemini primary; grok fallback for units gemini refuses with PROHIBITED_CONTENT
# (Gemini 3.x fail-closes on sexual content — non-configurable, D22.6; erotica judge = Grok). Grok is
# cross-family to the zh->ru editors (glm/mistral/deepseek) so no author==reviewer.
PROVIDERS = {
"gemini": dict(model="gemini-3.1-pro-preview", base="https://generativelanguage.googleapis.com/v1beta/openai",
key_env="GEMINI_API_KEY", max_tokens=20000, temperature=0.0,
price_in=2.0, price_out=12.0, reasoning="from_total"),
"grok": dict(model="grok-4.3", base="https://api.x.ai/v1",
key_env="XAI_API_KEY", max_tokens=8000, temperature=0.0,
price_in=1.25, price_out=2.50, reasoning="field"),
}
PRIMARY, FALLBACK = "gemini", "grok"
JUDGE_MODEL = PROVIDERS[PRIMARY]["model"] # display/aggregate reference
COST_CAP_USD = 8.0 # hard internal ceiling (well under the $15 experiment cap)
ARMS = ["glm", "mistral", "dspro"]
ARM_LABEL = {"glm": "glm-5", "mistral": "mistral-large-2512", "dspro": "deepseek-v4-pro"}
REPS = 2 # repeats per (pair, unit, order) — 2 orders x 2 reps = 4 votes/pair-unit
_clients = {}
def client(prov):
if prov not in _clients:
cfg = PROVIDERS[prov]
key = os.environ.get(cfg["key_env"])
if not key:
sys.exit(f"FATAL: {cfg['key_env']} not set (eval/.env not loaded?)")
_clients[prov] = OpenAI(base_url=cfg["base"], api_key=key, timeout=240)
return _clients[prov]
class Ledger:
def __init__(self, path):
self.path = path
self.total = 0.0
if path.exists():
for ln in path.read_text().splitlines():
try: self.total += json.loads(ln).get("cost_usd", 0.0)
except Exception: pass
def add(self, rec):
self.total += rec.get("cost_usd", 0.0)
with open(self.path, "a") as f:
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
LEDGER = Ledger(OUT / "ledger.jsonl")
def _reasoning_tokens(u, mode):
if mode == "from_total": # gemini: thinking only in total_tokens
pt = getattr(u, "prompt_tokens", 0) or 0
ct = getattr(u, "completion_tokens", 0) or 0
tt = getattr(u, "total_tokens", 0) or 0
return max(0, tt - pt - ct)
det = getattr(u, "completion_tokens_details", None) # grok/openai: reasoning_tokens field (additive)
return (getattr(det, "reasoning_tokens", 0) if det else 0) or getattr(u, "reasoning_tokens", 0) or 0
def _raw_call(prov, system, user, tag):
"""One provider call. Returns (text, rec). Ledgers cost. Retries transient; empty on hard fail."""
if LEDGER.total >= COST_CAP_USD:
sys.exit(f"FATAL: cost cap ${COST_CAP_USD} reached (spent ${LEDGER.total:.4f}) — STOP")
cfg = PROVIDERS[prov]
backoff = [5, 15, 35, 60]
for attempt in range(len(backoff) + 1):
try:
r = client(prov).chat.completions.create(
model=cfg["model"],
messages=[{"role": "system", "content": system},
{"role": "user", "content": user}],
temperature=cfg["temperature"], max_tokens=cfg["max_tokens"],
)
ch = r.choices[0]
finish = ch.finish_reason
m = getattr(ch, "message", None)
txt = ((m.content if m is not None else None) or "").strip()
u = r.usage
pt = getattr(u, "prompt_tokens", 0) or 0
ct = getattr(u, "completion_tokens", 0) or 0
tt = getattr(u, "total_tokens", 0) or 0
reasoning = _reasoning_tokens(u, cfg["reasoning"])
out_billed = ct + reasoning
cost = pt * cfg["price_in"] / 1e6 + out_billed * cfg["price_out"] / 1e6
rec = dict(tag=tag, judge=cfg["model"], finish=finish, prompt_tok=pt, completion_tok=ct,
reasoning_tok=reasoning, total_tok=tt, cost_usd=cost, empty=(not txt))
LEDGER.add(rec)
if finish != "stop":
rec["ANOMALY"] = f"finish_reason={finish!r}"
print(f" [{prov} {tag}] finish={finish!r} empty={not txt}")
return txt, rec
except Exception as e:
if attempt < len(backoff):
print(f" [retry {prov} {tag}] {str(e)[:100]} — sleep {backoff[attempt]}s")
time.sleep(backoff[attempt])
else:
print(f" [FAIL {prov} {tag}] {str(e)[:180]}")
LEDGER.add(dict(tag=tag, judge=cfg["model"], error=str(e)[:300], cost_usd=0.0))
return "", dict(tag=tag, judge=cfg["model"], error=str(e)[:300], cost_usd=0.0)
def _refused(rec, txt):
f = str(rec.get("finish") or "")
return (not txt) and ("content_filter" in f or "PROHIBITED" in f or "SAFETY" in f)
def gemini_call(system, user, tag):
"""Judge with gemini primary; on PROHIBITED_CONTENT refusal, fall back to grok (erotica judge, D22.6)."""
txt, rec = _raw_call(PRIMARY, system, user, tag)
if _refused(rec, txt):
print(f" [fallback->grok {tag}] gemini {rec.get('finish')}")
txt2, rec2 = _raw_call(FALLBACK, system, user, tag + "::grok")
rec2["fallback_from"] = f"gemini:{rec.get('finish')}"
return txt2, rec2
return txt, rec
# ---------------- data ----------------
def load_arms():
arms = {}
for a in ARMS:
d = json.load(open(RUN / f"export-{a}.json"))
by = {}
for c in d["chunks"]:
by[(c["chapter"], c["chunk_idx"])] = c
arms[a] = by
keys = sorted(set().union(*[set(v) for v in arms.values()]),
key=lambda k: (int(k[0]), int(k[1])))
return arms, keys
# ---------------- prompts ----------------
PAIR_SYS = (
"Ты — строгий эксперт по художественному переводу с китайского на русский (веб-новелла, жанр сянься). "
"Тебе дают КИТАЙСКИЙ исходник и ДВА русских перевода: A и B. Оцени, какой перевод лучше как "
"ХУДОЖЕСТВЕННЫЙ русский текст, по трём осям в порядке важности: (1) ВЕРНОСТЬ — смысл исходника "
"передан без искажений, пропусков и отсебятины; (2) ЕСТЕСТВЕННЫЙ ЛИТЕРАТУРНЫЙ РУССКИЙ — читается как "
"родная русская проза, без переводческого канцелярита и кальки; (3) СОГЛАСОВАННОСТЬ имён/терминов. "
"Отметь КАТАСТРОФУ у стороны, если есть: искажение смысла, пропуск предложения, неверный род персонажа, "
"непереведённый китайский в тексте, сломанная русская морфология. "
"Ответь СТРОГИМ JSON одной строкой, без пояснений вокруг:\n"
'{"winner":"A|B|tie","margin":"clear|slight","reason":"<=25 слов, с краткой цитатой-уликой>",'
'"catastrophe_A":true|false,"catastrophe_B":true|false,"catastrophe_detail":"<кратко или пусто>"}')
CAT_SYS = (
"Ты — строгий редактор-контролёр перевода с китайского на русский. Тебе дают КИТАЙСКИЙ исходник и ОДИН "
"русский перевод. Найди КАТАСТРОФИЧЕСКИЕ дефекты (те, из-за которых взыскательный читатель забракует "
"отрывок): искажение смысла, пропуск целого предложения/абзаца, неверный род персонажа, непереведённый "
"китайский (иероглифы) в русском тексте, сломанная русская словоформа, грубая ошибка в числах/единицах "
"времени. Мелкие стилистические придирки НЕ катастрофа. "
"Ответь СТРОГИМ JSON одной строкой:\n"
'{"catastrophe":true|false,"severity":"none|minor|major|critical","items":["<кратко с цитатой>", ...]}')
def pair_user(src, ta, tb):
return (f"=== КИТАЙСКИЙ ИСХОДНИК ===\n{src}\n\n=== ПЕРЕВОД A ===\n{ta}\n\n=== ПЕРЕВОД B ===\n{tb}\n\n"
"Верни JSON-вердикт.")
def cat_user(src, t):
return f"=== КИТАЙСКИЙ ИСХОДНИК ===\n{src}\n\n=== РУССКИЙ ПЕРЕВОД ===\n{t}\n\nВерни JSON."
def parse_json(txt):
if not txt: return None
s = txt.strip()
if s.startswith("```"):
s = s.strip("`")
s = s[s.find("{"):]
i, j = s.find("{"), s.rfind("}")
if i < 0 or j < 0: return None
try: return json.loads(s[i:j+1])
except Exception: return None
# ---------------- resumable vote store ----------------
def load_votes(path):
seen = {}
if path.exists():
for ln in path.read_text().splitlines():
try:
v = json.loads(ln)
seen[v["vote_id"]] = v
except Exception: pass
return seen
def append_vote(path, v):
with open(path, "a") as f:
f.write(json.dumps(v, ensure_ascii=False) + "\n")
# ---------------- passes ----------------
def run_probe():
print(f"=== LIVE PROBE {JUDGE_MODEL} ===")
txt, rec = gemini_call(
"Ты judge. Ответь строгим JSON.",
'Верни ровно: {"ok":true,"lang":"ru"}',
"probe")
print("finish:", rec.get("finish"), "| cost $", round(rec.get("cost_usd", 0), 5),
"| tokens p/c/r:", rec.get("prompt_tok"), rec.get("completion_tok"), rec.get("reasoning_tok"))
print("response:", txt[:200])
ok = rec.get("finish") == "stop" and parse_json(txt) is not None
print("PROBE", "OK" if ok else "FAILED (check slug/vendor doc per two-directions rule)")
return ok
def run_floor(arms, keys):
"""Identical-text position-bias floor: judge glm vs glm (same text), both orders.
Deviation from 'tie' = judge noise. (Full A0<->A0' independent-regen floor deferred; owner read ratifies.)"""
path = OUT / "votes_floor.jsonl"
seen = load_votes(path)
print(f"=== FLOOR pass (glm vs glm identical), {len(keys)} units x 2 orders ===")
for k in keys:
c = arms["glm"][k]
for order in ("AB", "BA"):
vid = f"floor::{k[0]}-{k[1]}::{order}"
if vid in seen: continue
txt, rec = gemini_call(PAIR_SYS, pair_user(c["source"], c["final_text"], c["final_text"]),
vid)
v = parse_json(txt) or {}
rec_v = dict(vote_id=vid, kind="floor", unit=f"{k[0]}-{k[1]}", order=order,
winner=v.get("winner"), margin=v.get("margin"), reason=v.get("reason"),
raw=txt[:400], finish=rec.get("finish"), judge=rec.get("judge"))
append_vote(path, rec_v)
print(f" floor {k[0]}-{k[1]} {order}: winner={v.get('winner')} (identical) ${LEDGER.total:.3f}")
def run_catastrophe(arms, keys):
"""Per-arm absolute catastrophe screen — ALL arms incl. favorite."""
path = OUT / "votes_catastrophe.jsonl"
seen = load_votes(path)
print(f"=== CATASTROPHE screen, {len(keys)} units x {len(ARMS)} arms ===")
for a in ARMS:
for k in keys:
c = arms[a][k]
vid = f"cat::{a}::{k[0]}-{k[1]}"
if vid in seen: continue
txt, rec = gemini_call(CAT_SYS, cat_user(c["source"], c["final_text"]), vid)
v = parse_json(txt) or {}
rec_v = dict(vote_id=vid, kind="catastrophe", arm=a, unit=f"{k[0]}-{k[1]}",
disposition=c.get("disposition"),
catastrophe=bool(v.get("catastrophe")), severity=v.get("severity"),
items=v.get("items"), raw=txt[:500], finish=rec.get("finish"), judge=rec.get("judge"))
append_vote(path, rec_v)
flag = "CATA" if v.get("catastrophe") else "ok"
print(f" cat {a} {k[0]}-{k[1]}: {flag} ({v.get('severity')}) ${LEDGER.total:.3f}")
def run_pairwise(arms, keys):
"""Round-robin pairwise, both orders, REPS reps. Full windows. Order-normalized winner."""
path = OUT / "votes_pairwise.jsonl"
seen = load_votes(path)
pairs = list(itertools.combinations(ARMS, 2))
print(f"=== PAIRWISE, {len(pairs)} pairs x {len(keys)} units x 2 orders x {REPS} reps ===")
for (x, y) in pairs:
for k in keys:
cx, cy = arms[x][k], arms[y][k]
for order in ("AB", "BA"):
A, B = (x, y) if order == "AB" else (y, x)
cA, cB = arms[A][k], arms[B][k]
for rep in range(REPS):
vid = f"pair::{x}-{y}::{k[0]}-{k[1]}::{order}::r{rep}"
if vid in seen: continue
txt, rec = gemini_call(PAIR_SYS, pair_user(cA["source"], cA["final_text"], cB["final_text"]),
vid)
v = parse_json(txt) or {}
w = v.get("winner")
# normalize winner label (A/B) -> actual arm
if w == "A": win_arm = A
elif w == "B": win_arm = B
elif w == "tie": win_arm = "tie"
else: win_arm = None
rec_v = dict(vote_id=vid, kind="pairwise", pair=f"{x}-{y}", unit=f"{k[0]}-{k[1]}",
order=order, rep=rep, posA=A, posB=B,
winner_arm=win_arm, margin=v.get("margin"), reason=v.get("reason"),
cata_A=bool(v.get("catastrophe_A")), cata_B=bool(v.get("catastrophe_B")),
cata_detail=v.get("catastrophe_detail"), raw=txt[:400],
finish=rec.get("finish"), judge=rec.get("judge"))
append_vote(path, rec_v)
print(f" pair {x}-{y} {k[0]}-{k[1]} {order} r{rep}: win={win_arm} ({v.get('margin')}) ${LEDGER.total:.3f}")
# ---------------- aggregate ----------------
def aggregate():
import statistics
res = {"judge": JUDGE_MODEL, "arms": {a: ARM_LABEL[a] for a in ARMS}}
# floor
fl = [json.loads(l) for l in (OUT/"votes_floor.jsonl").read_text().splitlines()] if (OUT/"votes_floor.jsonl").exists() else []
tie = sum(1 for v in fl if v.get("winner") == "tie")
res["floor"] = dict(n=len(fl), tie=tie, tie_rate=round(tie/len(fl), 3) if fl else None,
non_tie=[{"unit": v["unit"], "order": v["order"], "winner": v["winner"]} for v in fl if v.get("winner") != "tie"])
# catastrophe
ca = [json.loads(l) for l in (OUT/"votes_catastrophe.jsonl").read_text().splitlines()] if (OUT/"votes_catastrophe.jsonl").exists() else []
cat_by_arm = {}
for a in ARMS:
rows = [v for v in ca if v["arm"] == a]
hits = [v for v in rows if v.get("catastrophe")]
cat_by_arm[a] = dict(n=len(rows), catastrophes=len(hits),
detail=[{"unit": v["unit"], "severity": v["severity"], "items": v.get("items"), "disposition": v.get("disposition")} for v in hits])
res["catastrophe_screen"] = cat_by_arm
# pairwise
pw = [json.loads(l) for l in (OUT/"votes_pairwise.jsonl").read_text().splitlines()] if (OUT/"votes_pairwise.jsonl").exists() else []
# per-arm points: win=1, tie=0.5
points = {a: 0.0 for a in ARMS}; games = {a: 0 for a in ARMS}
pair_tally = {}
order_flip = 0; order_pairs = 0
for v in pw:
x, y = v["pair"].split("-")
w = v["winner_arm"]
for a in (x, y): games[a] += 1
if w == "tie":
points[x] += 0.5; points[y] += 0.5
elif w in (x, y):
points[w] += 1.0
pair_tally.setdefault(v["pair"], {x: 0.0, y: 0.0, "tie": 0})
if w == "tie": pair_tally[v["pair"]]["tie"] += 1
elif w in (x, y): pair_tally[v["pair"]][w] += 1
# order consistency (noise proxy): per (pair,unit,rep) compare AB vs BA winner
by_cell = {}
for v in pw:
c = (v["pair"], v["unit"], v["rep"])
by_cell.setdefault(c, {})[v["order"]] = v["winner_arm"]
for c, od in by_cell.items():
if "AB" in od and "BA" in od:
order_pairs += 1
if od["AB"] != od["BA"]: order_flip += 1
res["pairwise"] = dict(
win_points={a: round(points[a], 1) for a in ARMS},
games={a: games[a] for a in ARMS},
win_rate={a: round(points[a]/games[a], 3) if games[a] else None for a in ARMS},
pair_tally=pair_tally,
order_consistency=dict(cells=order_pairs, flips=order_flip,
flip_rate=round(order_flip/order_pairs, 3) if order_pairs else None),
cata_flags_relative={a: sum(1 for v in pw for side in (("A", v["posA"]), ("B", v["posB"]))
if side[1] == a and v.get(f"cata_{side[0]}")) for a in ARMS},
)
res["judges_used"] = {}
for v in fl + ca + pw:
j = v.get("judge") or "unknown"
res["judges_used"][j] = res["judges_used"].get(j, 0) + 1
res["cost_usd"] = round(LEDGER.total, 4)
(OUT/"judge_results.json").write_text(json.dumps(res, ensure_ascii=False, indent=2))
print(json.dumps(res, ensure_ascii=False, indent=2))
return res
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--probe", action="store_true")
ap.add_argument("--run", action="store_true")
ap.add_argument("--aggregate", action="store_true")
a = ap.parse_args()
if a.probe:
sys.exit(0 if run_probe() else 1)
arms, keys = load_arms()
print(f"loaded {len(ARMS)} arms x {len(keys)} units")
if a.run:
run_floor(arms, keys)
run_catastrophe(arms, keys)
run_pairwise(arms, keys)
aggregate()
print(f"\nTOTAL judge cost: ${LEDGER.total:.4f}")
elif a.aggregate:
aggregate()
if __name__ == "__main__":
main()