Land the rerun2 judge rig and blind-packet builder into eval: D39.7 full-window pairwise judging and D39.8 per-chapter-salted anonymization, outputs stay on the stand
This commit is contained in:
parent
cdf7bdd8f7
commit
cd107e1281
2 changed files with 486 additions and 0 deletions
105
eval/rerun2_judge/build_blind.py
Normal file
105
eval/rerun2_judge/build_blind.py
Normal file
|
|
@ -0,0 +1,105 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""D39.8 blind-read package builder for the pere-run (rerun2).
|
||||||
|
|
||||||
|
3 editor arms (glm-5 / mistral / deepseek-pro) of 蛊真人 ch1-5, zh->ru. Per CHAPTER, the 3 versions are
|
||||||
|
shuffled to neutral labels A/B/C with a PER-CHAPTER salt (so the reader cannot accumulate "C = one model"
|
||||||
|
across chapters). Source at the top of each chapter for fidelity comparison. Auto-QA (CJK-in-ru residual)
|
||||||
|
runs before the human. Correspondence key -> a SEPARATE file, withheld from the owner until verdict (D39.8).
|
||||||
|
"""
|
||||||
|
import json, random, re
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
RUN = Path("/home/ubuntu/books/gu-zhenren/rerun2")
|
||||||
|
OUT = RUN / "blind"
|
||||||
|
OUT.mkdir(exist_ok=True)
|
||||||
|
ARMS = ["glm", "mistral", "dspro"]
|
||||||
|
|
||||||
|
def load():
|
||||||
|
arms = {}
|
||||||
|
for a in ARMS:
|
||||||
|
d = json.load(open(RUN / f"export-{a}.json"))
|
||||||
|
arms[a] = d["chunks"]
|
||||||
|
return arms
|
||||||
|
|
||||||
|
def by_chapter(chunks):
|
||||||
|
ch = {}
|
||||||
|
for c in sorted(chunks, key=lambda c: (int(c["chapter"]), int(c["chunk_idx"]))):
|
||||||
|
ch.setdefault(c["chapter"], {"final": [], "source": [], "disp": []})
|
||||||
|
ch[c["chapter"]]["final"].append(c["final_text"])
|
||||||
|
ch[c["chapter"]]["source"].append(c["source"])
|
||||||
|
ch[c["chapter"]]["disp"].append(c["disposition"])
|
||||||
|
return {k: {"final": "\n\n".join(v["final"]), "source": "\n\n".join(v["source"]),
|
||||||
|
"disp": v["disp"]} for k, v in ch.items()}
|
||||||
|
|
||||||
|
CJK = re.compile(r"[㐀-鿿豈-]")
|
||||||
|
def auto_qa(text):
|
||||||
|
flags = []
|
||||||
|
cjk = CJK.findall(text)
|
||||||
|
if cjk:
|
||||||
|
flags.append(f"CJK-in-ru residual: {len(cjk)} char(s) e.g. {''.join(cjk[:6])}")
|
||||||
|
# crude latin-word leak (>=3 latin letters run, excluding common ok tokens)
|
||||||
|
lat = re.findall(r"[A-Za-z]{3,}", text)
|
||||||
|
if lat:
|
||||||
|
flags.append(f"latin run(s): {lat[:5]}")
|
||||||
|
return flags
|
||||||
|
|
||||||
|
def main():
|
||||||
|
arms = load()
|
||||||
|
chap = {a: by_chapter(arms[a]) for a in ARMS}
|
||||||
|
chapters = sorted(chap["glm"].keys(), key=int)
|
||||||
|
|
||||||
|
key = {}
|
||||||
|
packet = []
|
||||||
|
packet.append("# Слепой пакет чтения — пере-прогон rerun2 (蛊真人, главы 1–5, zh→ru)\n")
|
||||||
|
packet.append(
|
||||||
|
"Три версии перевода каждой главы (**A / B / C**) — это три редакторских «руки» одного и того же "
|
||||||
|
"чернового перевода. Метки **перетасованы заново в каждой главе** (в гл.1 «A» и в гл.2 «A» — это, "
|
||||||
|
"скорее всего, РАЗНЫЕ руки). Соответствие меток и моделей — в отдельном файле "
|
||||||
|
"`blind_read_KEY.json`, **его не смотреть до вердикта**.\n")
|
||||||
|
packet.append(
|
||||||
|
"**Как читать (планка: ≤2 претензии на версию):** по каждой главе отметьте для A/B/C претензии "
|
||||||
|
"по ярусам — **критические** (искажение смысла / пропуск / неверный род / непереведённый "
|
||||||
|
"китайский / сломанное слово), **смысловые** (неточность, единицы времени, числа), "
|
||||||
|
"**редакторские** (стиль, ритм, канцелярит) — и назовите предпочтительную версию.\n")
|
||||||
|
packet.append("---\n")
|
||||||
|
|
||||||
|
for ch in chapters:
|
||||||
|
rng = random.Random(f"rerun2-blind-ch{ch}")
|
||||||
|
order = ARMS[:]
|
||||||
|
rng.shuffle(order)
|
||||||
|
labels = ["A", "B", "C"]
|
||||||
|
key[ch] = {labels[i]: order[i] for i in range(3)}
|
||||||
|
src = chap[order[0]][ch]["source"] # same source across arms
|
||||||
|
packet.append(f"## Глава {ch}\n")
|
||||||
|
packet.append(f"<details><summary>Китайский исходник (для сверки верности)</summary>\n\n```\n{src}\n```\n</details>\n")
|
||||||
|
for i, lab in enumerate(labels):
|
||||||
|
arm = order[i]
|
||||||
|
txt = chap[arm][ch]["final"]
|
||||||
|
qa = auto_qa(txt)
|
||||||
|
qa_note = f"\n> _(авто-QA: {'; '.join(qa)})_\n" if qa else ""
|
||||||
|
packet.append(f"### Глава {ch} — версия {lab}\n{qa_note}\n{txt}\n")
|
||||||
|
packet.append(f"\n**Ваши претензии по главе {ch}:**\n"
|
||||||
|
"- Версия A — критич.: … · смысл.: … · редакт.: …\n"
|
||||||
|
"- Версия B — критич.: … · смысл.: … · редакт.: …\n"
|
||||||
|
"- Версия C — критич.: … · смысл.: … · редакт.: …\n"
|
||||||
|
"- Предпочтение: …\n\n---\n")
|
||||||
|
|
||||||
|
(OUT / "blind_read_packet.md").write_text("\n".join(packet))
|
||||||
|
(OUT / "blind_read_KEY.json").write_text(json.dumps(
|
||||||
|
{"note": "WITHHELD from owner until verdict (D39.8). Per-chapter label->arm mapping.",
|
||||||
|
"arm_label": {"glm": "glm-5 (base)", "mistral": "mistral-large-2512", "dspro": "deepseek-v4-pro"},
|
||||||
|
"mapping": key}, ensure_ascii=False, indent=2))
|
||||||
|
# QA summary (mine, not owner-facing)
|
||||||
|
qa_summary = {}
|
||||||
|
for ch in chapters:
|
||||||
|
for arm in ARMS:
|
||||||
|
f = auto_qa(chap[arm][ch]["final"])
|
||||||
|
if f: qa_summary.setdefault(arm, {})[ch] = f
|
||||||
|
(OUT / "auto_qa.json").write_text(json.dumps(qa_summary, ensure_ascii=False, indent=2))
|
||||||
|
print("packet:", OUT / "blind_read_packet.md")
|
||||||
|
print("key (withheld):", OUT / "blind_read_KEY.json")
|
||||||
|
print("chapters:", chapters)
|
||||||
|
print("auto-QA hits by arm:", {a: list(v.keys()) for a, v in qa_summary.items()})
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
381
eval/rerun2_judge/judge.py
Normal file
381
eval/rerun2_judge/judge.py
Normal file
|
|
@ -0,0 +1,381 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""D39.7 judge rig for the pere-run (rerun2) — 3 editor arms of 蛊真人 ch1-5, zh->ru.
|
||||||
|
|
||||||
|
Judge = gemini-3.1-pro-preview (cross-family to the zh->ru editors glm-5/mistral/deepseek-pro,
|
||||||
|
so no author==reviewer). Full windows (whole source unit + whole both translations, NO truncation).
|
||||||
|
Per-vote JSONL persistence + resume. Position-swap (both orders). Per-arm catastrophe screen (ALL
|
||||||
|
arms incl. the favorite). Identical-text floor pass (judge-noise baseline). Hard cost cap.
|
||||||
|
|
||||||
|
Metrics NOMINATE; the owner blind read RATIFIES. Backend deterministic signals (echo/flags) are
|
||||||
|
reported ALONGSIDE, never fused (no manufactured convergence).
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
eval/.venv/bin/python eval/rerun2_judge/judge.py --probe # live-probe gemini slug ($ ~0.001)
|
||||||
|
eval/.venv/bin/python eval/rerun2_judge/judge.py --run # full rig
|
||||||
|
eval/.venv/bin/python eval/rerun2_judge/judge.py --aggregate # re-aggregate from JSONL only ($0)
|
||||||
|
"""
|
||||||
|
import os, sys, json, time, argparse, itertools
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
HERE = Path(__file__).resolve().parent
|
||||||
|
EVAL = HERE.parent
|
||||||
|
RUN = Path("/home/ubuntu/books/gu-zhenren/rerun2")
|
||||||
|
OUT = RUN / "judge"
|
||||||
|
OUT.mkdir(exist_ok=True)
|
||||||
|
|
||||||
|
# --- keys: load eval/.env via dotenv (we never read .env ourselves; the lib does) ---
|
||||||
|
try:
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
load_dotenv(EVAL / ".env")
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
from openai import OpenAI # openai 2.44.0 in eval/.venv
|
||||||
|
|
||||||
|
# Judge providers: gemini primary; grok fallback for units gemini refuses with PROHIBITED_CONTENT
|
||||||
|
# (Gemini 3.x fail-closes on sexual content — non-configurable, D22.6; erotica judge = Grok). Grok is
|
||||||
|
# cross-family to the zh->ru editors (glm/mistral/deepseek) so no author==reviewer.
|
||||||
|
PROVIDERS = {
|
||||||
|
"gemini": dict(model="gemini-3.1-pro-preview", base="https://generativelanguage.googleapis.com/v1beta/openai",
|
||||||
|
key_env="GEMINI_API_KEY", max_tokens=20000, temperature=0.0,
|
||||||
|
price_in=2.0, price_out=12.0, reasoning="from_total"),
|
||||||
|
"grok": dict(model="grok-4.3", base="https://api.x.ai/v1",
|
||||||
|
key_env="XAI_API_KEY", max_tokens=8000, temperature=0.0,
|
||||||
|
price_in=1.25, price_out=2.50, reasoning="field"),
|
||||||
|
}
|
||||||
|
PRIMARY, FALLBACK = "gemini", "grok"
|
||||||
|
JUDGE_MODEL = PROVIDERS[PRIMARY]["model"] # display/aggregate reference
|
||||||
|
COST_CAP_USD = 8.0 # hard internal ceiling (well under the $15 experiment cap)
|
||||||
|
|
||||||
|
ARMS = ["glm", "mistral", "dspro"]
|
||||||
|
ARM_LABEL = {"glm": "glm-5", "mistral": "mistral-large-2512", "dspro": "deepseek-v4-pro"}
|
||||||
|
REPS = 2 # repeats per (pair, unit, order) — 2 orders x 2 reps = 4 votes/pair-unit
|
||||||
|
|
||||||
|
_clients = {}
|
||||||
|
def client(prov):
|
||||||
|
if prov not in _clients:
|
||||||
|
cfg = PROVIDERS[prov]
|
||||||
|
key = os.environ.get(cfg["key_env"])
|
||||||
|
if not key:
|
||||||
|
sys.exit(f"FATAL: {cfg['key_env']} not set (eval/.env not loaded?)")
|
||||||
|
_clients[prov] = OpenAI(base_url=cfg["base"], api_key=key, timeout=240)
|
||||||
|
return _clients[prov]
|
||||||
|
|
||||||
|
class Ledger:
|
||||||
|
def __init__(self, path):
|
||||||
|
self.path = path
|
||||||
|
self.total = 0.0
|
||||||
|
if path.exists():
|
||||||
|
for ln in path.read_text().splitlines():
|
||||||
|
try: self.total += json.loads(ln).get("cost_usd", 0.0)
|
||||||
|
except Exception: pass
|
||||||
|
def add(self, rec):
|
||||||
|
self.total += rec.get("cost_usd", 0.0)
|
||||||
|
with open(self.path, "a") as f:
|
||||||
|
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
||||||
|
|
||||||
|
LEDGER = Ledger(OUT / "ledger.jsonl")
|
||||||
|
|
||||||
|
def _reasoning_tokens(u, mode):
|
||||||
|
if mode == "from_total": # gemini: thinking only in total_tokens
|
||||||
|
pt = getattr(u, "prompt_tokens", 0) or 0
|
||||||
|
ct = getattr(u, "completion_tokens", 0) or 0
|
||||||
|
tt = getattr(u, "total_tokens", 0) or 0
|
||||||
|
return max(0, tt - pt - ct)
|
||||||
|
det = getattr(u, "completion_tokens_details", None) # grok/openai: reasoning_tokens field (additive)
|
||||||
|
return (getattr(det, "reasoning_tokens", 0) if det else 0) or getattr(u, "reasoning_tokens", 0) or 0
|
||||||
|
|
||||||
|
def _raw_call(prov, system, user, tag):
|
||||||
|
"""One provider call. Returns (text, rec). Ledgers cost. Retries transient; empty on hard fail."""
|
||||||
|
if LEDGER.total >= COST_CAP_USD:
|
||||||
|
sys.exit(f"FATAL: cost cap ${COST_CAP_USD} reached (spent ${LEDGER.total:.4f}) — STOP")
|
||||||
|
cfg = PROVIDERS[prov]
|
||||||
|
backoff = [5, 15, 35, 60]
|
||||||
|
for attempt in range(len(backoff) + 1):
|
||||||
|
try:
|
||||||
|
r = client(prov).chat.completions.create(
|
||||||
|
model=cfg["model"],
|
||||||
|
messages=[{"role": "system", "content": system},
|
||||||
|
{"role": "user", "content": user}],
|
||||||
|
temperature=cfg["temperature"], max_tokens=cfg["max_tokens"],
|
||||||
|
)
|
||||||
|
ch = r.choices[0]
|
||||||
|
finish = ch.finish_reason
|
||||||
|
m = getattr(ch, "message", None)
|
||||||
|
txt = ((m.content if m is not None else None) or "").strip()
|
||||||
|
u = r.usage
|
||||||
|
pt = getattr(u, "prompt_tokens", 0) or 0
|
||||||
|
ct = getattr(u, "completion_tokens", 0) or 0
|
||||||
|
tt = getattr(u, "total_tokens", 0) or 0
|
||||||
|
reasoning = _reasoning_tokens(u, cfg["reasoning"])
|
||||||
|
out_billed = ct + reasoning
|
||||||
|
cost = pt * cfg["price_in"] / 1e6 + out_billed * cfg["price_out"] / 1e6
|
||||||
|
rec = dict(tag=tag, judge=cfg["model"], finish=finish, prompt_tok=pt, completion_tok=ct,
|
||||||
|
reasoning_tok=reasoning, total_tok=tt, cost_usd=cost, empty=(not txt))
|
||||||
|
LEDGER.add(rec)
|
||||||
|
if finish != "stop":
|
||||||
|
rec["ANOMALY"] = f"finish_reason={finish!r}"
|
||||||
|
print(f" [{prov} {tag}] finish={finish!r} empty={not txt}")
|
||||||
|
return txt, rec
|
||||||
|
except Exception as e:
|
||||||
|
if attempt < len(backoff):
|
||||||
|
print(f" [retry {prov} {tag}] {str(e)[:100]} — sleep {backoff[attempt]}s")
|
||||||
|
time.sleep(backoff[attempt])
|
||||||
|
else:
|
||||||
|
print(f" [FAIL {prov} {tag}] {str(e)[:180]}")
|
||||||
|
LEDGER.add(dict(tag=tag, judge=cfg["model"], error=str(e)[:300], cost_usd=0.0))
|
||||||
|
return "", dict(tag=tag, judge=cfg["model"], error=str(e)[:300], cost_usd=0.0)
|
||||||
|
|
||||||
|
def _refused(rec, txt):
|
||||||
|
f = str(rec.get("finish") or "")
|
||||||
|
return (not txt) and ("content_filter" in f or "PROHIBITED" in f or "SAFETY" in f)
|
||||||
|
|
||||||
|
def gemini_call(system, user, tag):
|
||||||
|
"""Judge with gemini primary; on PROHIBITED_CONTENT refusal, fall back to grok (erotica judge, D22.6)."""
|
||||||
|
txt, rec = _raw_call(PRIMARY, system, user, tag)
|
||||||
|
if _refused(rec, txt):
|
||||||
|
print(f" [fallback->grok {tag}] gemini {rec.get('finish')}")
|
||||||
|
txt2, rec2 = _raw_call(FALLBACK, system, user, tag + "::grok")
|
||||||
|
rec2["fallback_from"] = f"gemini:{rec.get('finish')}"
|
||||||
|
return txt2, rec2
|
||||||
|
return txt, rec
|
||||||
|
|
||||||
|
# ---------------- data ----------------
|
||||||
|
def load_arms():
|
||||||
|
arms = {}
|
||||||
|
for a in ARMS:
|
||||||
|
d = json.load(open(RUN / f"export-{a}.json"))
|
||||||
|
by = {}
|
||||||
|
for c in d["chunks"]:
|
||||||
|
by[(c["chapter"], c["chunk_idx"])] = c
|
||||||
|
arms[a] = by
|
||||||
|
keys = sorted(set().union(*[set(v) for v in arms.values()]),
|
||||||
|
key=lambda k: (int(k[0]), int(k[1])))
|
||||||
|
return arms, keys
|
||||||
|
|
||||||
|
# ---------------- prompts ----------------
|
||||||
|
PAIR_SYS = (
|
||||||
|
"Ты — строгий эксперт по художественному переводу с китайского на русский (веб-новелла, жанр сянься). "
|
||||||
|
"Тебе дают КИТАЙСКИЙ исходник и ДВА русских перевода: A и B. Оцени, какой перевод лучше как "
|
||||||
|
"ХУДОЖЕСТВЕННЫЙ русский текст, по трём осям в порядке важности: (1) ВЕРНОСТЬ — смысл исходника "
|
||||||
|
"передан без искажений, пропусков и отсебятины; (2) ЕСТЕСТВЕННЫЙ ЛИТЕРАТУРНЫЙ РУССКИЙ — читается как "
|
||||||
|
"родная русская проза, без переводческого канцелярита и кальки; (3) СОГЛАСОВАННОСТЬ имён/терминов. "
|
||||||
|
"Отметь КАТАСТРОФУ у стороны, если есть: искажение смысла, пропуск предложения, неверный род персонажа, "
|
||||||
|
"непереведённый китайский в тексте, сломанная русская морфология. "
|
||||||
|
"Ответь СТРОГИМ JSON одной строкой, без пояснений вокруг:\n"
|
||||||
|
'{"winner":"A|B|tie","margin":"clear|slight","reason":"<=25 слов, с краткой цитатой-уликой>",'
|
||||||
|
'"catastrophe_A":true|false,"catastrophe_B":true|false,"catastrophe_detail":"<кратко или пусто>"}')
|
||||||
|
|
||||||
|
CAT_SYS = (
|
||||||
|
"Ты — строгий редактор-контролёр перевода с китайского на русский. Тебе дают КИТАЙСКИЙ исходник и ОДИН "
|
||||||
|
"русский перевод. Найди КАТАСТРОФИЧЕСКИЕ дефекты (те, из-за которых взыскательный читатель забракует "
|
||||||
|
"отрывок): искажение смысла, пропуск целого предложения/абзаца, неверный род персонажа, непереведённый "
|
||||||
|
"китайский (иероглифы) в русском тексте, сломанная русская словоформа, грубая ошибка в числах/единицах "
|
||||||
|
"времени. Мелкие стилистические придирки НЕ катастрофа. "
|
||||||
|
"Ответь СТРОГИМ JSON одной строкой:\n"
|
||||||
|
'{"catastrophe":true|false,"severity":"none|minor|major|critical","items":["<кратко с цитатой>", ...]}')
|
||||||
|
|
||||||
|
def pair_user(src, ta, tb):
|
||||||
|
return (f"=== КИТАЙСКИЙ ИСХОДНИК ===\n{src}\n\n=== ПЕРЕВОД A ===\n{ta}\n\n=== ПЕРЕВОД B ===\n{tb}\n\n"
|
||||||
|
"Верни JSON-вердикт.")
|
||||||
|
|
||||||
|
def cat_user(src, t):
|
||||||
|
return f"=== КИТАЙСКИЙ ИСХОДНИК ===\n{src}\n\n=== РУССКИЙ ПЕРЕВОД ===\n{t}\n\nВерни JSON."
|
||||||
|
|
||||||
|
def parse_json(txt):
|
||||||
|
if not txt: return None
|
||||||
|
s = txt.strip()
|
||||||
|
if s.startswith("```"):
|
||||||
|
s = s.strip("`")
|
||||||
|
s = s[s.find("{"):]
|
||||||
|
i, j = s.find("{"), s.rfind("}")
|
||||||
|
if i < 0 or j < 0: return None
|
||||||
|
try: return json.loads(s[i:j+1])
|
||||||
|
except Exception: return None
|
||||||
|
|
||||||
|
# ---------------- resumable vote store ----------------
|
||||||
|
def load_votes(path):
|
||||||
|
seen = {}
|
||||||
|
if path.exists():
|
||||||
|
for ln in path.read_text().splitlines():
|
||||||
|
try:
|
||||||
|
v = json.loads(ln)
|
||||||
|
seen[v["vote_id"]] = v
|
||||||
|
except Exception: pass
|
||||||
|
return seen
|
||||||
|
|
||||||
|
def append_vote(path, v):
|
||||||
|
with open(path, "a") as f:
|
||||||
|
f.write(json.dumps(v, ensure_ascii=False) + "\n")
|
||||||
|
|
||||||
|
# ---------------- passes ----------------
|
||||||
|
def run_probe():
|
||||||
|
print(f"=== LIVE PROBE {JUDGE_MODEL} ===")
|
||||||
|
txt, rec = gemini_call(
|
||||||
|
"Ты judge. Ответь строгим JSON.",
|
||||||
|
'Верни ровно: {"ok":true,"lang":"ru"}',
|
||||||
|
"probe")
|
||||||
|
print("finish:", rec.get("finish"), "| cost $", round(rec.get("cost_usd", 0), 5),
|
||||||
|
"| tokens p/c/r:", rec.get("prompt_tok"), rec.get("completion_tok"), rec.get("reasoning_tok"))
|
||||||
|
print("response:", txt[:200])
|
||||||
|
ok = rec.get("finish") == "stop" and parse_json(txt) is not None
|
||||||
|
print("PROBE", "OK" if ok else "FAILED (check slug/vendor doc per two-directions rule)")
|
||||||
|
return ok
|
||||||
|
|
||||||
|
def run_floor(arms, keys):
|
||||||
|
"""Identical-text position-bias floor: judge glm vs glm (same text), both orders.
|
||||||
|
Deviation from 'tie' = judge noise. (Full A0<->A0' independent-regen floor deferred; owner read ratifies.)"""
|
||||||
|
path = OUT / "votes_floor.jsonl"
|
||||||
|
seen = load_votes(path)
|
||||||
|
print(f"=== FLOOR pass (glm vs glm identical), {len(keys)} units x 2 orders ===")
|
||||||
|
for k in keys:
|
||||||
|
c = arms["glm"][k]
|
||||||
|
for order in ("AB", "BA"):
|
||||||
|
vid = f"floor::{k[0]}-{k[1]}::{order}"
|
||||||
|
if vid in seen: continue
|
||||||
|
txt, rec = gemini_call(PAIR_SYS, pair_user(c["source"], c["final_text"], c["final_text"]),
|
||||||
|
vid)
|
||||||
|
v = parse_json(txt) or {}
|
||||||
|
rec_v = dict(vote_id=vid, kind="floor", unit=f"{k[0]}-{k[1]}", order=order,
|
||||||
|
winner=v.get("winner"), margin=v.get("margin"), reason=v.get("reason"),
|
||||||
|
raw=txt[:400], finish=rec.get("finish"), judge=rec.get("judge"))
|
||||||
|
append_vote(path, rec_v)
|
||||||
|
print(f" floor {k[0]}-{k[1]} {order}: winner={v.get('winner')} (identical) ${LEDGER.total:.3f}")
|
||||||
|
|
||||||
|
def run_catastrophe(arms, keys):
|
||||||
|
"""Per-arm absolute catastrophe screen — ALL arms incl. favorite."""
|
||||||
|
path = OUT / "votes_catastrophe.jsonl"
|
||||||
|
seen = load_votes(path)
|
||||||
|
print(f"=== CATASTROPHE screen, {len(keys)} units x {len(ARMS)} arms ===")
|
||||||
|
for a in ARMS:
|
||||||
|
for k in keys:
|
||||||
|
c = arms[a][k]
|
||||||
|
vid = f"cat::{a}::{k[0]}-{k[1]}"
|
||||||
|
if vid in seen: continue
|
||||||
|
txt, rec = gemini_call(CAT_SYS, cat_user(c["source"], c["final_text"]), vid)
|
||||||
|
v = parse_json(txt) or {}
|
||||||
|
rec_v = dict(vote_id=vid, kind="catastrophe", arm=a, unit=f"{k[0]}-{k[1]}",
|
||||||
|
disposition=c.get("disposition"),
|
||||||
|
catastrophe=bool(v.get("catastrophe")), severity=v.get("severity"),
|
||||||
|
items=v.get("items"), raw=txt[:500], finish=rec.get("finish"), judge=rec.get("judge"))
|
||||||
|
append_vote(path, rec_v)
|
||||||
|
flag = "CATA" if v.get("catastrophe") else "ok"
|
||||||
|
print(f" cat {a} {k[0]}-{k[1]}: {flag} ({v.get('severity')}) ${LEDGER.total:.3f}")
|
||||||
|
|
||||||
|
def run_pairwise(arms, keys):
|
||||||
|
"""Round-robin pairwise, both orders, REPS reps. Full windows. Order-normalized winner."""
|
||||||
|
path = OUT / "votes_pairwise.jsonl"
|
||||||
|
seen = load_votes(path)
|
||||||
|
pairs = list(itertools.combinations(ARMS, 2))
|
||||||
|
print(f"=== PAIRWISE, {len(pairs)} pairs x {len(keys)} units x 2 orders x {REPS} reps ===")
|
||||||
|
for (x, y) in pairs:
|
||||||
|
for k in keys:
|
||||||
|
cx, cy = arms[x][k], arms[y][k]
|
||||||
|
for order in ("AB", "BA"):
|
||||||
|
A, B = (x, y) if order == "AB" else (y, x)
|
||||||
|
cA, cB = arms[A][k], arms[B][k]
|
||||||
|
for rep in range(REPS):
|
||||||
|
vid = f"pair::{x}-{y}::{k[0]}-{k[1]}::{order}::r{rep}"
|
||||||
|
if vid in seen: continue
|
||||||
|
txt, rec = gemini_call(PAIR_SYS, pair_user(cA["source"], cA["final_text"], cB["final_text"]),
|
||||||
|
vid)
|
||||||
|
v = parse_json(txt) or {}
|
||||||
|
w = v.get("winner")
|
||||||
|
# normalize winner label (A/B) -> actual arm
|
||||||
|
if w == "A": win_arm = A
|
||||||
|
elif w == "B": win_arm = B
|
||||||
|
elif w == "tie": win_arm = "tie"
|
||||||
|
else: win_arm = None
|
||||||
|
rec_v = dict(vote_id=vid, kind="pairwise", pair=f"{x}-{y}", unit=f"{k[0]}-{k[1]}",
|
||||||
|
order=order, rep=rep, posA=A, posB=B,
|
||||||
|
winner_arm=win_arm, margin=v.get("margin"), reason=v.get("reason"),
|
||||||
|
cata_A=bool(v.get("catastrophe_A")), cata_B=bool(v.get("catastrophe_B")),
|
||||||
|
cata_detail=v.get("catastrophe_detail"), raw=txt[:400],
|
||||||
|
finish=rec.get("finish"), judge=rec.get("judge"))
|
||||||
|
append_vote(path, rec_v)
|
||||||
|
print(f" pair {x}-{y} {k[0]}-{k[1]} {order} r{rep}: win={win_arm} ({v.get('margin')}) ${LEDGER.total:.3f}")
|
||||||
|
|
||||||
|
# ---------------- aggregate ----------------
|
||||||
|
def aggregate():
|
||||||
|
import statistics
|
||||||
|
res = {"judge": JUDGE_MODEL, "arms": {a: ARM_LABEL[a] for a in ARMS}}
|
||||||
|
# floor
|
||||||
|
fl = [json.loads(l) for l in (OUT/"votes_floor.jsonl").read_text().splitlines()] if (OUT/"votes_floor.jsonl").exists() else []
|
||||||
|
tie = sum(1 for v in fl if v.get("winner") == "tie")
|
||||||
|
res["floor"] = dict(n=len(fl), tie=tie, tie_rate=round(tie/len(fl), 3) if fl else None,
|
||||||
|
non_tie=[{"unit": v["unit"], "order": v["order"], "winner": v["winner"]} for v in fl if v.get("winner") != "tie"])
|
||||||
|
# catastrophe
|
||||||
|
ca = [json.loads(l) for l in (OUT/"votes_catastrophe.jsonl").read_text().splitlines()] if (OUT/"votes_catastrophe.jsonl").exists() else []
|
||||||
|
cat_by_arm = {}
|
||||||
|
for a in ARMS:
|
||||||
|
rows = [v for v in ca if v["arm"] == a]
|
||||||
|
hits = [v for v in rows if v.get("catastrophe")]
|
||||||
|
cat_by_arm[a] = dict(n=len(rows), catastrophes=len(hits),
|
||||||
|
detail=[{"unit": v["unit"], "severity": v["severity"], "items": v.get("items"), "disposition": v.get("disposition")} for v in hits])
|
||||||
|
res["catastrophe_screen"] = cat_by_arm
|
||||||
|
# pairwise
|
||||||
|
pw = [json.loads(l) for l in (OUT/"votes_pairwise.jsonl").read_text().splitlines()] if (OUT/"votes_pairwise.jsonl").exists() else []
|
||||||
|
# per-arm points: win=1, tie=0.5
|
||||||
|
points = {a: 0.0 for a in ARMS}; games = {a: 0 for a in ARMS}
|
||||||
|
pair_tally = {}
|
||||||
|
order_flip = 0; order_pairs = 0
|
||||||
|
for v in pw:
|
||||||
|
x, y = v["pair"].split("-")
|
||||||
|
w = v["winner_arm"]
|
||||||
|
for a in (x, y): games[a] += 1
|
||||||
|
if w == "tie":
|
||||||
|
points[x] += 0.5; points[y] += 0.5
|
||||||
|
elif w in (x, y):
|
||||||
|
points[w] += 1.0
|
||||||
|
pair_tally.setdefault(v["pair"], {x: 0.0, y: 0.0, "tie": 0})
|
||||||
|
if w == "tie": pair_tally[v["pair"]]["tie"] += 1
|
||||||
|
elif w in (x, y): pair_tally[v["pair"]][w] += 1
|
||||||
|
# order consistency (noise proxy): per (pair,unit,rep) compare AB vs BA winner
|
||||||
|
by_cell = {}
|
||||||
|
for v in pw:
|
||||||
|
c = (v["pair"], v["unit"], v["rep"])
|
||||||
|
by_cell.setdefault(c, {})[v["order"]] = v["winner_arm"]
|
||||||
|
for c, od in by_cell.items():
|
||||||
|
if "AB" in od and "BA" in od:
|
||||||
|
order_pairs += 1
|
||||||
|
if od["AB"] != od["BA"]: order_flip += 1
|
||||||
|
res["pairwise"] = dict(
|
||||||
|
win_points={a: round(points[a], 1) for a in ARMS},
|
||||||
|
games={a: games[a] for a in ARMS},
|
||||||
|
win_rate={a: round(points[a]/games[a], 3) if games[a] else None for a in ARMS},
|
||||||
|
pair_tally=pair_tally,
|
||||||
|
order_consistency=dict(cells=order_pairs, flips=order_flip,
|
||||||
|
flip_rate=round(order_flip/order_pairs, 3) if order_pairs else None),
|
||||||
|
cata_flags_relative={a: sum(1 for v in pw for side in (("A", v["posA"]), ("B", v["posB"]))
|
||||||
|
if side[1] == a and v.get(f"cata_{side[0]}")) for a in ARMS},
|
||||||
|
)
|
||||||
|
res["judges_used"] = {}
|
||||||
|
for v in fl + ca + pw:
|
||||||
|
j = v.get("judge") or "unknown"
|
||||||
|
res["judges_used"][j] = res["judges_used"].get(j, 0) + 1
|
||||||
|
res["cost_usd"] = round(LEDGER.total, 4)
|
||||||
|
(OUT/"judge_results.json").write_text(json.dumps(res, ensure_ascii=False, indent=2))
|
||||||
|
print(json.dumps(res, ensure_ascii=False, indent=2))
|
||||||
|
return res
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser()
|
||||||
|
ap.add_argument("--probe", action="store_true")
|
||||||
|
ap.add_argument("--run", action="store_true")
|
||||||
|
ap.add_argument("--aggregate", action="store_true")
|
||||||
|
a = ap.parse_args()
|
||||||
|
if a.probe:
|
||||||
|
sys.exit(0 if run_probe() else 1)
|
||||||
|
arms, keys = load_arms()
|
||||||
|
print(f"loaded {len(ARMS)} arms x {len(keys)} units")
|
||||||
|
if a.run:
|
||||||
|
run_floor(arms, keys)
|
||||||
|
run_catastrophe(arms, keys)
|
||||||
|
run_pairwise(arms, keys)
|
||||||
|
aggregate()
|
||||||
|
print(f"\nTOTAL judge cost: ${LEDGER.total:.4f}")
|
||||||
|
elif a.aggregate:
|
||||||
|
aggregate()
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Loading…
Add table
Reference in a new issue