textmachine/eval/coverage_precision.py

357 lines
18 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""experiments/07 — precision (false-positive rate) of the EXCISION coverage gate on
real zh/ja→ru chunks. Gates the go-live flip of `gates.coverage.enabled` (D12/Q4).
WHAT THE GATE DOES (backend/internal/pipeline/coverage.go, ported 1:1 from the oracle
eval/refusal_bench.py::classify_output): given a stage OUTPUT and the ORIGINAL source,
flag `excision_suspect` when `sent_cov < 0.75` OR `len_ratio < corridor_low`
(zh<2.2, ja<1.4, en<0.70). Metric = characters WITHOUT spaces. Only the LOWER bound
is checked. Chunks with < min_chunk_chars non-space source are not gated.
WHY PRECISION, NOT RECALL: the backend already validated recall (≥90%) on a mini-set of
sentence-excised outputs. The blocker for the flip is the FALSE-POSITIVE rate — how many
GOOD translations the gate wrongly flags — especially on DIALOGUE-DENSE chapters, where
naive sentence segmentation can undercount RU sentences vs a CJK source (quote-absorbing
§3.7 is deferred to v1.1), dropping sent_cov below 0.75.
ARMS (both are GOOD translations — a flag on them is a false positive by construction):
1. LLM production stack: deepseek-v4-flash draft (+ optional grok-4.3 edit) per chunk.
2. Human canonical: published RU translation, aligned by chapter (clean, no fuzzy align).
Chunker mirrors chunker.go: blank-line paragraphs packed to ≤1500 est-tokens
(est = cjk + other/3, floor 16), descending to sentence boundaries on an oversize paragraph.
Usage:
eval/.venv/bin/python eval/coverage_precision.py --dry-run # plan, no API calls
eval/.venv/bin/python eval/coverage_precision.py --arm llm # run LLM arm
eval/.venv/bin/python eval/coverage_precision.py --arm human # human-canonical arm
eval/.venv/bin/python eval/coverage_precision.py --arm both --edit # + grok-4.3 edit pass
"""
from __future__ import annotations
import argparse, json, re, sys, time, unicodedata
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from refusal_bench import (load_env_file, call_provider, classify_output, split_sentences,
EXPECT_LEN_RATIO, SYSTEM_PROMPT, TARGET_NAMES)
load_env_file()
ROOT = Path(__file__).resolve().parent
SAMPLES = ROOT / "data" / "samples"
OUT = ROOT / "data" / "coverage_precision"
OUT.mkdir(parents=True, exist_ok=True)
# --- chunker (faithful port of backend/internal/pipeline/chunker.go) -------------
TARGET_TOKENS = 1500
CJK_CLASS = re.compile(r"[一-鿿㐀-䶿぀-ゟ゠-ヿ가-힯]")
def est_tokens(s: str) -> int:
cjk = len(CJK_CLASS.findall(s))
other = sum(1 for c in s if not c.isspace() and not CJK_CLASS.match(c))
est = cjk + other // 3
return max(16, est)
def split_paragraphs(chapter: str) -> list[str]:
return [p.strip() for p in re.split(r"\n\s*\n", chapter) if p.strip()]
def pack_chapter(chapter_no: int, paras: list[str]) -> list[dict]:
"""Greedy paragraph packing to TARGET_TOKENS; oversize paragraph → sentence packing."""
out, buf = [], []
def flush():
nonlocal buf
if buf:
text = "\n\n".join(buf)
out.append({"chapter": chapter_no, "idx": len(out), "text": text})
buf = []
for p in paras:
if est_tokens(p) > TARGET_TOKENS:
flush()
for sub in pack_sentences(p):
out.append({"chapter": chapter_no, "idx": len(out), "text": sub})
continue
if buf and est_tokens("\n\n".join(buf) + "\n\n" + p) > TARGET_TOKENS:
flush()
buf.append(p)
flush()
for i, c in enumerate(out): # ChunkIdx resets per chapter (chunker.go contract)
c["idx"] = i
return out
def pack_sentences(paragraph: str) -> list[str]:
segs = split_sentences(paragraph) or [paragraph]
out, buf = [], []
for s in segs:
if buf and est_tokens(" ".join(buf) + " " + s) > TARGET_TOKENS:
out.append(" ".join(buf)); buf = []
buf.append(s)
if buf:
out.append(" ".join(buf))
return out
# --- chapter splitting per language ---------------------------------------------
ZH_CH = re.compile(r"^\s*第[一二三四五六七八九十百]+章.*$", re.M)
RU_ROMAN = re.compile(r"^\s*[ⅠⅡⅢⅣⅤⅥⅦⅧⅨⅩ]+\b.*$", re.M)
def split_chapters(text: str, marker: re.Pattern | None) -> list[str]:
text = unicodedata.normalize("NFC", text)
if marker is None:
return [text.strip()]
idxs = [m.start() for m in marker.finditer(text)]
if not idxs:
return [text.strip()]
idxs.append(len(text))
return [text[idxs[i]:idxs[i + 1]].strip() for i in range(len(idxs) - 1)]
# --- dialogue-density stratification --------------------------------------------
# Quote/colon markers that introduce or bound speech in zh/ja prose.
DLG_MARK = re.compile(r"[「」『』“”《》::]")
def dialogue_density(src: str) -> tuple[float, int]:
n_sent = max(1, len(split_sentences(src)))
marks = len(DLG_MARK.findall(src))
return marks / n_sent, marks
def stratum(density: float) -> str:
return "dialogue" if density >= 0.5 else "narrative"
# --- sources ---------------------------------------------------------------------
LLM_SOURCES = [
("zh", "zh/luxun-ah-q-ch1-4.txt", ZH_CH),
("zh", "zh/luxun-ah-q-ch5-9.txt", ZH_CH),
("zh", "zh/luxun-zhufu.txt", None),
("ja", "ja/akutagawa-rashomon.txt", None),
("ja", "ja/akutagawa-hana.txt", None),
("ja", "ja/dazai-hashire-merosu.txt", None),
("ja", "ja/soseki-wagahai-neko-ch1.txt", None),
]
# Human-canonical pairs (source, canonical-ru), aligned by chapter (or whole story).
HUMAN_PAIRS = [
("ja", "ja/akutagawa-hana.txt", None, "ru/akutagawa-hana-ru.txt", None),
("zh", "zh/luxun-ah-q-ch1-4.txt", ZH_CH, "ru/luxun-ah-q-ru.txt", RU_ROMAN),
("zh", "zh/luxun-ah-q-ch5-9.txt", ZH_CH, "ru/luxun-ah-q-ru.txt", RU_ROMAN),
]
DRAFT = {"name": "deepseek", "base_url": "https://api.deepseek.com/v1",
"model": "deepseek-v4-flash", "api_key_env": "DEEPSEEK_API_KEY", "max_tokens": 12000}
# Alternate translator for the zh arm: grok-4.20 non-reasoning does NOT echo classical zh
# (deepseek-v4-flash echoes ~54% of Lu Xun zh chunks — a draft-model artifact, not a gate
# issue). Used to build a robust zh LLM coverage-FP arm from a non-echoing model.
DRAFT_GROK = {"name": "grok", "base_url": "https://api.x.ai/v1",
"model": "grok-4.20-0309-non-reasoning", "api_key_env": "XAI_API_KEY",
"max_tokens": 8000, "temperature": 0.3}
_TRANSLATOR = DRAFT
# grok-4.3 = the Phase-1 production EDITOR (monolingual ru→ru polish). reasoning-by-default;
# thinking OFF for a clean ledger (D3). temp 0.4 per the ratified editor config.
EDITOR = {"name": "grok", "base_url": "https://api.x.ai/v1", "model": "grok-4.3",
"api_key_env": "XAI_API_KEY", "max_tokens": 8000, "temperature": 0.4,
"extra_body": {}}
EDIT_SYS = ("Ты — литературный редактор. Отредактируй русский перевод: улучши стиль, "
"естественность и связность, сохранив ВСЕ детали, реплики и предложения без "
"пропусков. Выведи ТОЛЬКО отредактированный текст, без комментариев.")
def translate(chunk_text: str, target="ru") -> tuple[str | None, str | None, dict]:
sys_p = SYSTEM_PROMPT.format(target=TARGET_NAMES[target])
return call_provider(_TRANSLATOR, sys_p, chunk_text, timeout=240)
def edit(draft_text: str) -> tuple[str | None, str | None, dict]:
return call_provider(EDITOR, EDIT_SYS, draft_text, timeout=240)
# --- gate application (uses the sanctioned Python oracle) ------------------------
def gate(src: str, out: str, lang: str) -> dict:
return classify_output(src, out, None, EXPECT_LEN_RATIO[lang])
def _process_chunk(job: dict, edit_pass: bool) -> dict:
c, lang, rel = job["c"], job["lang"], job["file"]
dens, marks = dialogue_density(c["text"])
row = {"arm": "llm", "file": rel, "lang": lang, "chapter": c["chapter"],
"idx": c["idx"], "src_chars": sum(1 for x in c["text"] if not x.isspace()),
"src_tokens_est": est_tokens(c["text"]), "dlg_density": round(dens, 2),
"dlg_marks": marks, "stratum": stratum(dens)}
dtext, derr, dusage = translate(c["text"], "ru")
row["draft_err"] = derr
if derr or not (dtext or "").strip():
row["draft_verdict"] = {"verdict": "STAGE_ERROR", "detail": str(derr)}
print("E", end="", flush=True, file=sys.stderr)
return row
row["draft"] = dtext
row["draft_verdict"] = gate(c["text"], dtext, lang)
row["draft_usage"] = dusage
if edit_pass:
etext, eerr, eusage = edit(dtext)
row["edit_err"] = eerr
if etext and etext.strip():
row["edit"] = etext
row["edit_verdict"] = gate(c["text"], etext, lang)
row["edit_usage"] = eusage
print("x" if row["draft_verdict"]["verdict"] == "excision_suspect" else ".",
end="", flush=True, file=sys.stderr)
return row
def run_llm(edit_pass: bool, dry: bool, limit: int | None, workers: int) -> list[dict]:
jobs = []
for lang, rel, marker in LLM_SOURCES:
chapters = split_chapters((SAMPLES / rel).read_text(), marker)
chunks = []
for ci, ch in enumerate(chapters, 1):
chunks.extend(pack_chapter(ci, split_paragraphs(ch)))
if limit:
chunks = chunks[:limit]
print(f"[{rel}] {len(chapters)} chap → {len(chunks)} chunks", file=sys.stderr)
jobs += [{"c": c, "lang": lang, "file": rel} for c in chunks]
if dry:
return [{"arm": "llm", "file": j["file"], "lang": j["lang"],
"chapter": j["c"]["chapter"], "idx": j["c"]["idx"],
"src_chars": sum(1 for x in j["c"]["text"] if not x.isspace()),
"src_tokens_est": est_tokens(j["c"]["text"]),
"dlg_density": round(dialogue_density(j["c"]["text"])[0], 2),
"dlg_marks": dialogue_density(j["c"]["text"])[1],
"stratum": stratum(dialogue_density(j["c"]["text"])[0])} for j in jobs]
print(f"translating {len(jobs)} chunks with {workers} workers…", file=sys.stderr)
with ThreadPoolExecutor(max_workers=workers) as ex:
rows = list(ex.map(lambda j: _process_chunk(j, edit_pass), jobs))
print("", file=sys.stderr)
return rows
def run_human(dry: bool) -> list[dict]:
rows = []
for lang, srel, smark, rrel, rmark in HUMAN_PAIRS:
s_ch = split_chapters((SAMPLES / srel).read_text(), smark)
r_ch = split_chapters((SAMPLES / rrel).read_text(), rmark)
# Whole-story pair (single chapter both sides) OR chapter-index alignment.
if smark is None:
pairs = [(s_ch[0], r_ch[0], 1)]
else:
# Align by chapter index. The RU canon has a leading title/前言 block before
# ; RU_ROMAN captures from . Source ZH_CH captures from 第一章. ch5-9 file's
# chapters map to RU chapters 5..9 → offset applied by filename.
offset = 4 if "ch5-9" in srel else 0
pairs = []
for i, sc in enumerate(s_ch):
rj = i + offset
if rj < len(r_ch):
pairs.append((sc, r_ch[rj], rj + 1))
for sc, rc, chn in pairs:
dens, marks = dialogue_density(sc)
row = {"arm": "human", "file": srel, "lang": lang, "ru_chapter": chn,
"src_chars": sum(1 for x in sc if not x.isspace()),
"ru_chars": sum(1 for x in rc if not x.isspace()),
"dlg_density": round(dens, 2), "stratum": stratum(dens)}
if not dry:
row["verdict"] = gate(sc, rc, lang)
rows.append(row)
print(f" human {srel} ch{chn}: {row.get('verdict', {}).get('verdict','(dry)')}",
file=sys.stderr)
return rows
MIN_CHUNK_CHARS = 500 # production skips shorter chunks (pipeline-c1.yaml)
# Verdicts eligible for the coverage-FP denominator: only genuine complete translations
# reach the coverage gate in production. echo/empty/refusal are separate (correct)
# dispositions handled before the gate, so they are excluded from the FP measurement.
COVERAGE_ELIGIBLE = {"ok", "excision_suspect"}
def _which_fired(v: dict) -> str:
d = v.get("detail", "")
parts = []
if "sent_cov" in d:
parts.append("sent_cov")
if "len_ratio" in d:
parts.append("len_ratio")
return "+".join(parts) or "?"
def summarize(rows: list[dict]) -> dict:
llm = [r for r in rows if r["arm"] == "llm"]
summ = {"min_chunk_chars": MIN_CHUNK_CHARS}
for stage in ("draft", "edit"):
key = f"{stage}_verdict"
# production-gated: complete translation AND src >= min_chunk_chars
gated = [r for r in llm if key in r and r[key]["verdict"] in COVERAGE_ELIGIBLE
and r["src_chars"] >= MIN_CHUNK_CHARS]
if not gated:
continue
fp = [r for r in gated if r[key]["verdict"] == "excision_suspect"]
s = {"n_gated": len(gated), "flags": len(fp),
"flag_rate": round(len(fp) / max(1, len(gated)), 3),
"fired": {}, "flagged_chunks": []}
for r in fp:
f = _which_fired(r[key])
s["fired"][f] = s["fired"].get(f, 0) + 1
s["flagged_chunks"].append({"file": r["file"], "ch": r["chapter"], "idx": r["idx"],
"stratum": r["stratum"], "detail": r[key]["detail"],
"sent_cov": r[key].get("sent_cov"),
"len_ratio": r[key].get("len_ratio")})
for st in ("narrative", "dialogue"):
sub = [r for r in gated if r["stratum"] == st]
subfp = [r for r in sub if r[key]["verdict"] == "excision_suspect"]
s[st] = {"n": len(sub), "flags": len(subfp),
"flag_rate": round(len(subfp) / max(1, len(sub)), 3)}
# per-chapter flag counts → informs the "N flags per chapter" threshold
perch = {}
for r in gated:
k = f"{r['file']}::ch{r['chapter']}"
perch.setdefault(k, {"chunks": 0, "flags": 0})
perch[k]["chunks"] += 1
perch[k]["flags"] += int(r[key]["verdict"] == "excision_suspect")
s["per_chapter_max_flags"] = max((v["flags"] for v in perch.values()), default=0)
s["per_chapter"] = perch
# sent_cov / len_ratio distribution over gated OK chunks (for threshold tuning)
covs = sorted(round(r[key]["sent_cov"], 3) for r in gated if r[key].get("sent_cov") is not None)
lrs = sorted(round(r[key]["len_ratio"], 3) for r in gated if r[key].get("len_ratio") is not None)
if covs:
s["sent_cov_min_observed"] = covs[0]
s["sent_cov_p05"] = covs[max(0, int(0.05 * len(covs)) - 1)]
if lrs:
s["len_ratio_min_observed"] = lrs[0]
summ[stage] = s
# sub-min chunks excluded from prod gating (reported for transparency)
submin = [r for r in llm if "draft_verdict" in r and r["src_chars"] < MIN_CHUNK_CHARS]
summ["submin_excluded"] = len(submin)
stage_err = [r for r in llm if r.get("draft_verdict", {}).get("verdict")
not in COVERAGE_ELIGIBLE and "draft_verdict" in r]
summ["non_coverage_dispositions"] = [
{"file": r["file"], "ch": r["chapter"], "idx": r["idx"],
"verdict": r["draft_verdict"]["verdict"]} for r in stage_err]
# human arm: ANY flag on a published complete translation is a clean false positive
human = [r for r in rows if r["arm"] == "human" and "verdict" in r]
hfp = [r for r in human if r["verdict"]["verdict"] == "excision_suspect"]
summ["human"] = {"n": len(human), "flags": len(hfp),
"flag_rate": round(len(hfp) / max(1, len(human)), 3),
"flagged": [{"file": r["file"], "ru_ch": r.get("ru_chapter"),
"stratum": r["stratum"], "detail": r["verdict"]["detail"],
"sent_cov": r["verdict"].get("sent_cov"),
"len_ratio": r["verdict"].get("len_ratio")} for r in hfp],
"sent_cov_all": sorted(round(r["verdict"].get("sent_cov", 9), 3) for r in human),
"len_ratio_all": sorted(round(r["verdict"].get("len_ratio", 9), 3) for r in human)}
return summ
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--arm", choices=["llm", "human", "both"], default="both")
ap.add_argument("--edit", action="store_true", help="also run grok-4.3 editor pass")
ap.add_argument("--dry-run", action="store_true")
ap.add_argument("--limit", type=int, help="max chunks per file (LLM arm)")
ap.add_argument("--workers", type=int, default=5)
ap.add_argument("--translator", choices=["deepseek", "grok"], default="deepseek",
help="draft model for the LLM arm (grok = non-echoing zh translator)")
ap.add_argument("--langs", help="comma-separated source langs to include (e.g. zh)")
ap.add_argument("--out", default=str(OUT / "results.json"))
args = ap.parse_args()
global _TRANSLATOR, LLM_SOURCES
_TRANSLATOR = DRAFT_GROK if args.translator == "grok" else DRAFT
if args.langs:
keep = set(args.langs.split(","))
LLM_SOURCES = [s for s in LLM_SOURCES if s[0] in keep]
rows = []
if args.arm in ("llm", "both"):
rows += run_llm(args.edit, args.dry_run, args.limit, args.workers)
if args.arm in ("human", "both"):
rows += run_human(args.dry_run)
summ = summarize(rows)
Path(args.out).write_text(json.dumps({"summary": summ, "rows": rows},
ensure_ascii=False, indent=2))
print("\n=== SUMMARY ===")
print(json.dumps(summ, ensure_ascii=False, indent=2))
print(f"\nwrote {args.out}")
if __name__ == "__main__":
main()