From 667eec0dc530a20798929d45176c994a5ae61497 Mon Sep 17 00:00:00 2001 From: "Claude (backend session)" Date: Sun, 19 Jul 2026 00:24:24 +0300 Subject: [PATCH] Land exp16 bank-mining: the which-what split with the 0.97 recall code detector, banknote as the dst channel, the clean exp14b re-audit, and the stop-only truncation errata --- docs/experiments/16-bank-mining.md | 128 ++++++++++++++++- eval/exp16/a6_local.py | 133 ++++++++++++++++++ eval/exp16/a6b_verify.py | 129 +++++++++++++++++ eval/exp16/canon_lift.py | 89 ++++++++++++ eval/exp16/coldstart_analyze.py | 198 ++++++++++++++++++++++++++ eval/exp16/coldstart_gen.py | 12 +- eval/exp16/emit_owner_sheets.py | 219 +++++++++++++++++++++++++++++ 7 files changed, 902 insertions(+), 6 deletions(-) create mode 100644 eval/exp16/a6_local.py create mode 100644 eval/exp16/a6b_verify.py create mode 100644 eval/exp16/canon_lift.py create mode 100644 eval/exp16/coldstart_analyze.py create mode 100644 eval/exp16/emit_owner_sheets.py diff --git a/docs/experiments/16-bank-mining.md b/docs/experiments/16-bank-mining.md index 1b7b42f..6804438 100644 --- a/docs/experiments/16-bank-mining.md +++ b/docs/experiments/16-bank-mining.md @@ -187,6 +187,130 @@ ja→ru реплика — след-за-§D (отдельный мини-фри --- -## §2. GT и материал (см. §1.1) · §3. Результаты по армам · §4. Решающие правила + слепые зоны · §5. Деньги · §6. Самопроверка +## §2. GT и материал +См. §1.1 (GT=57 + blurb-правило, страты f≥10:40/f3-9:7/f<3:10, records.json 57 чанков, конфаунд инъекции). +Артефакты прогонов — `books/gu-zhenren/exp16/*.json` (вне git). -*(заполняются после прогонов; тест-половина — первичка)* +## §3. Результаты по армам + +### 3.1. A1–A3+абл — recall@PROPOSED (ось слепоты), ТЕСТ-половина ch16-25 (первичка, анти-подгонка) +| Арм | overall | f≥10 | f3-9 | f<3 | catastrophe | +|---|---|---|---|---|---| +| **A1 V-A** | 0.818 | **1.00** | 0.917 | **0.00** (частотно слеп) | PASS | +| **A2 V-B** | 0.818 | 1.00 | 0.917 | 0.00 | PASS | +| **A3 V-C** | **0.932** | 1.00 | 0.917 | **0.714** | PASS | +| A3-абл | 0.932 | 1.00 | 0.917 | 0.714 | PASS | + +По типам (A3 тест): name 1.0 · title 1.0 · place 1.0 · term 0.83 · nickname 1.0. Промахи A3@proposed +(3): 开窍(f<3) · 月兰花(f3-9) · 困境(f<3). **V-C recall на f≥3 ≈ 0.97 на held-out** — §D5 планка 85% +пройдена. Full-срез: V-A f≥3=0.98, V-C f≥3=1.0, overall 0.965 (промахи 一气金光虫[ядро 金光虫 предложено] +· 长生). Trade-off (full A3): top-90 recall 0.53/pseudo-prec 0.30 · top-250 0.64/0.13 · @proposed 0.965/0.004 +(pseudo-prec низка by design — сид не исчерпывающий, §D1; истинная точность — §4 precision@30). + +**Вклад сигналов (§D4-а, раздельно):** ПАТТЕРНЫ (A3−A1 @proposed): f<3 0.0→0.71, place 0.4→1.0, name→1.0 — +закрывают частотно-слепые классы. SPREAD (A2−A1): 0 новых кандидатов; на инъектированных λ>0 монотонно +ВРЕДИТ ранжированию (top90 0.625→0.475@λ=8) — конфаунд §D1, изолирован. Сенситивность ±50% (§1.3): +A3@f≥3=1.0 инвариантно по floor/over_rep/subsume/partners; catastrophe PASS везде. + +### 3.2. Алиас-tier1 (§B2) +Известные алиасы сида (古月方源=方源, 古月方正=方正): **recall 2/2**. Прецизионно-безопасные R1–R3 при R4: +R4-iii блокирует 族长=四代族长 (разный approved dst); компози-гард блокирует цепочку 古月+族长 (сохраняя +fullname 古月+方源); кап-гейт имён-лемм убирает Палладий-FP «найти»=най+ти. Остаточный tier-1-шум (方源看, +古月赤 — фрагменты) — известная ошибка §B2 (мерж только по подписи). Recall кластеризации — только по +мини-голду владельца (§E-1, сайдкар `alias_minigold.md`). + +### 3.3. Канон (§C2) — ВЕДУЩАЯ находка: dst-извлечение = слабое звено +Канон-recovery (совпадение с подписанным dst, chunk-co-occurrence): +- full-инъект 57ч **0.68** · инъект-14ч **0.35** · **cold-start-14ч 0.20**. +**§D5-гейт canon-recovery≥70% НЕ пройден.** НО причина — не модель, а **метод извлечения dst**: chunk-уровневая +ко-оккуренция (Dice) вырождается для (а) вездесущих термов (蛊/方源 в ~каждом чанке → топ-ассоциат = частое +контекстное слово, не рендеринг) и (б) локализованных (все леммы чанка тождественны по co-occ). Проверено: +lift/PMI-коррекция базовой ставки (`canon_lift.py`) — ЕЩЁ ХУЖЕ (0.29): проблема в ГРАНУЛЯРНОСТИ (chunk), не +в базовой ставке; нужна предложение-уровневая привязка (§B1-митигация, не реализована) ИЛИ прямой dst-канал. +**Имена извлекаются (Палладий+кап-лемма: 方源→фан/юань, 古月→гуюэ); термы — нет.** ⟶ вывод §4. + +### 3.4. A6 / A6b — локальная 9B на стенде (huihui_ai/qwen3.5-abliterated:9b, $0) +- **A6 спот по вебновелл-source:** recall **0.895** (f≥10 0.93 / f3-9 1.0 / f<3 0.70), health 56/57, + **3 галлюцинации**. **Оговорка exp06 ПОДТВЕРЖДЕНА:** спот падает 0.98(PD-классика)→0.895(вебновелл). + **Код-детектор V-C (@proposed 0.965) ≥ локальная 9B (0.895)**, детерминирован, 0 галлюцинаций. +- **A6b Z1-верификатор** (30 пар, авто-голд): accuracy 0.60, **wrong-catch 0.333**, correct-accept 0.867. + Локальная 9B **принимает правдоподобно-неверное** (族长→«старейшина», 人上之人→«среди людей» = correct!); + ловит лишь грубое (元石→«метеорит»). **⟶ Z1 маршрутизируется в ОБЛАКО, не локаль** (§B4 подтверждён; + согласуется с exp06 render 0.48–0.55). + +### 3.5. Cold-start срез (платн., конфаунд снят) + A4/A5 канал сноски +- **Spread инъект→cold-start: 0.013→0.031** (конфаунд §D1 реален, но скромен на 14ч). Ядровые имена + консистентны без инъекции (方源/古月 spread~0 — частотно-заякорены); термы/титулы дрейфуют (花酒行者 + 0→0.25, 蛊虫 0→0.2, 丙等 0→0.2, 族长 0→0.143). Canon-recovery cold-start 0.20 (см. 3.3 — извлечение). +- **A4/A5 banknote-v1:** 95 строк / 14 чанков, distinct_src 65, **parse_fail 0.0%, truncated 0**. + Recall банкноты: all 37/57, f≥10 30/40, f3-9 2/7, f<3 5/10. **Маржинальный recall над V-C на f<3: +1 + терм (一气金光虫, ~10 п.п.)** — V-C уже держит остальные (код > банкнота по recall f<3: 8/10 vs 5/10). + **Отличительная ценность банкноты — ПРЯМОЙ dst:** переводчик сам эмитит 蛊→гу, 血颅蛊→«Гу кровавого + черепа», 一气金光虫→«Золотоносный червь…», 希望蛊→«Гу надежды», 方源→«Фан Юань» — ровно те dst, что + co-occurrence извлечь НЕ смог (3.3), + нашёл НЕ-сидовую 龙公→«Лун Гун». Дельта качества (длина + banknote/plain): mean **0.935**, std 0.19 — лёгкое сокращение в пределах детерм-шума; смысл-судья на + таком маржинальном эффекте — под измеренным шум-полом 0.261 (addendum §1), уверенного вывода о + деградации НЕ делаю. + +## §4. Решающие правила (§D5) применены + маршрутизация слепых зон + +**Гипотеза №1 («код сам разметит банк») — РАСЩЕПЛЕНА на две:** +- **WHICH (какие сущности) — ПОДТВЕРЖДЕНА:** V-C recall f≥3 ≈0.97 (held-out ≥85% ✓), overall @proposed + 0.965, катастроф-скрин PASS, кластеризация R1–R3 (known-alias 2/2). Код детерминированно и за $0 + размечает, ЧТО есть сущность банка. +- **WHAT (какой dst/канон) — НЕ подтверждена для co-occurrence:** canon-recovery 0.20–0.68 < 70%; dst- + извлечение из черновиков ко-оккуренцией — слабое звено (вырождение по гранулярности). Имена берутся + Палладием; термы — нет. ⟶ dst маршрутизируется в **банкноту (переводчик эмитит) / облако+Палладий**, НЕ + в чистую ко-оккуренцию. Это прямой вход бэкенд-дизайна слоя 4. +**Канал сноски (§D5-правило):** parse_fail 0%<5% ✓; маржинальный recall f<3 ~+10 п.п. (на пороге); дельта +качества — лёгкая, под шум-полом. **Вердикт: банкнота ЖИЗНЕСПОСОБНА и решает dst-пробел** (её ниша — не +recall редкого, где код сильнее, а ПРЯМАЯ доставка dst для вездесущих+редких термов); цена копеечная ($0.023 +/ 14 чанков → ~$0.01–0.1/ранобэ, класс §B3 подтверждён). Рекомендация — не «сноска в архив», а «сноска = +dst-канал поверх кода-детектора». + +**Слепые зоны лучшего арма (V-C) — ручная Z-классификация промахов (§D3-5, главный вход бэкенд-дизайна):** +| GT-промах | страта | зона §B4 | почему код слеп | маршрут | +|---|---|---|---|---| +| 长生 (бессмертие) | f<3 | Z-common-word | частый общеслов, контраст давит; не паттерн | облако/сноска | +| 困境 (Беды=имя стаи) | f<3 | Z1 stable-wrong | обычное слово 困境 как ИМЯ; termhood давится | облако-судья + KWIC | +| 开窍 (обряд) | f<3 | Z3 rare | редкий; 窍-формант дал 空窍, не 开窍 (границы) | сноска/локаль-спот | +| 月兰花 | f3-9 | Z3 rare | пороговый; 花-не-формант | код (порог) / сноска | +| 一气金光虫 | f<3 | Z3 rare | 5-char, паттерн дал ядро 金光虫 | **банкнота (получено!)** | + +## §5. Деньги +Платный спенд ИТОГО: **$0.0226 / кап $10** (28 flash-вызовов cold-start+banknote, все finish=stop, 0 +length-reject, 0 err, 0 parse_fail). $0-армы: код+стенд, $0. DeepSeek-долина соблюдена (UTC ~21–22, окно +скидки). Проекция сноски на ранобэ: ~$0.01–0.1 (§B3). + +## §6. Самопроверка (мандат 12.07) + отклонения + +**Инкрементальный ревью исполнением — поймал 4 бага ДО результатов:** (1) shadowing exp15/arms.py поверх +exp16 (sys.path insert→append); (2) дефектный R4-эксцепшн алиаса (сливал 族长=四代族长); (3) Палладий-FP +«найти»=най+ти (→ кап-лемма-гейт); (4) конфаунд spread λ=8 (изолирован). + +**Консолидированный адверсариальный само-ревью (6 осей, refute-by-default, независимый рекомпьют, воркфлоу):** +ВСЕ 6 держатся (4 holds / 2 holds_with_caveats), **0 critical**. Несущее: +- **MAJOR ПОЙМАН И ПОЧИНЕН ДО СПЕНДА:** truncation-гейт был `finish=='length'`-only → усечёнка с иным + finish_reason (DeepSeek `insufficient_system_resource` peak-overload) прошла бы. Фикс: принимать ТОЛЬКО + `finish=='stop'`. Прогон подтвердил: 28/28 stop, 0 reject. +- Катастроф-скрин ГЕНУИНЕН (независимая реимплементация, 0 расхождений top-50); g(L)=log₂(L+1) несущий + (log₂(L)→蛊 rank 4753, скрин FAIL). A1/A2 кандидат-сеты байт-идентичны; subsume ДЕФЛЯЦИРУЕТ recall (не + льстит). Фриз-целостность: инлайн translator-промпт == translator.md brief-filled (SHA рантайм-сверен). +- **Каветы в реестр (не блокеры):** gt_occurrences сумма по поверхностям — латентный двойной счёт (только + 方源/方正, оба f≥10, ноль влияния на страты); docstring c-value «heavy»→«light» дисконт; surname-канал + over-generates типо-неверные (丁等 как name); canon-lift — негативный результат (документирован); банкнота + truncation-tolerance недостижима (гейт stop-only). **Провайдер-аномалия к владельцу:** пик-окно D39.7 + (UTC 01-04/06-10) НЕ совпадает с документированным DeepSeek off-peak-скидочным окном 16:30–00:30 UTC — + правило двух направлений (00-provider-quirks): сверить с вендор-докой ДО след. платных прогонов, не + абсорбировать интерпретацию. (Прогон шёл в 16:30–00:30 — дешёвом окне по обеим трактовкам.) + +**Отклонения от пре-рега:** нет по армам/метрикам/гардам. Blurb-правило — как объявлено (GT=57). +Length-гейт усилён (stop-only) — это errata frozen `coldstart_gen.py` (SHA 70c8d9c1 устарел), поймана +мандатным само-ревью ДО спенда; лендит оркестратор. + +## §7. Владельцу (сайдкары + тучпойнты) +- `signature_map.md` — карта подписи (канон-предложение [банкнота-dst где есть], dst-варианты, KWIC, пол + 他/她-evidence, зоны §B4). **Пол = ваш вердикт** (人祖=male/古月=n-a подписаны; 白凝冰 hidden — 他=… в тексте). +- `alias_minigold.md` — мини-голд алиасов (~20–30 мин: вычеркнуть/дополнить рёбра → recall кластеризации). +- `precision_at30.md` — топ-30 не-сидовых A3 для адъюдикации (T/F/? → precision@30). +- Открытый вопрос §E-4/§B5: жанр-паки НЕ строились (универсальные каналы); реплика ja→ru — отдельный мини-фриз. diff --git a/eval/exp16/a6_local.py b/eval/exp16/a6_local.py new file mode 100644 index 0000000..8246933 --- /dev/null +++ b/eval/exp16/a6_local.py @@ -0,0 +1,133 @@ +#!/usr/bin/env python3 +"""exp16 — A6: local 9B entity spotter over the webnovel source (research/20 §D2 A6, §B4 Z3). $0 (stand). + +Re-measures the exp06 caveat (exp06 spot recall 0.95-1.0 was on PD-classic; the webnovel-slice re-measure +is the eval/README-flagged open question). Prompts a local 9B to spot named entities/terms per source chunk, +aggregates across the slice, and measures spot-recall against GT + hallucination. Health-checked per exp06 +(empty/echo/unparseable -> retry with bigger budget -> HEALTH=failed, NOT counted as recall 0). Deterministic. + +localhost ollama needs no-proxy transport (stand quirk: 403 on localhost under the corp proxy). +""" +from __future__ import annotations + +import json +import re +import sys +import urllib.request +from pathlib import Path + +import exp16_common as X + +MODEL = "huihui_ai/qwen3.5-abliterated:9b" # exp06 best local spot recall (1.0) +OLLAMA = "http://localhost:11434/api/chat" +OUT = Path("/home/ubuntu/books/gu-zhenren/exp16") + +# no-proxy opener (stand quirk) +_opener = urllib.request.build_opener(urllib.request.ProxyHandler({})) + +SPOT_PROMPT = ( + "Ты — экстрактор именованных сущностей из китайского текста веб-новеллы. " + "Найди в СЛЕДУЮЩЕМ фрагменте все имена персонажей, топонимы, титулы/ранги и уникальные термины мира " + "книги (артефакты, техники, существа). Верни ТОЛЬКО JSON-массив строк — сами китайские подстроки как " + "они в тексте, без перевода и пояснений. Пример: [\"方源\",\"蛊师\",\"青茅山\"].\n\nФрагмент:\n") + + +def call_local(prompt: str, num_predict=1024, timeout=120): + # think:false + /no_think — spotting is extraction, not reasoning; thinking mode is slow + noisy here + body = json.dumps({"model": MODEL, "think": False, + "messages": [{"role": "user", "content": prompt + "\n/no_think"}], + "stream": False, "options": {"temperature": 0.0, "num_predict": num_predict}}).encode() + req = urllib.request.Request(OLLAMA, data=body, headers={"Content-Type": "application/json"}) + try: + with _opener.open(req, timeout=timeout) as r: + d = json.loads(r.read()) + return (d.get("message", {}) or {}).get("content", ""), None + except Exception as e: + return "", f"{type(e).__name__}: {str(e)[:150]}" + + +def parse_entities(text: str): + """Extract the JSON array of zh strings; tolerant of blocks and code fences.""" + text = re.sub(r".*?", "", text, flags=re.S) + m = re.search(r"\[.*\]", text, re.S) + if not m: + # fallback: any Han runs of length 2-6 in quotes + return [q for q in re.findall(r'"([㐀-鿿]{1,8})"', text)] + try: + arr = json.loads(m.group(0)) + return [str(x).strip() for x in arr if isinstance(x, (str,)) and re.search(r"[㐀-鿿]", str(x))] + except json.JSONDecodeError: + return [q for q in re.findall(r'"([㐀-鿿]{1,8})"', text)] + + +def main(): + chunks = X.load_chunks() + gt = X.load_gt() + limit = int(sys.argv[1]) if len(sys.argv) > 1 else len(chunks) + spotted = set() # normalized spotted substrings + spotted_raw = [] + health = {"ok": 0, "failed": 0, "retried": 0} + per_chunk = [] + for c in chunks[:limit]: + prompt = SPOT_PROMPT + c.source + ents, err, npred = [], None, 1024 + for attempt in range(2): + out, err = call_local(prompt, num_predict=npred) + ents = parse_entities(out) + if ents and not err: + if attempt > 0: + health["retried"] += 1 + break + npred = 2048 # bigger budget on retry + if err or not ents: + health["failed"] += 1 + per_chunk.append({"id": f"{c.chapter}.{c.chunk_idx}", "health": "failed", "err": err}) + print(f" ch{c.chapter}.{c.chunk_idx}: HEALTH=failed ({err})") + continue + health["ok"] += 1 + for e in ents: + en = X.norm(e) + if en: + spotted.add(en) + spotted_raw.append(e) + per_chunk.append({"id": f"{c.chapter}.{c.chunk_idx}", "health": "ok", "n": len(ents)}) + print(f" ch{c.chapter}.{c.chunk_idx}: {len(ents)} ents") + + # recall vs GT: a GT entity is spotted if any of its normalized surfaces is a spotted substring + # OR a spotted string contains/is contained in the surface (surface-level match, spot-lenient) + def gt_spotted(ent): + for ns in ent.norm_surfaces: + if ns in spotted: + return True + for sp in spotted: + if ns in sp or sp in ns: + return True + return False + + hits = [e for e in gt if gt_spotted(e)] + by_stratum = {"f>=10": [], "f3-9": [], "f<3": []} + for e in gt: + by_stratum[X.freq_stratum(X.gt_occurrences(e, chunks))].append(gt_spotted(e)) + recall = len(hits) / len(gt) + # hallucination proxy: spotted normalized strings that do NOT occur in ANY source (as substring) + allsrc = "".join(c.nsource for c in chunks) + halluc = [s for s in spotted if s not in allsrc] + + print(f"\n=== A6 local 9B spot ({MODEL}) — {limit} chunks ===") + print(f"health: {health}") + print(f"spot recall vs GT: {len(hits)}/{len(gt)} = {recall:.3f}") + for st, v in by_stratum.items(): + print(f" {st}: {sum(v)}/{len(v)} = {sum(v)/len(v):.2f}" if v else f" {st}: n/a") + print(f"distinct spotted: {len(spotted)}; hallucinated (not in source): {len(halluc)} {list(halluc)[:10]}") + miss = [e.src for e in gt if not gt_spotted(e)] + print(f"GT misses: {miss}") + json.dump({"model": MODEL, "limit": limit, "health": health, "recall": recall, + "by_stratum": {k: (sum(v), len(v)) for k, v in by_stratum.items()}, + "n_spotted": len(spotted), "halluc": list(halluc), "misses": miss, + "spotted": sorted(spotted)}, + open(OUT / "a6_local_spot.json", "w"), ensure_ascii=False, indent=1) + print(f"[written] {OUT/'a6_local_spot.json'}") + + +if __name__ == "__main__": + main() diff --git a/eval/exp16/a6b_verify.py b/eval/exp16/a6b_verify.py new file mode 100644 index 0000000..dc88cd1 --- /dev/null +++ b/eval/exp16/a6b_verify.py @@ -0,0 +1,129 @@ +#!/usr/bin/env python3 +"""exp16 — A6b: local 9B as a Z1 VERIFIER of translation pairs (research/20 §D2 A6b, §B4 Z1). $0 (stand). + +§B4 routes SPOT to code (A6 showed code>=local 9B) and reserves the local 9B for VERIFICATION of the +stably-wrong class Z1 (蛊→«гусеница»: high termhood, zero spread, dst is a common word — invisible to the +spread signal). This mini-bench asks: can a local 9B judge whether a ru rendering of a zh term IN CONTEXT +is faithful? Gold is AUTOMATIC: CORRECT pairs use the owner-signed seed dst; WRONG pairs inject the Z1 +failure class (plausible common-word mistranslations + cross-term swaps). Accuracy + wrong-catch rate. + +30 pairs (15 correct + 15 wrong). Deterministic (temp 0, think off). Health-checked. +""" +from __future__ import annotations + +import json +import re +import urllib.request +from pathlib import Path + +import exp16_common as X + +MODEL = "huihui_ai/qwen3.5-abliterated:9b" +OLLAMA = "http://localhost:11434/api/chat" +OUT = Path("/home/ubuntu/books/gu-zhenren/exp16") +_opener = urllib.request.build_opener(urllib.request.ProxyHandler({})) + +# Z1 verification bench: (src_term, correct_dst, wrong_dst[Z1 stably-wrong class]). Wrong = a plausible +# common Russian word or a cross-term swap that the spread signal would NOT catch. +BENCH = [ + ("蛊", "гу", "гусеница"), # Z1 archetype: 蛊 as a common bug-word + ("蛊师", "гу-мастер", "заклинатель насекомых"), + ("蛊虫", "гу-червь", "личинка"), + ("转", "ранг", "оборот"), # 转 polysemy: rank vs turn/revolution + ("元石", "первобытный камень", "метеорит"), + ("真元", "истинная ци", "чистая энергия"), + ("空窍", "апертура", "полость"), + ("凡人", "смертный", "простолюдин"), + ("修炼", "культивация", "тренировка"), + ("长生", "бессмертие", "долголетие"), + ("族长", "глава клана", "старейшина"), # the D38 c3 confound as a Z1 pair + ("四代族长", "Четвёртый глава рода", "четвёртый старейшина"), + ("人上之人", "человек над людьми", "человек среди людей"), # the c1 catastrophe as Z1 + ("方源", "Фан Юань", "Квадратный Источник"), # name mistranslated literally + ("花酒行者", "Монах Цветочного Вина", "Странник цветов и вина"), +] + + +def call_local(prompt, timeout=90): + body = json.dumps({"model": MODEL, "think": False, + "messages": [{"role": "user", "content": prompt + "\n/no_think"}], + "stream": False, "options": {"temperature": 0.0, "num_predict": 200}}).encode() + req = urllib.request.Request(OLLAMA, data=body, headers={"Content-Type": "application/json"}) + try: + with _opener.open(req, timeout=timeout) as r: + d = json.loads(r.read()) + return (d.get("message", {}) or {}).get("content", ""), None + except Exception as e: + return "", f"{type(e).__name__}: {str(e)[:120]}" + + +def find_sentence(term, chunks): + tn = X.norm(term) + for c in chunks: + for sent in re.split(r"(?<=[。!?])", c.source): + if term in sent and 6 <= len(sent) <= 80: + return sent.strip() + return term + + +def verdict_yes_no(text): + text = re.sub(r".*?", "", text, flags=re.S).lower() + # first explicit yes/да or no/нет + m = re.search(r"\b(да|верно|correct|yes|правильн)\b", text) + n = re.search(r"\b(нет|неверно|incorrect|no|ошибочн|неправильн)\b", text) + if n and (not m or n.start() < m.start()): + return "no" + if m: + return "yes" + return "?" + + +def main(): + chunks = X.load_chunks() + pairs = [] + for term, ok_dst, bad_dst in BENCH: + sent = find_sentence(term, chunks) + pairs.append((term, sent, ok_dst, "correct")) + pairs.append((term, sent, bad_dst, "wrong")) + + rows = [] + health = {"ok": 0, "failed": 0} + for term, sent, dst, gold in pairs: + prompt = (f"Китайский термин «{term}» в предложении:\n{sent}\n\n" + f"Предложенный русский перевод этого термина: «{dst}».\n\n" + f"Верен ли этот перевод термина по смыслу в данном контексте? " + f"Ответь одним словом: ДА или НЕТ, затем короткое обоснование.") + out, err = call_local(prompt) + if err: + health["failed"] += 1 + rows.append(dict(term=term, dst=dst, gold=gold, pred="failed", err=err)); continue + health["ok"] += 1 + pred = verdict_yes_no(out) + model_says = "correct" if pred == "yes" else ("wrong" if pred == "no" else "?") + rows.append(dict(term=term, dst=dst, gold=gold, pred=pred, model_says=model_says, + correct=(model_says == gold))) + mark = "✓" if model_says == gold else ("?" if pred == "?" else "✗") + print(f" {mark} {term:<6} «{dst[:24]:<24}» gold={gold:<7} model={model_says:<7}") + + scored = [r for r in rows if r.get("pred") not in ("failed", "?")] + acc = sum(r["correct"] for r in scored) / len(scored) if scored else 0 + # wrong-catch (recall on the Z1 wrong class) + correct-accept (specificity) + wrong = [r for r in scored if r["gold"] == "wrong"] + correct = [r for r in scored if r["gold"] == "correct"] + wrong_catch = sum(r["model_says"] == "wrong" for r in wrong) / len(wrong) if wrong else 0 + correct_accept = sum(r["model_says"] == "correct" for r in correct) / len(correct) if correct else 0 + print(f"\n=== A6b Z1 verifier ({MODEL}) — {len(pairs)} pairs ===") + print(f"health: {health}; scored: {len(scored)}") + print(f"accuracy: {acc:.3f}") + print(f"wrong-catch (recall on Z1 mistranslations): {sum(r['model_says']=='wrong' for r in wrong)}/{len(wrong)} = {wrong_catch:.3f}") + print(f"correct-accept (specificity): {sum(r['model_says']=='correct' for r in correct)}/{len(correct)} = {correct_accept:.3f}") + missed = [r["term"] for r in wrong if r["model_says"] != "wrong"] + print(f"Z1 mistranslations the local 9B FAILED to catch: {missed}") + json.dump({"model": MODEL, "health": health, "accuracy": acc, "wrong_catch": wrong_catch, + "correct_accept": correct_accept, "rows": rows}, + open(OUT / "a6b_verify.json", "w"), ensure_ascii=False, indent=1) + print(f"[written] {OUT/'a6b_verify.json'}") + + +if __name__ == "__main__": + main() diff --git a/eval/exp16/canon_lift.py b/eval/exp16/canon_lift.py new file mode 100644 index 0000000..6522d72 --- /dev/null +++ b/eval/exp16/canon_lift.py @@ -0,0 +1,89 @@ +#!/usr/bin/env python3 +"""exp16 — ADDITIVE canon extractor: base-rate-corrected (lift/PMI) dst-variants (research/20 §B1 mitigation +for ubiquitous terms). $0. NOT frozen — an analysis-layer improvement over the pre-registered chunk-Dice +(spread.dst_variants), which degenerates for terms present in ~every chunk (方源/蛊/古月 — no 'chunks +without X' contrast, so every frequent ru lemma looks associated). + +lift(lemma, term) = |chunks(term) ∩ chunks(lemma)| / expected, expected = |chunks(term)|*|chunks(lemma)|/N. +A common ru word (год, everywhere) has lift≈1; the term's actual rendering (гу — only near 蛊) has high +lift. This surfaces the rendering even for ubiquitous terms. Reported ALONGSIDE the frozen method so the +pre-reg canon number and the true recoverability are both visible (§D4-а, no manufactured convergence). +""" +from __future__ import annotations + +import exp16_common as X +import spread as SP +import palladius as PAL + + +def dst_variants_lift(sm: SP.SpreadModel, cand_norm: str, top=6, min_lift=1.5, min_cooccur=2): + cidx = set(sm.cand_chunk_indices(cand_norm)) + N = len(sm.chunks) + if len(cidx) < 1: + return [] + from collections import Counter + counts = Counter() + for i in cidx: + counts.update(sm.chunk_lemmas[i]) + scored = [] + for lm, co in counts.items(): + if co < min_cooccur: + continue + base = len(sm.lemma_chunks[lm]) + expected = len(cidx) * base / N + lift = co / expected if expected else 0.0 + if lift >= min_lift: + scored.append((lm, round(lift, 2), co)) + scored.sort(key=lambda t: (-t[1], -t[2], t[0])) + return scored[:top] + + +def canon_propose_lift(sm, ent, top=6): + lifted = dst_variants_lift(sm, X.norm(ent.src), top=top) + lemmas = [lm for lm, _, _ in lifted] + if ent.typ in ("name", "place", "nickname"): + proposed = [lm for lm in lemmas if PAL.is_palladius_token(lm, 1) and sm.is_name_lemma(lm)] + head = [lm for lm in lemmas if lm in ("гора", "деревня")] + proposed = head[:1] + proposed if ent.typ == "place" else proposed + else: + proposed = lemmas[:4] + return proposed, lifted + + +def canon_recovery_lift(sm, ent, cover_frac=0.5): + import canon as CANON + signed = set(CANON.signed_content_lemmas(ent.dst)) + if not signed: + return None, [], [] + proposed, lifted = canon_propose_lift(sm, ent) + assoc = {lm for lm, _, _ in lifted} + covered = {s for s in signed if s in set(proposed) or s in assoc} + return (len(covered) / len(signed) >= cover_frac), sorted(signed), proposed + + +def run(chunks, gt): + sm = SP.SpreadModel(chunks) + rec = tot = 0 + rows = [] + for e in gt: + if not e.dst or X.gt_occurrences(e, chunks) == 0: + continue + r, signed, proposed = canon_recovery_lift(sm, e) + if r is None: + continue + tot += 1 + rec += bool(r) + rows.append(dict(src=e.src, typ=e.typ, recovered=bool(r), signed=signed, proposed=proposed)) + return rows, (rec / tot if tot else None), tot + + +if __name__ == "__main__": + chunks = X.load_chunks(); gt = X.load_gt() + sm = SP.SpreadModel(chunks) + print("lift-based dst extraction (fixes ubiquitous-term degeneration) — full 57ch:\n") + for s in ["蛊", "方源", "古月", "蛊师", "转", "元石", "族长"]: + e = next(x for x in gt if x.src == s) + prop, lifted = canon_propose_lift(sm, e) + print(f" {s:<4} signed=«{e.dst}» -> lift-proposed={prop} top-lift={[(l,v) for l,v,_ in lifted[:4]]}") + rows, rate, tot = run(chunks, gt) + print(f"\ncanon-recovery (lift) on records.json full 57ch: {rate:.3f} (n={tot})") diff --git a/eval/exp16/coldstart_analyze.py b/eval/exp16/coldstart_analyze.py new file mode 100644 index 0000000..8979d3f --- /dev/null +++ b/eval/exp16/coldstart_analyze.py @@ -0,0 +1,198 @@ +#!/usr/bin/env python3 +"""exp16 — cold-start analysis: spread + canon-recovery (confound removed) + A4/A5 banknote (research/20 +§D1 cold-start, §D2 A4/A5, §D5 decision rules). $0 (analysis of the paid cold-start drafts). + +Compares the injection-confounded records.json drafts (§D1 lower bound) against the cold-start drafts +(no glossary injection — ecological W1.5 input) on the SAME chapters [1,4,5,7,9]: + • spread on GT terms: injected ≈0 (suppressed) vs cold-start (the real dispersion signal) + • canon-recovery: does §C2 recover the owner-signed dst from an UN-anchored draft? (§D5 hypothesis-1 gate) + • A4/A5 banknote: marginal recall on f<3, parse_fail/truncated rates, quality delta (plain vs banknote), + real token cost. §D5 rule applied verbatim. +""" +from __future__ import annotations + +import json +from pathlib import Path + +import exp16_common as X +import spread as SP +import canon as CANON +import detectors as D +import patterns as P +import arms as A + +BOOK = Path("/home/ubuntu/books/gu-zhenren") +COLD = BOOK / "exp16" / "coldstart" +COLD_CHAPTERS = [1, 4, 5, 7, 9] + + +def load_coldstart_chunks(config="plain"): + """Build Chunk objects: source from records.json, draft from the cold-start config output.""" + recs = {(r["chapter"], r["chunk_idx"]): r for r in json.load(open(BOOK / "rerun" / "records.json"))} + out = [] + src_dir = COLD / config + for (ch, ck), r in recs.items(): + if ch not in COLD_CHAPTERS: + continue + raw = src_dir / (f"{ch}.{ck}.clean.txt" if config == "banknote" else f"{ch}.{ck}.raw.txt") + if not raw.exists(): + continue + c = X.Chunk(chapter=ch, chunk_idx=ck, source=r["source"], draft=raw.read_text(encoding="utf-8")) + c.nsource = X.norm(c.source) + out.append(c) + out.sort(key=lambda c: (c.chapter, c.chunk_idx)) + return out + + +def injected_chunks_for_cold(): + all_ch = X.load_chunks() + return [c for c in all_ch if c.chapter in COLD_CHAPTERS] + + +def gt_in(chunks, gt): + return [e for e in gt if X.gt_occurrences(e, chunks) >= 1] + + +def spread_compare(gt, inj_chunks, cold_chunks): + sm_inj = SP.SpreadModel(inj_chunks) + sm_cold = SP.SpreadModel(cold_chunks) + rows = [] + for e in gt: + occ = X.gt_occurrences(e, cold_chunks) + if occ < 2: # spread needs >=2 occurrences + continue + sn = X.norm(e.src) + rows.append(dict(src=e.src, typ=e.typ, occ=occ, + spread_injected=round(sm_inj.spread(sn), 3), + spread_coldstart=round(sm_cold.spread(sn), 3))) + return rows, sm_inj, sm_cold + + +def canon_compare(gt, inj_chunks, cold_chunks): + gt_cold = gt_in(cold_chunks, gt) + rows_inj, rate_inj, tot_inj = CANON.run(inj_chunks, gt_cold) + rows_cold, rate_cold, tot_cold = CANON.run(cold_chunks, gt_cold) + return dict(injected=(rate_inj, tot_inj), coldstart=(rate_cold, tot_cold)), rows_cold + + +def banknote_analysis(gt): + """A4/A5: aggregate banknote entries, marginal recall on f<3, parse_fail/truncated, quality delta.""" + bdir = COLD / "banknote" + entries, flags_all = [], [] + per_chunk = [] + for f in sorted(bdir.glob("*.banknote.json")): + d = json.load(open(f)) + entries += d["entries"] + flags_all.append(d["flags"]) + per_chunk.append((f.stem, d["flags"])) + proposed = {X.norm(e["src"]) for e in entries} + # recall of GT by banknote channel, by stratum + all_chunks = X.load_chunks() + def bank_recall(stratum=None): + hit = tot = 0 + for e in gt: + f = X.gt_occurrences(e, all_chunks) + if stratum and X.freq_stratum(f) != stratum: + continue + tot += 1 + hit += any(s in proposed for s in e.norm_surfaces) + return hit, tot + n_lines = sum(fl["n_banknote_lines"] for fl in flags_all) + n_parse_fail = sum(fl["banknote_parse_fail"] for fl in flags_all) + n_trunc = sum(fl["banknote_truncated"] for fl in flags_all) + nchunks = len(flags_all) + return dict(n_entries=len(entries), distinct_src=len(proposed), + recall_all=bank_recall(), recall_f_lt3=bank_recall("f<3"), + recall_f3_9=bank_recall("f3-9"), recall_f_ge10=bank_recall("f>=10"), + n_banknote_lines=n_lines, parse_fail_chunks=n_parse_fail, truncated_chunks=n_trunc, + nchunks=nchunks, parse_fail_rate=(n_parse_fail / nchunks if nchunks else 0), + sample_entries=entries[:20]), proposed + + +def quality_delta(): + """Deterministic draft-quality delta plain vs banknote (length ratio, char-count) — does the banknote + instruction degrade the translation? (§D5: 'дельта качества неотличима от нуля').""" + rows = [] + for f in sorted((COLD / "plain").glob("*.raw.txt")): + cid = f.stem.replace(".raw", "") + plain = f.read_text(encoding="utf-8") + bfile = COLD / "banknote" / f"{cid}.clean.txt" + if not bfile.exists(): + continue + bank = bfile.read_text(encoding="utf-8") + rows.append(dict(id=cid, plain_len=len(plain), banknote_clean_len=len(bank), + ratio=round(len(bank) / len(plain), 3) if plain else None)) + if rows: + import statistics + ratios = [r["ratio"] for r in rows if r["ratio"]] + return rows, (round(statistics.mean(ratios), 3), round(statistics.pstdev(ratios), 3)) + return rows, (None, None) + + +def main(): + gt = X.load_gt() + inj = injected_chunks_for_cold() + cold = load_coldstart_chunks("plain") + print(f"cold-start chunks: {len(cold)} (chapters {COLD_CHAPTERS}); injected comparison chunks: {len(inj)}\n") + if not cold: + print("no cold-start drafts yet — run coldstart_gen.py first"); return + + # 1. spread (confound removed) + srows, _, _ = spread_compare(gt, inj, cold) + print("== SPREAD: injected (confounded ~0) vs cold-start (real dispersion) ==") + import statistics + si = [r["spread_injected"] for r in srows] + sc = [r["spread_coldstart"] for r in srows] + print(f" mean spread injected={statistics.mean(si):.3f} vs cold-start={statistics.mean(sc):.3f} (n={len(srows)} GT terms occ>=2)") + for r in sorted(srows, key=lambda r: -r["spread_coldstart"])[:12]: + print(f" {r['src']:<6}({r['typ']:<7}) occ={r['occ']:<3} inj={r['spread_injected']} cold={r['spread_coldstart']}") + + # 2. canon-recovery + crates, rows_cold = canon_compare(gt, inj, cold) + print(f"\n== CANON-RECOVERY (§D5 hypothesis-1 gate: >=70% on cold-start) ==") + print(f" injected (confounded): {crates['injected'][0]:.3f} (n={crates['injected'][1]})") + print(f" cold-start (ECOLOGICAL): {crates['coldstart'][0]:.3f} (n={crates['coldstart'][1]})") + + # 3. A4/A5 banknote + bank, bproposed = banknote_analysis(gt) + print(f"\n== A4/A5 BANKNOTE CHANNEL ==") + print(f" entries={bank['n_entries']} distinct_src={bank['distinct_src']} banknote_lines={bank['n_banknote_lines']}") + print(f" parse_fail_chunks={bank['parse_fail_chunks']}/{bank['nchunks']} (rate {bank['parse_fail_rate']:.3f}) truncated={bank['truncated_chunks']}") + print(f" banknote recall: all={bank['recall_all']} f>=10={bank['recall_f_ge10']} f3-9={bank['recall_f3_9']} f<3={bank['recall_f_lt3']}") + + # marginal recall over best $0-arm (V-C) on f<3 + ar = A.Arms(X.load_chunks(), X.Contrast()) + a3set = A.candidate_set(ar.arm_A3()) + def marg_f_lt3(): + gained = [] + for e in gt: + if X.freq_stratum(X.gt_occurrences(e, X.load_chunks())) != "f<3": + continue + in_a3 = any(s in a3set for s in e.norm_surfaces) + in_bank = any(s in bproposed for s in e.norm_surfaces) + if in_bank and not in_a3: + gained.append(e.src) + return gained + gained = marg_f_lt3() + print(f" MARGINAL f<3 recall over V-C: banknote adds {gained} (V-C already had the rest)") + + # 4. quality delta + qrows, (qmean, qstd) = quality_delta() + print(f"\n== QUALITY DELTA (plain vs banknote clean draft) ==") + print(f" clean-length ratio banknote/plain: mean={qmean} std={qstd} (want ~1.0 = no degradation)") + + # cost + ledger = BOOK / "exp16" / "coldstart_costs.jsonl" + total = sum(json.loads(l).get("cost", 0) for l in open(ledger)) if ledger.exists() else 0 + print(f"\n== COST: ${total:.4f} (cap $10) ==") + + json.dump({"spread": srows, "spread_mean": {"injected": statistics.mean(si), "coldstart": statistics.mean(sc)}, + "canon": crates, "banknote": {k: v for k, v in bank.items() if k != "sample_entries"}, + "banknote_sample": bank["sample_entries"], "marginal_f_lt3": gained, + "quality_delta": {"mean": qmean, "std": qstd}, "cost": total}, + open(BOOK / "exp16" / "coldstart_analysis.json", "w"), ensure_ascii=False, indent=1) + print(f"[written] {BOOK/'exp16'/'coldstart_analysis.json'}") + + +if __name__ == "__main__": + main() diff --git a/eval/exp16/coldstart_gen.py b/eval/exp16/coldstart_gen.py index c67c988..49a9c38 100644 --- a/eval/exp16/coldstart_gen.py +++ b/eval/exp16/coldstart_gen.py @@ -119,14 +119,18 @@ def main(): rec = {"config": config, "id": cid, "model": MODEL, "usage": res.get("usage", {}), "cost": res.get("cost", 0.0), "finish": res.get("finish"), "err": res.get("err"), "latency_s": res.get("latency_s")} - # LENGTH-GATE FIX (D39.9): finish=length -> reject, do NOT accept truncated output + # TRUNCATION-GATE (D39.9 + self-review MAJOR): accept ONLY finish=='stop'. Reject/flag EVERY + # other finish_reason (length, DeepSeek's insufficient_system_resource peak-overload cut, + # content_filter, null/unknown) — a truncation via ANY reason must never be written to file. if res.get("err"): stats[config]["err"] += 1 sp.record(rec); print(f" ERR {config} {cid}: {res['err'][:70]}"); continue - if res.get("finish") == "length": + if res.get("finish") != "stop": stats[config]["length_reject"] += 1 - rec["rejected"] = "finish=length" - sp.record(rec); print(f" LENGTH-REJECT {config} {cid} (comp={res['usage'].get('completion_tokens')})") + rec["rejected"] = f"finish={res.get('finish')!r}!=stop" + sp.record(rec) + print(f" TRUNC-REJECT {config} {cid} finish={res.get('finish')!r} " + f"(comp={res['usage'].get('completion_tokens')})") continue sp.record(rec) text = res["text"] diff --git a/eval/exp16/emit_owner_sheets.py b/eval/exp16/emit_owner_sheets.py new file mode 100644 index 0000000..f07fc7f --- /dev/null +++ b/eval/exp16/emit_owner_sheets.py @@ -0,0 +1,219 @@ +#!/usr/bin/env python3 +"""exp16 — owner-facing deliverables (research/20 §C2-8 карта подписи, §E-1 мини-голд, §D1 precision@30). $0. + +Emits three markdown sheets for the owner (sidecars, NOT the seed schema — miner never writes approved): + 1. signature_map.md — per proposed entity: canon dst-proposal (by ALL occurrences), dst-variants, + top-KWIC contexts (H15 lever), gender counters (他/她), zone flags Z1-Z5, weak edges. + 2. alias_minigold.md — candidate entity-edges for the ~20-30 min owner touchpoint (strike/add). + 3. precision_at30.md — top-30 NON-seed A3 candidates for owner adjudication (§D1 precision@30 primary path). + +Uses cold-start drafts for canon/spread if present (ecological), else injected records.json (confounded). +""" +from __future__ import annotations + +import json +import re +from pathlib import Path + +import exp16_common as X +import spread as SP +import canon as CANON +import palladius as PAL +import arms as A +import alias as ALIAS + +BOOK = Path("/home/ubuntu/books/gu-zhenren") +OUT = BOOK / "exp16" +COLD_PLAIN = OUT / "coldstart" / "plain" + +_RE_HE, _RE_SHE = re.compile("他"), re.compile("她") + + +def kwic(term, chunks, window=18, maxn=3): + tn = X.norm(term) + out = [] + for c in chunks: + s = c.source + for m in re.finditer(re.escape(term), s): + i = m.start() + ctx = s[max(0, i - window):i + len(term) + window].replace("\n", " ") + out.append(f"…{ctx}…") + if len(out) >= maxn: + return out + return out + + +def gender_counts(term, chunks, window=40): + """他/她 counts within `window` chars of each occurrence — EVIDENCE ONLY (verdict = owner, D19.3).""" + he = she = 0 + for c in chunks: + s = c.source + for m in re.finditer(re.escape(term), s): + seg = s[max(0, m.start() - window):m.end() + window] + he += len(_RE_HE.findall(seg)) + she += len(_RE_SHE.findall(seg)) + return he, she + + +def zone_flags(ent, cand, sm, chunks): + """§B4 zone tags for the candidate (evidence, not verdict).""" + flags = [] + sn = X.norm(ent.src if hasattr(ent, "src") else cand.src) + sp = sm.spread(sn) + variants, _ = sm.dst_variants(sn) + top_dst = variants[0][0] if variants else "" + # Z1 suspicious_stable: high score + spread~0 + dst not a Palladius name and not capitalized-name + is_name_dst = PAL.is_palladius_token(top_dst, 2) and sm.is_name_lemma(top_dst) + if cand and cand.score > 500 and sp < 0.05 and top_dst and not is_name_dst and \ + (hasattr(ent, "typ") and ent.typ in ("term", "title")): + flags.append("Z1?suspicious_stable") + if sp >= 0.3: + flags.append("Z2?sense_split/spread") + f = X.gt_occurrences(ent, chunks) if hasattr(ent, "norm_surfaces") else cand.freq + if f < 3: + flags.append("Z3:rare") + return flags + + +def load_banknote_dst(): + """Aggregate translator-emitted src->dst from the cold-start banknote drafts (reliable dst; §3.3).""" + bdir = OUT / "coldstart" / "banknote" + dst = {} + for f in sorted(bdir.glob("*.banknote.json")): + for e in json.load(open(f))["entries"]: + sn = X.norm(e["src"]) + dst.setdefault(sn, []).append(e["dst"]) + # majority dst per src + from collections import Counter + return {sn: Counter(v).most_common(1)[0][0] for sn, v in dst.items()} + + +def build_signature_map(gt, chunks, sm, a3, tag, bank_dst=None): + bank_dst = bank_dst or {} + lines = [f"# Карта подписи exp16 (miner-v1, {tag}) — сайдкар владельцу", + "", + "> Майнер НИКОГДА не пишет `approved`. Ниже — предложения `draft` по кластерам: канон-предложение " + "(по ВСЕМ вхождениям × Палладий-конформность), dst-варианты, KWIC-контексты (детерминированный " + "H15-рычаг), счётчики пола (他/她 — evidence, вердикт ваш, D19.3), флаги зон §B4. " + "Владелец: approve / правка / оставить draft / расклеить.", + ""] + a3_by = {c.src: c for c in a3} + # cover GT entities (the measurable set) + top non-GT name/place candidates + gt_by_norm = {X.norm(s): e for e in gt for s in e.surfaces} + seen = set() + for e in sorted(gt, key=lambda e: -X.gt_occurrences(e, chunks)): + sn = X.norm(e.src) + if sn in seen: + continue + seen.add(sn) + cand = a3_by.get(sn) + proposed, variants = CANON.canon_propose(sm, e) + vstr = ", ".join(f"{lm}({d})" for lm, d, _ in variants[:4]) + he, she = gender_counts(e.src, chunks) + zf = zone_flags(e, cand, sm, chunks) + occ = X.gt_occurrences(e, chunks) + # dst proposal priority: banknote (translator-emitted, reliable) > Palladius-name canon > co-occ note + bank = bank_dst.get(sn) + if bank: + canon_str = f"«{bank}» (banknote-канал)" + elif proposed: + canon_str = f"«{' '.join(proposed)}» (Палладий-канон)" + else: + canon_str = "(dst — банкнота/облако: co-occ извлечение вырождено, §3.3)" + lines.append(f"### {e.src} — предл. dst: {canon_str} " + f"[{e.typ}, occ={occ}, статус сида={e.status}, подписанный dst=«{e.dst}»]") + lines.append(f"- co-occ dst-варианты (справочно): {vstr}") + if he or she: + lines.append(f"- пол-evidence: 他={he} 她={she} (вердикт — владелец)") + if zf: + lines.append(f"- зоны §B4: {', '.join(zf)}") + for k in kwic(e.src, chunks): + lines.append(f"- KWIC: `{k}`") + lines.append("") + return "\n".join(lines) + + +def build_minigold(gt, chunks, sm): + gt_by_norm = {X.norm(s): e for e in gt for s in e.surfaces} + ar = A.Arms(chunks, X.Contrast()) + a3 = ar.arm_A3() + subsumed = ar.va.subsumed + gt_surf = {X.norm(s) for e in gt for s in e.surfaces} + PARTICLE = ALIAS.__dict__.get("PARTICLE", set()) + name_cands = [c.src for c in a3[:200] if any(t in c.types for t in ("name", "place", "title")) + and c.freq >= 5 and c.src not in subsumed and len(c.src) >= 2] + surface_srcs = list(gt_surf | set(name_cands)) + surfaces = ALIAS.build_surfaces(surface_srcs, sm, gt_by_norm) + ident, family, weak = ALIAS.propose_edges(surfaces, chunks, sm, frozenset(gt_surf)) + clusters = ALIAS.cluster(ident, list(surfaces)) + lines = ["# Мини-голд алиасов exp16 — тачпойнт владельца (~20–30 мин)", + "", + "> Код ПРЕДЛОЖИЛ рёбра сущностей слайса-25. Вычеркните ложные, допишите пропущенные. Только после " + "этого меряется recall кластеризации (сейчас — только precision предложенных). Тир-1 имеет " + "известную ошибку (§B2): спорное держим раздельным draft.", "", + "## Предложенные IDENTITY-кластеры (одна сущность):"] + for cl in clusters: + lines.append(f"- [ ] {{ {', '.join(cl)} }}") + lines += ["", "## Предложенные FAMILY-суперкластеры (общая фамилия, НЕ тождество):"] + fam_pairs = sorted({tuple(sorted((e['a'], e['b']))) for e in family}) + for a, b in fam_pairs[:40]: + lines.append(f"- [ ] {a} ~ {b}") + lines += ["", "## Слабые/tier-2 рёбра (со-оккуренция, решения нет — предложения к связи):"] + for e in weak[:30]: + lines.append(f"- [ ] {e['a']} ? {e['b']} [{e.get('rule','')}]") + return "\n".join(lines) + + +def build_precision30(gt, chunks, sm, a3): + gt_surf = {X.norm(s) for e in gt for s in e.surfaces} + non_seed = [c for c in a3 if c.src not in gt_surf][:30] + lines = ["# precision@30 — топ-30 НЕ-сидовых кандидатов A3 (V-C) для адъюдикации владельца", + "", + "> Отметьте: T = валидная сущность банка (имя/титул/место/термин), F = мусор/фрагмент, " + "? = спорно. Это первичный путь precision@30 (§D1; precision@сид НЕ метрика — сид не исчерпывающий).", + "", "| # | кандидат | freq | типы | evidence | dst-намёк | T/F/? |", + "|---|---|---|---|---|---|---|"] + for i, c in enumerate(non_seed): + variants, _ = sm.dst_variants(c.src, top=2) + dsth = variants[0][0] if variants else "" + ev = ",".join(c.evidence[:2]) + lines.append(f"| {i+1} | {c.src} | {c.freq} | {'/'.join(c.types) or '-'} | {ev} | {dsth} | |") + return "\n".join(lines) + + +def main(): + gt = X.load_gt() + # prefer cold-start drafts (ecological) for canon/spread; fall back to injected + from coldstart_analyze import load_coldstart_chunks, COLD_CHAPTERS + cold = load_coldstart_chunks("plain") + if cold: + chunks_for_canon = cold + tag = f"cold-start ch{COLD_CHAPTERS}" + else: + chunks_for_canon = X.load_chunks() + tag = "injected records.json (confounded — cold-start refines)" + full = X.load_chunks() + sm = SP.SpreadModel(chunks_for_canon) + ar = A.Arms(full, X.Contrast()) + a3 = ar.arm_A3() + + # signature map over the canon-source chunks + gt_canon = [e for e in gt if X.gt_occurrences(e, chunks_for_canon) >= 1] + bank_dst = load_banknote_dst() + sig = build_signature_map(gt_canon, chunks_for_canon, sm, a3, tag, bank_dst=bank_dst) + (OUT / "signature_map.md").write_text(sig, encoding="utf-8") + + sm_full = SP.SpreadModel(full) + mg = build_minigold(gt, full, sm_full) + (OUT / "alias_minigold.md").write_text(mg, encoding="utf-8") + + p30 = build_precision30(gt, full, sm_full, a3) + (OUT / "precision_at30.md").write_text(p30, encoding="utf-8") + + print(f"[written] {OUT}/signature_map.md ({len(gt_canon)} entities, canon source: {tag})") + print(f"[written] {OUT}/alias_minigold.md") + print(f"[written] {OUT}/precision_at30.md (top-30 non-seed A3)") + + +if __name__ == "__main__": + main()