diff --git a/.gitignore b/.gitignore index de8d598..5df23d8 100644 --- a/.gitignore +++ b/.gitignore @@ -30,3 +30,6 @@ backend/dist/ # Claude Code — личные (некоммитируемые) настройки прав .claude/settings.local.json *.key.json + +# exp16 versioned contrast corpus (jieba 0.42.1 dict.txt, 5MB) — SHA pinned in the report, reproducible +eval/exp16/data/ diff --git a/docs/experiments/16-bank-mining.md b/docs/experiments/16-bank-mining.md new file mode 100644 index 0000000..1b7b42f --- /dev/null +++ b/docs/experiments/16-bank-mining.md @@ -0,0 +1,192 @@ +# Эксперимент 16. Банк-майнинг W1.5 (полигон-стадия research/20 §D) + +> **Статус:** полигон-сессия exp16 (2026-07-18), исполняет пре-рег дизайн **research/20 §D** (дизайн-оф-рекорд, +> D39.6). Зона: `eval/exp16/` + этот файл. Артефакты — ВНЕ git (`/home/ubuntu/books/gu-zhenren/exp16/`). +> **Коммичу ТОЛЬКО пре-рег фриз** (единственный коммит, ДО первого платного вызова — D30.10/D37); результаты +> сдаются этим отчётом, лендит оркестратор. Бюджет платных армов: жёсткий кап **$10** (решение владельца 17–18.07). +> +> Структура: **§0** пре-задача (ре-аудит exp14b, D39.9) · §1 пре-рег фриз · §2 GT/материал · §3 результаты по +> армам · §4 решающие правила + маршрутизация слепых зон · §5 деньги · §6 самопроверка + отклонения. + +--- + +## §0. Пре-задача ($0, ДО фриза): ре-аудит exp14b исправленными правилами (D39.9) + +**Мандат (промт §пре-задача):** Q4a (D39.9) вскрыл, что (i) правила скорера exp14b +(`rule_counterfactual`/`polarity_envy`/`b1`) дефектны, а исправленные оверрайды лежат в +[q4a_traps.py](../../eval/exp15/q4a_traps.py); (ii) харнес [exp14_common.call](../../eval/exp14/exp14_common.py#L104-L113) +НЕ отклоняет `finish=length` — усечённый выход возвращается как валидный (`err=None`). Задача: (1) пере-скорить +СОХРАНЁННЫЕ exp14b-выходы исправленными правилами; (2) свип `finish_reason=length` по exp14/14b-леджерам; +(3) отчёт-дельта: какие пер-класс вердикты D38 сдвинулись. **Флип любого несущего → СТОП + пинг оркестратору +(errata — его зона).** + +Код (аддитивный, риг exp14/14b НЕ тронут — импорт read-only): [reaudit_14b.py](../../eval/exp16/reaudit_14b.py), +[length_sweep.py](../../eval/exp16/length_sweep.py). Length-гейт для СВОИХ будущих генераций чинится у себя +(exp16-харнес), риг exp14 не трогаю. + +### §0.1. Результат ре-скоринга (задача 1) — D38 DET-ядро ДЕРЖИТСЯ + +D38 (`14b-meaning-battery.md` §2.1) скорил DET **ручной span-читкой** (авто-скорер был само-отвергнут как +бракованный). Исправленный `q4a_traps` — это тот самый скорер, но починенный (предложение-скоуп, исправленные +a1/a2/b1/b2/c1/d1/d2; звук b2, c2=`rule_grade_bing`, c3=`rule_patriarch` реюзаны verbatim). Self-test +корректных правил: 19/19 pass. Ре-скор всех сохранённых выходов арма C/F/X/D/M (+ K где есть, + P0-референс): + +| Трап (класс) | glm-5 C | gpt-5.4 F | grok X | deepseek-pro D | mistral M | сходится с D38? | +|---|---|---|---|---|---|---| +| **a1** двойн.отриц (крит) | fail | **fix** (лок.gap→разрешён) | fail | fix | fix | ✅ все | +| **a2** 无不 | fix | fix | fix | fix | fix | ✅ все | +| **b2** 四成四=44% | fix | fix | fix | fix | fix | ✅ все | +| **c1** 人上之人 (крит) | fix | **fix** (лок.gap→разрешён) | fail | fail | fix* | ✅ (M: `?`→fix, апгрейд) | +| **c3** 族长 (глосс-конфаунд) | fail | fix | fix | fix | fail | ✅ все | + +**Каждая ячейка, которую исправленный скорер СМОГ локализовать, совпадает с ручным вердиктом D38 дословно** — +включая паттерн глосс-конфаунда c3 (C/M=fail «старейшина», F/X/D=fix «глава») и c1 M=`?`. Единственные два +расхождения — **a1 F** и **c1 F**: D38=fix → исправленный скорер=`?` (пустой span). Это НЕ семантический флип, +а **пробел покрытия локатора** q4a на перефразе gpt-5.4 (окно story→knowledge 90 симв. перелетело; `над людьми` +разорвано словом `обычными`). `?` = «не локализовано / не классифицируемо — ИСКЛЮЧЕНО из ставок, НИКОГДА не +тихий pass». Разрешение span-читкой (мануально + независимо, §0.3): gpt-5.4 верно рендерит оба +(«…не было ни одного человека, который бы его не знал» = верная двойная негация; «…подняться над обычными +людьми» = верное 人上之人). **D38 «fix» стоит.** + +Исправленные fix-rate по армам (9 DET-инстансов, `?` исключён из знаменателя): C 7/9 · F 5/6 · X 7/9 · +D 7/9 · M 5/7 · P0 4/5. (b1/c2/d1/d2 не были в ручной D38-таблице §2.1 — свежие данные, не «сдвиги».) + +**Несущие выводы D38 под исправленными правилами — БЕЗ изменений:** +- «Только gpt-5.4 чинит смысл» **опровергнуто**: a1 чинят F, **D и M**; валят glm(C) и grok(X). Провалы + РАЗБРОСАНЫ по модель×конструкция (a1: C+X; c1: X+D; c3-глосс: C+M). Нет модели безупречной/безнадёжной. +- c3 = **глоссарий-енфорсмент**, не способность (C/M подтянули approved-сосед `家老→старейшина` к draft-целевому + `四代族长`) — подтверждено внутренней консистентностью (ни один арм не мешает глава/старейшина). + +### §0.2. Свип finish_reason=length (задача 2) — ни одна несущая DET-ячейка не заражена + +8 леджеров, 477 записей: `stop` 413 · `length` **32** · `(none)` 32. Разбор 32 `length`-записей: + +| Источник | n | Приняты как валидные? | Затрагивает несущую DET-ячейку? | +|---|---|---|---| +| gemini-3.1-pro (G-base/G-disc, exp14) | 11 | да (баг-гейт) | **нет** — gemini ДРОПНУТ как арм (exp14b §1.1); не в C/F/X/D/M | +| kimi-k2.6 (K-арм + судьи, exp14/14b) | 19 | 2 приняты (10.1/11.1), 17 отвергнуты (empty) | **нет** — K-арм ДРОПНУТ (D38); 10.1/11.1 не несущие | +| deepseek-v4-pro **D 2.1** (exp14b) | 1 | **да** (comp_tok=8000, оборван на «…Ступа[й]») | **нет** — 2.1 = e2 (SOFT/судейский), не DET; **НОВАЯ находка** | +| gpt-5-mini (regate-судья) | 1 | нет (отвергнут empty) | нет — судья, не редактор | + +Прямая проверка длин 7 несущих DET-чанков (7.0/17.0/16.0/6.0/9.1/10.1/19.0) по C/F/X/D/M: все консистентны и +завершены (ни один аномально короткий, ни один в свипе). **Несущие DET-вердикты сидят на полных выходах.** +D 6.0 (упомянут в D38 как оборванный→перегнан на 16k) — сохранённый файл полный (10054 симв, `stop`); в свипе +как `length` не значится (финальная перезапись). **Новая находка:** D 2.1 (e2, SOFT) — принятая усечёнка; +судьи e2 скорили оборванный выход D. Влияния на несущее нет (e2-вердикт D38 = «большинство correct, +чэнъюй-сплющивание универсально» — усечёнка не рождает ложную катастрофу), но фиксирую как риг-гигиену. + +### §0.3. Независимая верификация несущих (author≠reviewer) + дельта-вердикт (задача 3) + +Промт требует: «несущие a1-вердикты span-верифицированы независимо». Исполнено воркфлоу из 4 независимых +refute-by-default ридеров (a1 ×2, c1, c3), читавших СЫРЫЕ выходы армов, вердикт fix/fail/omit + цитата-span. +**Все 4 подтвердили каждую несущую ячейку**, включая двойное подтверждение a1 F=fix и c1 F=fix (моё +разрешение локатор-gap'ов независимо подтверждено). a1: C=fail×2, F=fix×2, X=fail×2, D=fix×2, M=fix×2. c1: +C/F/M=fix, X/D=fail. c3: C/M=fail, F/X/D=fix. + +**ДЕЛЬТА-ВЕРДИКТ:** **0 семантических флипов несущих вердиктов D38.** DET-ядро D38 держится под исправленными +правилами, тройно подтверждено (D38-ручной × исправленный авто-скорер × независимый refute-by-default ридер). +Два расхождения (a1 F, c1 F) — пробелы покрытия локатора, разрешены в `fix`; одно улучшение (M c1 `?`→fix). +**СТОП-гейт НЕ сработал; пинг оркестратору по флипу НЕ требуется** (условие пинга — флип, его нет). + +**Риг-находки в реестр (не блокеры, для протокола):** (а) исправленный `q4a_traps`-локатор имеет пробелы +покрытия на не-S2′ материале → выдаёт `?` (безопасно: не тихий pass) — подтверждает правоту D38 скорить DET +ручной span-читкой; (б) харнес exp14 принял усечёнку D 2.1 (e2 SOFT) — единственная новая жертва length-бага +вне дропнутых армов; (в) для СВОИХ генераций exp16 length-гейт реализуется в exp16-харнесе (отклонять +`finish=length` как reject/flag) — §1. + +Артефакты: `exp16/reaudit_14b.json`, `exp16/length_sweep.json`, `exp16/pretask_independent_verify.json`. + +**⟶ Гейт пройден: ядро держится. Перехожу к пре-рег фризу (§1).** + +--- + +## §1. Пре-рег фриз (miner-v1) — коммит ДО первого платного вызова (D30.10/D37) + +Замораживается дословно из research/20 §D + конкретизации. Детектор-код детерминирован (сортировки, +версия `miner-v1`, версионированные словари) — повторный прогон байт-в-байт (§D4-е). + +### 1.1. Материал и Ground Truth +- **GT = сид v2 пост-reseed**, `guzhenren-seed-v2.yaml` **sha256 `175a67ad2f05…c408f4`** (совпадает с + реф-хешем промта; 人祖=male, 古月=n-a подписаны — exp15-преп). **GT = 57 термов + 2 алиас-поверхности.** +- **Фильтр вхождения (§D1):** каждый GT-терм имеет ≥1 вхождение src/алиаса в source (Aho-Corasick-эквивалент + БЕЗ suppressContained, D24.2). Проверено исполнением: 57/57 проходят **с** аннотацией, 54/57 **без**. +- **BLURB-ПРАВИЛО (решение полигона, пре-рег):** аннотация гл.1/чанк0 СЧИТАЕТСЯ вхождением ⇒ **GT=57**. + 3 blurb-терма (血颅蛊 since193, 一气金光虫/青丝蛊 since47) встречаются по 1 разу только в аннотации ⇒ + страта **f<3**. (Альтернатива «аннотация ≠ вхождение» дала бы GT=54; репортим f<3-recall в обоих + разрезах для этих 3 термов.) **Частотные страты GT (incl. аннотация): f≥10 — 40 · f3-9 — 7 · f<3 — 10.** +- **(source, draft) пары:** `rerun/records.json` **sha256 `3de49eaf…928d`** — 57 чанков (гл.1–25), + черновики deepseek-v4-flash. ⚠ **Конфаунд инъекции (§D1):** черновики генерились С глоссарий-инъекцией + сида (budget 800, `pipeline-rerun.yaml`) ⇒ спред на GT подавлен ⇒ метрики спреда/канона на них = + НИЖНЯЯ граница; несущие выводы — по cold-start срезу (§1.6). + +### 1.2. Армы (A1–A3+абл) +| Арм | Детектор | Определение | +|---|---|---| +| **A1** | V-A | Han-n-граммы 1–6 × частотный floor 3 × c-value (nested-дисконт, g(L)=log₂(L+1)) × weirdness (контраст с общим zh) | +| **A2** | V-B | A1 + translation-spread (Dice/LTCR по ru-черновикам, pymorphy3-леммы) + dst-варианты | +| **A3** | V-C | A2 + паттерны: 百家姓-якорь, титул/топо-суффиксы, rank/grade-композиция, продуктивная морфология (авто-формант по over-rep), Палладий ru-канал | +| **A3-абл** | V-C, λ=0 | V-C без spread-сигнала — изолирует вклад каждого сигнала (§D2) | + +### 1.3. Замороженные пороги (miner-v1) — выбраны на тюнинг-половине ch1–15, ±50%-стабильны +`freq_floor=3` (§A1 C-value канон) · `subsume_alpha=0.80` (nested-консолидация ранжирования; 0.5 +слишком агрессивно роняет GT) · `ngram_max=6` · `formant_min_over_rep=15.0` · `formant_min_partners=3` +· `lam=0.0` на инъектированных черновиках (spread вредит ранжированию — конфаунд §D1; на cold-start λ +свипается) · pattern-бонусы `{name/place 140, title 120, term 80}` × source-weight. **Свип ±50% (§1 +tuning_sensitivity.json): A3@f≥3=1.0 инвариантно по ВСЕМ настройкам; catastrophe PASS везде.** + +### 1.4. Метрики (все детерминированные, $0, кроме адъюдикации) +1. **recall@PROPOSED** (членство в кандидат-сете = ось «где арм слеп», §D3) — первичная recall-метрика, + по типам (name/title/place/term/nickname) и стратам (f≥10/3-9/<3). +2. **recall@top-K + pseudo-precision@K** (ось ранжирования/точности) + trade-off кривая. +3. **precision@30 адъюдицированная** (топ-30 не-сидовых A3): владелец по карте ИЛИ кросс-семейный судья + закалённого паттерна `judges.py` (per-vote, оба порядка, полно-evidence) — precision@сид НЕ метрика (§D1). +4. **canon-recovery** (§C2): покрытие подписанных лемм предложением канона; несущее — на cold-start. +5. **алиас-precision** рёбер R1–R3 (entity-B³, не MUC) + мини-голд владельца для recall (§E-1 тачпойнт). +6. **стабильность порогов** ±50%. 7. **A6b** accuracy верификации/линковки. + +### 1.5. Порог-дисциплина 60/40 (анти-подгонка, §D1) +Тюнинг = ch1–15, тест = ch16–25. Пороги выбраны на тюнинге, заморожены этим коммитом; **первичный +репортируемый результат — тест-половина**; full-срез — вторичка (catastrophe/trade-off/дамп). + +### 1.6. Cold-start срез (пре-рег фиксирует главы) +**Главы [1, 4, 5, 7, 9]** (жадное GT-покрытие: **52/57** сущностей, **14 чанков** — 6 в гл.1). Свежие +flash-черновики БЕЗ глоссарий-инъекции, 2 конфига: `plain` (=срез, spread+canon-recovery) и `banknote` +(§B3-инструкция, канал сноски A4). Промт translator — `backend/prompts/translator.md` **sha256 +`35d3ad6e…`**, brief из `rerun/book.yaml`, БЕЗ инъекц-блока (порт `coldstart_gen.py`). ⚠ **DeepSeek-долина +обязательна** (пики UTC 01–04/06–10 ×2, D39.7 — `coldstart_gen.in_deepseek_peak()` отказывает в пик). + +### 1.7. Катастроф-скрин (§D4-в) +方源 · 蛊 · 蛊师 · 古月 обязаны быть в **топ-50 победителя**; промах = провал арма. **Пройден на full +(ranks 0/1/2/13/…)** — g(L)=log₂(L+1) держит одночарный 蛊 (nested в 蛊师/蛊虫/月光蛊) на ранге 1. + +### 1.8. Карта независимости сигналов (§D4-а — против manufactured convergence) +Сигналы репортятся РАЗДЕЛЬНО: A1(частота×контраст), spread(A2−A1 / A3−A3абл), паттерны(A3−A1). Сведение +«N сигналов сходятся» — только с leave-one-out. Spread на инъектированных = ~0/негатив (изолирован), +измеряется экологически лишь на cold-start. + +### 1.9. Бюджет и порядок (§D5) +Платный кап **$10** (владелец 17–18.07), per-call gate + hard-cap (`exp15_llm.Spender`). Порядок: +A1–A3+абл ($0) → алиас ($0) → A6/A6b (стенд $0) → cold-start (копейки) → A4/A5 (платные, последними, +отменяемы). **Length-гейт (D39.9):** ВСЕ свои генерации отклоняют `finish=length` (не наследуем баг exp14). + +### 1.10. Хеши (версионирование детектора) +- Код `miner-v1` (SHA256): exp16_common `92bc737a` · detectors `89059e53` · patterns `7ac8240c` · + spread `5422415f` · palladius `a071295e` · arms `a5ed2ae2` · run_arms `c4e3e17d` · banknote `069ad8fe` · + coldstart_gen `70c8d9c1` · alias `354172fa` · canon `e947ebf3` · reaudit_14b `8d95ad58` · length_sweep `59a60b15`. +- Словарь-артефакт (контраст): `data/jieba_dict_general_zh.txt` **sha256 `7197c321…e12a8`** (jieba 0.42.1 + `dict.txt`, общий zh word-freq; вне git — воспроизводим из jieba 0.42.1, SHA пиннится здесь). +- Сид `175a67ad…` · records `3de49eaf…` · translator.md `35d3ad6e…` · бэкенд HEAD `b9e62a7`. + +### 1.11. Решения владельца, вшитые (17–18.07, НЕ пересматривать) +Жанр-паки НЕ строятся (V-C только универсальные каналы; авто-формант, не хардкод 蛊). Майнер НИКОГДА +не пишет `approved` (auto/draft + карта подписи); канон = по ВСЕМ вхождениям × Палладий-конформность × +полнота леммы. Пол в сиде подписан (人祖=male, 古月=n-a); счётчики 他/她 = evidence, вердикт = владелец. +ja→ru реплика — след-за-§D (отдельный мини-фриз). + +**⟶ Фриз коммитится. Результаты (§3+) — ПОСЛЕ, лендит оркестратор.** + +--- + +## §2. GT и материал (см. §1.1) · §3. Результаты по армам · §4. Решающие правила + слепые зоны · §5. Деньги · §6. Самопроверка + +*(заполняются после прогонов; тест-половина — первичка)* diff --git a/eval/exp16/alias.py b/eval/exp16/alias.py new file mode 100644 index 0000000..34c1afb --- /dev/null +++ b/eval/exp16/alias.py @@ -0,0 +1,188 @@ +#!/usr/bin/env python3 +"""exp16 — alias tier-1 clustering (research/20 §B2). miner-v1. $0. + +PROP-layer, precision-safe rules over a set of src surfaces + their dominant ru rendering: + R1 surface containment X ⊂ Y, |X|>=2 -> 'extension' edge (方源 ⊂ 古月方源) + R2 surname anchor+compose shared surname -> FAMILY supercluster edge (NOT identity) + R3 shared ru rendering same dominant ru lemma-set -> strong IDENTITY edge (古月方源,方源 -> Фан Юань) + R4 NEGATIVE constraints (block a merge): + (i) same surname + DIFFERENT given names (方源 vs 方正) -> not identity + (ii) different confirmed gender + (iii)different approved dst + (iv) co-presence in one source sentence/dialogue turn -> not the same entity + (v) title-nesting with different seed dst (族长 ⊂ 四代族长) -> different entities + +Output: identity clusters (R1/R3 minus R4), family superclusters (R2), and tier-2 weak-link proposals. +Precision measured entity-level (B³ style, §A5 — NOT MUC link metric). Recall needs the owner mini-gold +(§E-1): this module also emits a candidate-edge sheet for the ~20-30 min owner touchpoint. +""" +from __future__ import annotations + +import json +from collections import defaultdict +from dataclasses import dataclass, field + +import exp16_common as X +import spread as SP +import palladius as PAL + +SURNAMES = None # lazy import to avoid cycle + + +def _surname_of(src_norm: str): + import patterns as P + return P.is_surname_start(src_norm) + + +@dataclass +class Surface: + src: str # normalized src surface + dst_lemmas: list = field(default_factory=list) # dominant ru rendering lemma(s) + gender: str = "" + approved_dst: str = "" + typ: str = "" + + +def dominant_dst(sm: SP.SpreadModel, src_norm: str, top=3): + variants, _ = sm.dst_variants(src_norm, top=top) + # keep the ru lemmas that look like a NAME rendering (Palladius) OR the very top associates + return [lm for lm, d, _ in variants] + + +def cooccur_same_sentence(chunks, a_norm: str, b_norm: str) -> bool: + """R4-iv: do a and b co-occur within one source sentence anywhere?""" + import regex + for c in chunks: + for sent in regex.split(r"[。!?\n]", c.nsource): + if a_norm in sent and b_norm in sent: + return True + return False + + +def build_surfaces(surface_srcs, sm, gt_by_norm) -> dict[str, Surface]: + out = {} + for s in surface_srcs: + sn = X.norm(s) + if not sn: + continue + ent = gt_by_norm.get(sn) + out[sn] = Surface(src=sn, dst_lemmas=dominant_dst(sm, sn), + gender=(ent.gender if ent else ""), + approved_dst=(ent.dst if ent and ent.status == "approved" else ""), + typ=(ent.typ if ent else "")) + return out + + +def propose_edges(surfaces: dict[str, Surface], chunks, sm, gt_norm_surfaces=frozenset()): + _SM = sm + ident, family, weak = [], [], [] + keys = list(surfaces) + for i in range(len(keys)): + for j in range(i + 1, len(keys)): + a, b = surfaces[keys[i]], surfaces[keys[j]] + # ── R4 HARD BLOCKS (checked first; no exceptions) ── + # R4-ii different confirmed gender + if a.gender and b.gender and a.gender != b.gender and "hidden" not in (a.gender, b.gender): + continue + # R4-iii/v different approved dst -> different entities, EVEN IF one contains the other + # (this is the 族长⊂四代族长 lesson D38 §3 and the clan⊂person case 古月⊂古月方源). + if a.approved_dst and b.approved_dst and X.norm(a.approved_dst) != X.norm(b.approved_dst): + continue + sur_a, sur_b = _surname_of(a.src), _surname_of(b.src) + # R1 containment -> extension edge (a and b are the SAME entity, fuller vs shorter surface). + # Guard against boundary fragments: the SHORTER surface must itself be a GT surface or a + # multi-char pattern-typed entity (not a longer±particle artifact). + if len(a.src) >= 2 and len(b.src) >= 2 and (a.src in b.src or b.src in a.src) and a.src != b.src: + short, long_ = (a, b) if len(a.src) < len(b.src) else (b, a) + short_is_entity = bool(short.typ) or short.src in gt_norm_surfaces + # COMPOSITIONAL guard: long = short + Y (or Y + short) where Y is ANOTHER known entity of + # type title/place/term => long is a PHRASE (古月+族长 = clan-head phrase), not an alias. + # clan/surname + NAME stays a fullname alias (古月+方源). Blocks the 古月↔族长 chaining. + rest = long_.src[len(short.src):] if long_.src.startswith(short.src) else long_.src[:-len(short.src)] + rest_ent = surfaces.get(rest) + compositional = bool(rest) and rest in gt_norm_surfaces and \ + (rest_ent is None or rest_ent.typ in ("title", "place", "term")) + if compositional: + weak.append(dict(a=a.src, b=b.src, rule="R1-compositional", kind="phrase_not_alias", part=rest)) + elif short_is_entity: + ident.append(dict(a=a.src, b=b.src, rule="R1", kind="extension")) + else: + weak.append(dict(a=a.src, b=b.src, rule="R1-fragment?", kind="containment_weak")) + continue + # R2 surname anchor -> family (NOT identity) + if sur_a and sur_b and sur_a == sur_b and a.src != b.src: + # R4-i same surname + different given names -> NOT identity, family only + given_a, given_b = a.src[len(sur_a):], b.src[len(sur_b):] + if given_a != given_b and given_a and given_b: + if cooccur_same_sentence(chunks, a.src, b.src): + family.append(dict(a=a.src, b=b.src, rule="R2+R4iv", kind="family_copresent")) + else: + family.append(dict(a=a.src, b=b.src, rule="R2", kind="family")) + continue + # R3 shared ru rendering -> identity (strong), unless R4-iv co-presence. + # Palladius tokens must ALSO be capitalized-predominant name lemmas (blocks 'найти'=най+ти FP). + palla = [lm for lm in a.dst_lemmas[:2] if PAL.is_palladius_token(lm, 2) and _SM.is_name_lemma(lm)] + pallb = [lm for lm in b.dst_lemmas[:2] if PAL.is_palladius_token(lm, 2) and _SM.is_name_lemma(lm)] + if palla and pallb and set(palla) & set(pallb): + if cooccur_same_sentence(chunks, a.src, b.src): + weak.append(dict(a=a.src, b=b.src, rule="R3-blockedR4iv", kind="shared_render_copresent")) + else: + ident.append(dict(a=a.src, b=b.src, rule="R3", kind="shared_render", render=list(set(palla) & set(pallb)))) + return ident, family, weak + + +def cluster(ident_edges, all_surfaces): + """Union-find over identity edges -> entity clusters.""" + parent = {s: s for s in all_surfaces} + def find(x): + while parent[x] != x: + parent[x] = parent[parent[x]]; x = parent[x] + return x + for e in ident_edges: + if e["a"] in parent and e["b"] in parent: + parent[find(e["a"])] = find(e["b"]) + clusters = defaultdict(list) + for s in all_surfaces: + clusters[find(s)].append(s) + return [sorted(v) for v in clusters.values() if len(v) > 1] + + +if __name__ == "__main__": + chunks = X.load_chunks(); gt = X.load_gt() + sm = SP.SpreadModel(chunks) + gt_by_norm = {X.norm(s): e for e in gt for s in e.surfaces} + # surface set: GT surfaces (measure precision vs the 2 known aliases) + high-conf name candidates + import arms as A + ar = A.Arms(chunks, X.Contrast()) + a3 = ar.arm_A3() + # input = GT surfaces + genuine name/place/title candidates (freq>=5, typed, NOT boundary fragments). + # Exclude GT-surface+trailing-particle artifacts (方源在/方源心 from the surname window). + subsumed = ar.va.subsumed + gt_surf_norm = {X.norm(s) for e in gt for s in e.surfaces} + PARTICLE = set("的了在是和就也都不心面前后上下今少多大小和与之其") + def frag(src): + return len(src) >= 2 and src[:-1] in gt_surf_norm and src[-1] in PARTICLE + name_cands = [c.src for c in a3[:200] + if any(t in c.types for t in ("name", "place", "title")) + and c.freq >= 5 and c.src not in subsumed and len(c.src) >= 2 and not frag(c.src)] + gt_norm_surfaces = frozenset(X.norm(s) for e in gt for s in e.surfaces) + surface_srcs = list(gt_norm_surfaces | set(name_cands)) + surfaces = build_surfaces(surface_srcs, sm, gt_by_norm) + ident, family, weak = propose_edges(surfaces, chunks, sm, gt_norm_surfaces) + clusters = cluster(ident, list(surfaces)) + print(f"surfaces: {len(surfaces)} identity-edges: {len(ident)} family-edges: {len(family)} weak: {len(weak)}") + print("\n== identity edges (R1/R3) ==") + for e in ident: + print(f" {e['a']} = {e['b']} [{e['rule']}/{e['kind']}] {e.get('render','')}") + print("\n== family superclusters (R2, NOT identity) ==") + for e in family[:20]: + print(f" {e['a']} ~ {e['b']} [{e['rule']}/{e['kind']}]") + print(f"\n== identity clusters (>1) ==") + for cl in clusters: + print(f" {{ {', '.join(cl)} }}") + # precision vs known seed aliases (古月方源=方源, 古月方正=方正) + known = {frozenset([X.norm("古月方源"), X.norm("方源")]), frozenset([X.norm("古月方正"), X.norm("方正")])} + id_pairs = {frozenset([e["a"], e["b"]]) for e in ident if e["kind"] in ("extension", "shared_render")} + tp = len(id_pairs & known) + print(f"\nknown-alias recall: {tp}/{len(known)}; identity-edge count {len(id_pairs)} " + f"(precision vs seed not meaningful — seed alias set is thin, needs owner mini-gold §E-1)") diff --git a/eval/exp16/arms.py b/eval/exp16/arms.py new file mode 100644 index 0000000..1ba4c67 --- /dev/null +++ b/eval/exp16/arms.py @@ -0,0 +1,215 @@ +#!/usr/bin/env python3 +"""exp16 — arm assembly A1/A2/A3/A3-abl + §D3 metrics (research/20 §D2-D4). miner-v1. $0. + + A1 = V-A freq × contrast × c-value + A2 = V-B = A1 + spread translation-spread boost + dst-variants (confounded lower bound on records.json) + A3 = V-C = A2 + patterns surname/title/topo/formant/rank-grade + Palladius; PROPOSES sub-floor terms + A3-abl = V-C with λ=0 patterns WITHOUT the spread signal (isolates each signal's contribution) + +Metric structure (§D3, §D4-а — signals reported SEPARATELY, no manufactured convergence): + • recall @PROPOSED (candidate-set membership) — the BLINDNESS axis: "where is each arm blind?" + A1/A2 share a candidate set (spread only re-ranks); A3/A3-abl share the pattern-extended set. + This is where V-A (freq-floor-blind on f<3) and V-C (patterns close f<3) differ. + • recall@top-K + pseudo-precision@K — the RANKING/precision axis + trade-off curve. + • catastrophe screen (top-50 of the WINNER), threshold ±50% sensitivity. + • spread's contribution (A2 vs A1, A3 vs A3-abl) is ISOLATED — on injected drafts it is ~0/negative + (confound §D1); measured ecologically only on the cold-start slice. + +Thresholds chosen on TUNING half (ch1-15), frozen (FROZEN), TEST half (ch16-25) reported (§D1). +""" +from __future__ import annotations + +import json +from dataclasses import dataclass, field + +import exp16_common as X +import detectors as D +import patterns as P +import spread as SP +import palladius as PAL + +# ── frozen miner-v1 config (thresholds tuned on ch1-15; pinned in the freeze commit) ────────────── +FROZEN = dict( + freq_floor=X.FREQ_FLOOR, # 3 + subsume_alpha=D.SUBSUME_ALPHA, # 0.80 + ngram_max=X.NGRAM_MAX, # 6 + lam=0.0, # V-B spread coefficient — 0 on injected drafts (confound §D1); >0 on cold-start + formant_min_partners=3, + formant_min_over_rep=15.0, + bonus=dict(name=140.0, place=140.0, title=120.0, term=80.0), + pattern_source_weight=dict(surname=1.0, ordinal_title=1.0, title_suffix=0.9, topo_suffix=0.9, + rank_grade=1.0, formant_suffix=0.6, formant_prefix=0.5, + title_bare=0.4, palladius=0.8), + subfloor_freq_scale=40.0, + spread_freq_min=5, # only compute spread for candidates with freq>=this (cheap; rare ill-defined) + top_k=90, # operating point for recall@K/precision reporting +) + + +@dataclass +class ScoredCand: + src: str + score: float + freq: int + types: list = field(default_factory=list) + evidence: list = field(default_factory=list) + dst_variants: list = field(default_factory=list) + spread: float = 0.0 + from_pattern: bool = False + + +class Arms: + def __init__(self, chunks, contrast, cfg=FROZEN, use_spread=True): + self.chunks = chunks + self.C = contrast + self.cfg = cfg + self.va = D.VA(chunks, contrast, freq_floor=cfg["freq_floor"], + subsume_alpha=cfg["subsume_alpha"]).build() + self.va_ranked = self.va.score_all() # A1 + self.pats, self.formants = P.pattern_candidates( + chunks, self.va.cand_freq, contrast, + min_partners=cfg["formant_min_partners"], min_over_rep=cfg["formant_min_over_rep"]) + self.sm = SP.SpreadModel(chunks) if use_spread else None + + def _spread_of(self, src, freq): + if self.sm and freq >= self.cfg["spread_freq_min"]: + return self.sm.spread(src) + return 0.0 + + def _dst_of(self, src, freq): + if self.sm and freq >= self.cfg["spread_freq_min"]: + return self.sm.dst_variants(src)[0] + return [] + + def arm_A1(self): + return list(self.va_ranked) + + def arm_A2(self, lam=None): + lam = self.cfg["lam"] if lam is None else lam + out = [] + for c in self.va_ranked: + sp = self._spread_of(c.src, c.freq) + out.append(ScoredCand(src=c.src, score=c.score * (1 + lam * sp), freq=c.freq, spread=sp, + dst_variants=self._dst_of(c.src, c.freq))) + out.sort(key=lambda s: (-s.score, -s.freq, s.src)) + return out + + def arm_A3(self, lam=None): + lam = self.cfg["lam"] if lam is None else lam + cfg = self.cfg + merged: dict[str, ScoredCand] = {} + for c in self.va_ranked: + sp = self._spread_of(c.src, c.freq) if lam else 0.0 + merged[c.src] = ScoredCand(src=c.src, score=c.score * (1 + lam * sp), freq=c.freq, spread=sp, + dst_variants=self._dst_of(c.src, c.freq)) + for cand, info in self.pats.items(): + pw = max((cfg["pattern_source_weight"].get(ev.split(":")[0], 0.3) for ev in info["evidence"]), + default=0.3) + typ = info["types"][0] if info["types"] else "term" + bonus = cfg["bonus"].get(typ, cfg["bonus"]["term"]) * pw + if cand in merged: + merged[cand].score += bonus + for t in info["types"]: + if t not in merged[cand].types: + merged[cand].types.append(t) + merged[cand].evidence += info["evidence"][:3] + merged[cand].from_pattern = True + else: + f = X.count_occurrences(cand, self.chunks) + merged[cand] = ScoredCand(src=cand, score=bonus + cfg["subfloor_freq_scale"] * f * pw, + freq=f, types=list(info["types"]), evidence=info["evidence"][:3], + dst_variants=self._dst_of(cand, f), from_pattern=True) + # Palladius ru-side confirmation (gated on capitalized-predominant name lemmas — blocks the + # 'найти'=най+ти class of false positives) + for sc in merged.values(): + for lm, d, _ in sc.dst_variants[:2]: + if PAL.is_palladius_token(lm, min_syllables=2) and (self.sm and self.sm.is_name_lemma(lm)): + sc.score += cfg["bonus"]["name"] * cfg["pattern_source_weight"]["palladius"] * 0.5 + if "name" not in sc.types: + sc.types.append("name") + sc.evidence.append(f"palladius:{lm}") + break + out = list(merged.values()) + out.sort(key=lambda s: (-s.score, -s.freq, s.src)) + return out + + def arm_A3_abl(self): + return self.arm_A3(lam=0.0) + + +# ── metrics ─────────────────────────────────────────────────────────────────────────────────────── +def candidate_set(ranked): + return {c.src for c in ranked} + + +def recall_by(ranked, gt, chunks, top_k=None, include_annotation=True): + """recall by stratum & type. top_k=None => @PROPOSED (candidate-set membership = blindness axis).""" + if top_k is None: + cset = candidate_set(ranked) + def hit(e): + return any(s in cset for s in e.norm_surfaces) + else: + idx = {c.src: i for i, c in enumerate(ranked)} + def hit(e): + return any((idx.get(s) is not None and idx[s] < top_k) for s in e.norm_surfaces) + buckets = {k: [] for k in ("overall", "f>=10", "f3-9", "f<3", + "name", "title", "place", "term", "nickname")} + misses = {} + for e in gt: + f = X.gt_occurrences(e, chunks, include_annotation) + h = hit(e) + buckets["overall"].append(h) + buckets[X.freq_stratum(f)].append(h) + if e.typ in buckets: + buckets[e.typ].append(h) + if not h: + misses.setdefault("overall", []).append(f"{e.src}({X.freq_stratum(f)})") + return {k: (sum(v) / len(v) if v else None, len(v)) for k, v in buckets.items()}, misses + + +def pseudo_precision(ranked, gt, top_k): + gtsurf = {s for e in gt for s in e.norm_surfaces} + top = ranked[:top_k] + return (sum(1 for c in top if c.src in gtsurf) / len(top)) if top else 0.0 + + +def catastrophe_screen(ranked, top_k=50): + idx = {c.src: i for i, c in enumerate(ranked)} + res = {s: idx.get(X.norm(s)) for s in X.CATASTROPHE} + return all(r is not None and r < top_k for r in res.values()), res + + +def gt_in_chapters(gt, chunks, chapters, include_annotation=True): + sub = [c for c in chunks if c.chapter in chapters] + return [e for e in gt if X.gt_occurrences(e, sub, include_annotation) >= 1] + + +def summarize(name, ranked, gt, chunks, K): + rp, miss_p = recall_by(ranked, gt, chunks, top_k=None) # @proposed + rk, _ = recall_by(ranked, gt, chunks, top_k=K) # @top-K + ok, res = catastrophe_screen(ranked, 50) + pp = pseudo_precision(ranked, gt, K) + return dict(name=name, n=len(ranked), catastrophe_ok=ok, catastrophe=res, pseudo_prec_at_K=round(pp, 3), + recall_proposed={k: (round(v[0], 3) if v[0] is not None else None, v[1]) for k, v in rp.items()}, + recall_at_K={k: (round(v[0], 3) if v[0] is not None else None, v[1]) for k, v in rk.items()}, + misses_proposed=miss_p.get("overall", [])) + + +if __name__ == "__main__": + chunks = X.load_chunks() + gt = X.load_gt() + C = X.Contrast() + arms = Arms(chunks, C) + A = {"A1": arms.arm_A1(), "A2": arms.arm_A2(lam=8.0), "A3": arms.arm_A3(lam=0.0), + "A3-abl": arms.arm_A3_abl()} + K = FROZEN["top_k"] + print(f"=== exp16 arms — FULL slice (57 chunks); K={K} ===") + print(f"detected formants: {sorted(arms.formants, key=lambda c: -arms.formants[c]['over_rep'])}\n") + for name, ranked in A.items(): + s = summarize(name, ranked, gt, chunks, K) + rp, rk = s["recall_proposed"], s["recall_at_K"] + print(f"{name}: n={s['n']} catastrophe={'PASS' if s['catastrophe_ok'] else 'FAIL'} {s['catastrophe']}") + print(f" recall@PROPOSED: overall={rp['overall'][0]} f>=10={rp['f>=10'][0]} f3-9={rp['f3-9'][0]} f<3={rp['f<3'][0]}") + print(f" recall@top{K}: overall={rk['overall'][0]} f>=10={rk['f>=10'][0]} f3-9={rk['f3-9'][0]} f<3={rk['f<3'][0]} pseudo-prec={s['pseudo_prec_at_K']}") + print(f" by type @proposed: " + " ".join(f"{t}={rp[t][0]}({rp[t][1]})" for t in ("name","title","place","term","nickname") if rp[t][0] is not None)) + print() diff --git a/eval/exp16/banknote.py b/eval/exp16/banknote.py new file mode 100644 index 0000000..a428a12 --- /dev/null +++ b/eval/exp16/banknote.py @@ -0,0 +1,89 @@ +#!/usr/bin/env python3 +"""exp16 — banknote-v1 in-band footnote channel: format, instruction, parser (research/20 §B3). miner-v1. + +The translator, after the translation, may emit a versioned separator + tab-delimited term lines for +NEW terms only (not in the injected glossary block). Code slices the block off BEFORE any gate/editor +(here: the polygon reads the parsed artifact; the editor never sees the block). Parser is tolerant of a +truncated final line (flags banknote_truncated) and flags malformed residual (banknote_parse_fail). + +Separator ⟦TM-BANK-v1⟧ is chosen to (a) not occur in natural text, (b) NOT be the words +'Примечание/Сноска/Комментарий' which the backend sanitizer's trailing-note class reserves (§B3-1). +""" +from __future__ import annotations + +import re + +SEP = "⟦TM-BANK-v1⟧" +MAX_LINES = 12 # max banknote lines budgeted per chunk (§B3-5 max_tokens sizing) + +# instruction appended to the translator system prompt for the banknote (config-b) arm. +BANKNOTE_INSTRUCTION = ( + "\n\nПОСЛЕ полного перевода, если в этом фрагменте встретились НОВЫЕ имена собственные, " + "топонимы или уникальные термины мира книги (которых нет в уже данном тебе глоссарии, если он " + "был), добавь В САМОМ КОНЦЕ ответа ОТДЕЛЬНЫЙ блок ровно в таком формате:\n" + f"{SEP}\n" + "исходный_терминтвой_переводтип\n" + "где тип — одно из: name (имя), place (топоним), title (титул/ранг), term (термин). Каждый термин " + "на своей строке, поля разделены символом табуляции. Не более " + str(MAX_LINES) + " строк, только " + "самые важные новые термины. Если новых терминов нет — НЕ добавляй ни блок, ни разделитель. " + "Сам перевод не должен содержать этот блок внутри себя — только в самом конце." +) + +_TYPE_OK = {"name", "place", "title", "term", "nickname"} + + +def split_banknote(model_output: str): + """Slice the banknote block off the raw model output. Returns (clean_translation, raw_block_or_'').""" + idx = model_output.find(SEP) + if idx < 0: + return model_output.rstrip(), "" + clean = model_output[:idx].rstrip() + block = model_output[idx + len(SEP):].strip("\n") + return clean, block + + +def parse_banknote(block: str, truncated_generation: bool = False): + """Parse the tab-delimited block. Returns (entries, flags). + entries: [{'src','dst','type'}]. flags: {'n_lines','parse_fail','truncated'}.""" + entries, bad = [], 0 + flags = {"n_banknote_lines": 0, "banknote_parse_fail": False, "banknote_truncated": False} + if not block.strip(): + return entries, flags + lines = [ln for ln in block.split("\n") if ln.strip()] + for i, ln in enumerate(lines): + # tolerate the LAST line being truncated by generation length (§B3-5) + parts = re.split(r"\t| {2,}|\s*\|\s*", ln.strip()) + parts = [p.strip() for p in parts if p.strip()] + if len(parts) < 2: + if i == len(lines) - 1 and truncated_generation: + flags["banknote_truncated"] = True + continue + bad += 1 + continue + src, dst = parts[0], parts[1] + typ = parts[2].lower() if len(parts) >= 3 else "term" + if typ not in _TYPE_OK: + typ = "term" + # src should contain a Han char (this is a zh->ru channel); else malformed + if not re.search(r"[㐀-鿿]", src): + bad += 1 + continue + entries.append({"src": src, "dst": dst, "type": typ}) + flags["n_banknote_lines"] = len(entries) + if bad > 0: + flags["banknote_parse_fail"] = True + return entries, flags + + +if __name__ == "__main__": + demo = ("Перевод фрагмента... последнее предложение.\n\n" + f"{SEP}\n方源\tФан Юань\tname\n蛊师\tгу-мастер\ttitle\n青茅山\tгора Цинмао\tplace") + clean, block = split_banknote(demo) + ents, flags = parse_banknote(block) + print("clean tail:", repr(clean[-40:])) + print("entries:", ents) + print("flags:", flags) + # truncated last line + demo2 = f"текст\n{SEP}\n方源\tФан Юань\tname\n蛊" + _, b2 = split_banknote(demo2) + print("\ntruncated:", parse_banknote(b2, truncated_generation=True)) diff --git a/eval/exp16/canon.py b/eval/exp16/canon.py new file mode 100644 index 0000000..d86fe17 --- /dev/null +++ b/eval/exp16/canon.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""exp16 — canon consolidation §C2 + canon-recovery metric §D3-3. miner-v1. $0. + +Canon by ALL occurrences (NOT first-wins, anti-LTCR): the winning dst for a term is the rendering +scored by frequency-across-all-chunks × Palladius-conformity (type name/place) × case-lemma coverage — +the VARIANT wins, not the first occurrence. Majority NEVER yields approved (canon proposal = draft). + +canon-recovery(ent): would the §C2 procedure, run over the drafts, propose a dst whose content lemmas +COVER the owner-signed seed dst (memnorm + lemma level)? On records.json (injected) this is trivialized +(spread≈0 — the drafts already carry the signed forms; §D1 confound). The ECOLOGICAL measurement is on +the cold-start drafts (no injection): does canon recover the signed dst from an un-anchored draft? +""" +from __future__ import annotations + +import regex + +import exp16_common as X +import spread as SP +import palladius as PAL + +_COMMON_HEAD = {"гора", "деревня", "село", "селение", "стан", "клан", "род", "море", "первобытный", + "камень", "истинный", "ци", "апертура"} # generic heads still counted, but not name-distinctive + + +def signed_content_lemmas(dst: str) -> list[str]: + """Distinctive lemmas of the signed dst (drop 2-char function words; keep name + key term words).""" + lem = SP.lemmatize_ru(dst) + return [l for l in lem if len(l) >= 3] + + +def canon_propose(sm: SP.SpreadModel, ent: X.GTEntity, top=6): + """Reconstruct the canon dst proposal from the drafts by ALL occurrences. + For name/place: ordered Palladius name lemmas among the top associates. For term/title: the top + content associates. Returns (proposed_lemmas, evidence).""" + variants, cidx = sm.dst_variants(X.norm(ent.src), top=top) + lemmas = [lm for lm, d, _ in variants] + if ent.typ in ("name", "place", "nickname"): + # canon = Palladius-conformant capitalized-name lemmas (the transliteration), by Dice rank + proposed = [lm for lm in lemmas if PAL.is_palladius_token(lm, 1) and sm.is_name_lemma(lm)] + # for places, prepend the generic head if strongly associated (гора/деревня) + head = [lm for lm in lemmas if lm in ("гора", "деревня")] + proposed = head[:1] + proposed if ent.typ == "place" else proposed + else: + proposed = [lm for lm in lemmas if len(lm) >= 3][:4] + return proposed, variants + + +def canon_recovery(sm: SP.SpreadModel, ent: X.GTEntity, cover_frac=0.5): + """Recovered if >= cover_frac of the signed distinctive lemmas appear among the proposed lemmas.""" + signed = set(signed_content_lemmas(ent.dst)) + if not signed: + return None, [], [] # nothing to recover (e.g. empty dst) + proposed, variants = canon_propose(sm, ent) + proposed_set = set(proposed) + # also allow coverage by ANY top associate lemma (the content word may not be a name token) + assoc = {lm for lm, d, _ in variants} + covered = {s for s in signed if s in proposed_set or s in assoc} + rec = len(covered) / len(signed) >= cover_frac + return rec, sorted(signed), proposed + + +def run(chunks, gt, label=""): + sm = SP.SpreadModel(chunks) + rows = [] + rec_n = tot = 0 + for e in gt: + if not e.dst: + continue + occ = X.gt_occurrences(e, chunks) + if occ == 0: + continue + rec, signed, proposed = canon_recovery(sm, e) + if rec is None: + continue + tot += 1 + rec_n += bool(rec) + rows.append(dict(src=e.src, dst=e.dst, typ=e.typ, occ=occ, recovered=bool(rec), + signed=signed, proposed=proposed)) + return rows, (rec_n / tot if tot else None), tot + + +if __name__ == "__main__": + chunks = X.load_chunks(); gt = X.load_gt() + rows, rate, tot = run(chunks, gt, "records.json (INJECTED — confounded upper bound, §D1)") + print(f"canon-recovery on records.json (INJECTED drafts — confound §D1 trivializes this): " + f"{sum(r['recovered'] for r in rows)}/{tot} = {rate:.3f}\n") + print("misses (signed vs proposed):") + for r in rows: + if not r["recovered"]: + print(f" {r['src']:<6}({r['typ']:<8}) signed={r['signed']} proposed={r['proposed']}") + print("\nsample recovered:") + for r in [x for x in rows if x["recovered"]][:12]: + print(f" {r['src']:<6}({r['typ']:<8}) signed={r['signed']} -> proposed={r['proposed']}") diff --git a/eval/exp16/coldstart_gen.py b/eval/exp16/coldstart_gen.py new file mode 100644 index 0000000..c67c988 --- /dev/null +++ b/eval/exp16/coldstart_gen.py @@ -0,0 +1,151 @@ +#!/usr/bin/env python3 +"""exp16 — cold-start / A4 generation (research/20 §D1 cold-start slice + §D2 A4). PAID (deepseek-v4-flash). + +FROZEN in the pre-reg commit; RUN later in the DeepSeek off-peak valley (D39.7: peaks UTC 01-04 & 06-10 +cost ×2 — this script REFUSES to run in a peak window unless --force). Generates, for the pre-registered +cold-start chapters [1,4,5,7,9] (14 chunks), fresh flash drafts WITHOUT glossary injection, in two configs: + plain — cold-start baseline (translator.md brief-filled, NO glossary block) -> spread + canon-recovery + banknote — plain + the banknote-v1 instruction (§B3) -> footnote channel + quality delta + +Discipline: thinking ON (deepseek echo-mine guardrail — never disabled). LENGTH-GATE FIX (D39.9): a +finish_reason=='length' response is REJECTED/flagged, NOT accepted as valid (exp14 harness bug not +inherited). Every call persists raw usage+cost (D30.10) via the exp15 Spender; per-call gate + $10 hard cap. +""" +from __future__ import annotations + +import argparse +import json +import sys +import time +from pathlib import Path + +HERE = Path(__file__).resolve().parent +sys.path.append(str(HERE.parent / "exp15")) +import exp15_llm as LLM +import banknote as BN + +BOOK = Path("/home/ubuntu/books/gu-zhenren") +RECORDS = BOOK / "rerun" / "records.json" +OUT = BOOK / "exp16" / "coldstart" +LEDGER = BOOK / "exp16" / "coldstart_costs.jsonl" + +COLD_CHAPTERS = [1, 4, 5, 7, 9] # FROZEN (greedy GT-cover: 52/57 entities, 14 chunks) +MODEL = "deepseek-v4-flash" +HARD_CAP = 10.0 # exp16 paid budget (owner 17-18.07), separate from exp15 +PER_CALL_CAP = 0.05 # flash is cheap; a single chunk far below this + +# ── FROZEN translator prompt: backend/prompts/translator.md (sha256 35d3ad6e...), brief from rerun/book.yaml, +# glossary-injection block OMITTED (cold-start = new book, no bank). ────────────────────────────── +BRIEF = dict(source_lang="zh", target_lang="ru", genre="вебновелла", audience="взрослые читатели вебновелл", + title="蛊真人", honorifics="keep", transcription="palladius", venuti="0.6", footnotes="minimal") + +TRANSLATOR_SYSTEM = ( + 'Ты — профессиональный литературный переводчик с языка «{source_lang}» на язык «{target_lang}».\n' + 'Жанр книги: {genre}. Аудитория: {audience}. Книга: «{title}».\n\n' + 'Правила перевода (translation brief):\n' + '- Хонорифики: {honorifics}. Транскрипция имён и реалий: {transcription}.\n' + '- Баланс форенизация/доместикация (0 — полная адаптация, 1 — сохранение чужого): {venuti}.\n' + '- Сноски: {footnotes}.\n' + '- Переводи ВСЕ предложения исходника: ничего не пропускай, не сокращай и не добавляй от себя.\n' + '- Вёрстка — по нормам русского языка, НЕ копируй построчную разбивку исходника: реплики прямой речи ' + 'оформляй с нового абзаца через тире «—».\n' + '- Пиши живым литературным русским языком; избегай канцелярита и калек с исходного языка.\n\n' + 'Выведи ТОЛЬКО перевод, без служебных преамбул, комментариев, пояснений и markdown-заголовков.' +).format(**BRIEF) + +TRANSLATOR_USER = "Переведи следующий фрагмент:\n\n{text}" +TRANSLATOR_MD_SHA = "35d3ad6e0656a7d1116fb8b034a4cea087d5c11ac6e043804185f94f15af5829" + + +def in_deepseek_peak() -> bool: + """UTC 01:00-04:00 and 06:00-10:00 are the ×2 peak windows (D39.7). new/argless time is banned in + workflow scripts but this is a plain script; use time.gmtime().""" + h = time.gmtime().tm_hour + return (1 <= h < 4) or (6 <= h < 10) + + +def load_cold_chunks(): + recs = json.load(open(RECORDS, encoding="utf-8")) + out = [r for r in recs if r["chapter"] in COLD_CHAPTERS] + out.sort(key=lambda r: (r["chapter"], r["chunk_idx"])) + return out + + +def gen_one(sp: LLM.Spender, source: str, config: str): + system = TRANSLATOR_SYSTEM + (BN.BANKNOTE_INSTRUCTION if config == "banknote" else "") + user = TRANSLATOR_USER.format(text=source) + msgs = [{"role": "system", "content": system}, {"role": "user", "content": user}] + in_tok = len(source) + len(system) # rough; gate uses conservative estimate below + ok, pred, why = sp.gate(MODEL, in_tok, LLM.MODELS[MODEL]["max_tokens"]) + if not ok: + return {"err": f"gate: {why}", "pred": pred} + r = LLM.call(MODEL, msgs, max_tokens=LLM.MODELS[MODEL]["max_tokens"]) + return r + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--config", choices=["plain", "banknote", "both"], default="both") + ap.add_argument("--force", action="store_true", help="override the DeepSeek peak-window guard") + ap.add_argument("--dry", action="store_true") + a = ap.parse_args() + + if in_deepseek_peak() and not a.force: + print(f"REFUSING: UTC hour {time.gmtime().tm_hour} is in a DeepSeek ×2 peak window (01-04/06-10). " + f"Run in the valley or pass --force. (D39.7)") + return + + chunks = load_cold_chunks() + print(f"cold-start chunks: {len(chunks)} (chapters {COLD_CHAPTERS})") + OUT.mkdir(parents=True, exist_ok=True) + sp = LLM.Spender(str(LEDGER), per_call_cap=PER_CALL_CAP, hard_cap=HARD_CAP) + print(f"spender: per_call_cap=${PER_CALL_CAP} hard_cap=${HARD_CAP} already_spent=${sp.total:.4f}") + + configs = ["plain", "banknote"] if a.config == "both" else [a.config] + stats = {c: {"ok": 0, "length_reject": 0, "err": 0, "banknote_lines": 0, "parse_fail": 0, "truncated": 0} + for c in configs} + for config in configs: + cdir = OUT / config + cdir.mkdir(parents=True, exist_ok=True) + for r in chunks: + cid = f"{r['chapter']}.{r['chunk_idx']}" + raw_fp = cdir / f"{cid}.raw.txt" + if raw_fp.exists(): + continue + if a.dry: + print(f" [dry] {config} {cid}") + continue + res = gen_one(sp, r["source"], config) + rec = {"config": config, "id": cid, "model": MODEL, "usage": res.get("usage", {}), + "cost": res.get("cost", 0.0), "finish": res.get("finish"), "err": res.get("err"), + "latency_s": res.get("latency_s")} + # LENGTH-GATE FIX (D39.9): finish=length -> reject, do NOT accept truncated output + if res.get("err"): + stats[config]["err"] += 1 + sp.record(rec); print(f" ERR {config} {cid}: {res['err'][:70]}"); continue + if res.get("finish") == "length": + stats[config]["length_reject"] += 1 + rec["rejected"] = "finish=length" + sp.record(rec); print(f" LENGTH-REJECT {config} {cid} (comp={res['usage'].get('completion_tokens')})") + continue + sp.record(rec) + text = res["text"] + raw_fp.write_text(text, encoding="utf-8") + if config == "banknote": + clean, block = BN.split_banknote(text) + ents, flags = BN.parse_banknote(block, truncated_generation=False) + (cdir / f"{cid}.clean.txt").write_text(clean, encoding="utf-8") + json.dump({"entries": ents, "flags": flags}, + open(cdir / f"{cid}.banknote.json", "w"), ensure_ascii=False, indent=1) + stats[config]["banknote_lines"] += flags["n_banknote_lines"] + stats[config]["parse_fail"] += flags["banknote_parse_fail"] + stats[config]["truncated"] += flags["banknote_truncated"] + stats[config]["ok"] += 1 + print(f" ok {config} {cid}: {len(text)}ch finish={res['finish']} ${res['cost']:.5f}") + + print(f"\n=== stats: {json.dumps(stats, ensure_ascii=False)} ===") + print(f"=== total spend this run: ${sp.total:.4f} / cap ${HARD_CAP} ===") + + +if __name__ == "__main__": + main() diff --git a/eval/exp16/detectors.py b/eval/exp16/detectors.py new file mode 100644 index 0000000..20de957 --- /dev/null +++ b/eval/exp16/detectors.py @@ -0,0 +1,218 @@ +#!/usr/bin/env python3 +"""exp16 — deterministic detectors V-A / V-B / V-C (research/20 §B1). miner-v1. $0. + +Common substrate: Han char n-grams 1–6 (exp16_common), freq floor 3, memnorm normalization, c-value +nested discount with g(L)=log2(L+1) (avoids the |a|=1 degeneracy that would zero out 蛊/转 — §B1), +GT occurrence counting via AC-equivalent WITHOUT suppressContained. + + V-A freq × contrast(weirdness, general-zh) × c-value(nested discount). Output: ranked src candidates. + V-B V-A + translation-spread signal from the ru drafts (Dice/LTCR-style; pymorphy3 lemmas). +dst variants. + V-C V-B + NER patterns / structural morphology / Palladius ru-side channel (patterns.py, palladius.py). + +Thresholds (floor, top-K, λ, m) are frozen after tuning on chapters 1–15; only chapters 16–25 are the +reported test half (§D1 anti-overfit). This module exposes the scorer; tuning/arms live in arms.py. +""" +from __future__ import annotations + +import math +from collections import defaultdict +from dataclasses import dataclass, field + +import exp16_common as X + + +# ── c-value (Frantzi et al. 2000, §A1) with §B1 length-multiplier fix ────────────────────────────── +def length_mult(L: int) -> float: + """g(L) = log2(L+1). §B1: NOT log2(L) — that zeros |a|=1 (蛊/转) and breaks the catastrophe screen.""" + return math.log2(L + 1) + + +def compute_cvalue(cand_freq: dict[str, int]) -> dict[str, float]: + """C-value with nested discount. For a nested in longer candidates T_a: + cval(a) = g(|a|)*(f(a) - (1/|T_a|)*Σ_{b∈T_a} f(b)); non-nested: g(|a|)*f(a).""" + cands = set(cand_freq) + tcount = defaultdict(int) # |T_a| + tfreqsum = defaultdict(int) # Σ f(b) + for b, fb in cand_freq.items(): + Lb = len(b) + if Lb < 2: + continue + subs = set() + for n in range(1, Lb): + for i in range(0, Lb - n + 1): + a = b[i:i + n] + if a in cands and a != b: + subs.add(a) + for a in subs: + tcount[a] += 1 + tfreqsum[a] += fb + cval = {} + for a, fa in cand_freq.items(): + g = length_mult(len(a)) + if tcount[a] > 0: + cval[a] = g * (fa - tfreqsum[a] / tcount[a]) + else: + cval[a] = g * fa + return cval + + +# ── V-A ─────────────────────────────────────────────────────────────────────────────────────────── +@dataclass +class Candidate: + src: str # normalized n-gram + freq: int + cvalue: float + weirdness: float + termhood: float + score: float + length: int + chapters: set = field(default_factory=set) + # V-B/V-C enrichments + dst_variants: dict = field(default_factory=dict) # lemma -> count + spread: float = 0.0 + pattern_types: list = field(default_factory=list) # name/title/place/term hints + pattern_evidence: list = field(default_factory=list) + + +SUBSUME_ALPHA = 0.80 # drop candidate a if a longer container b has f(b) >= alpha*f(a): a is not independent + +def subsumed_candidates(cand_freq: dict[str, int], alpha=SUBSUME_ALPHA) -> set[str]: + """Substring subsumption (nested consolidation for ranking): a candidate `a` that is (almost) + always part of a longer candidate `b` (f(b) >= alpha*f(a)) is a boundary fragment, not an + independent term — drop it. 方源(559) survives 方源的(85) [ratio 0.15]; 花酒行(65) is dropped by + 花酒行者(65) [ratio 1.0]; 光蛊(96) dropped by 月光蛊(96). Occurrence counting for GT is unaffected + (that path is suppress-free, D24.2) — this only cleans the candidate RANKING.""" + cands = set(cand_freq) + drop = set() + # index candidates by length for efficient container lookup + bylen = defaultdict(list) + for c in cands: + bylen[len(c)].append(c) + for a, fa in cand_freq.items(): + La = len(a) + # any longer candidate b that strictly contains a with f(b) >= alpha*f(a) subsumes a + found = False + for Lb in range(La + 1, X.NGRAM_MAX + 1): + for b in bylen[Lb]: + if a in b and cand_freq[b] >= alpha * fa: + found = True + break + if found: + break + if found: + drop.add(a) + return drop + + +class VA: + """Frequency × contrast × c-value. Params tunable; frozen after tuning half.""" + + def __init__(self, chunks, contrast: X.Contrast, freq_floor=X.FREQ_FLOOR, + weird_floor=1e-9, termhood_w=1.0, subsume_alpha=SUBSUME_ALPHA): + self.chunks = chunks + self.C = contrast + self.freq_floor = freq_floor + self.weird_floor = weird_floor + self.termhood_w = termhood_w + self.subsume_alpha = subsume_alpha + self.cand_freq = {} + self.N_book = 0 + + def build(self): + allc = X.enumerate_candidates(self.chunks) + self.cand_freq = {k: v for k, v in allc.items() if v >= self.freq_floor} + self.N_book = sum(allc.values()) # total n-gram token mass (all lengths) — weirdness denominator scale + self.subsumed = subsumed_candidates(self.cand_freq, self.subsume_alpha) + self.cval = compute_cvalue(self.cand_freq) + return self + + def weirdness(self, a: str, f: int) -> float: + """P_book(a) / P_general(a). P_general = max(word_rel, char_indep) smoothed (§A1 OOV smoothing).""" + p_book = f / self.N_book + gen = max(self.C.word_rel(a), self.C.char_indep_rel(a), self.weird_floor) + return p_book / gen + + def score_all(self, drop_subsumed=True) -> list[Candidate]: + out = [] + for a, f in self.cand_freq.items(): + if drop_subsumed and a in self.subsumed: + continue + w = self.weirdness(a, f) + termhood = math.log1p(w) + cv = self.cval[a] + # score: unithood (c-value) gated by termhood (domain specificity). Negative c-value + # (over-nested substring) floored at 0 — it is dominated by its container, not a term itself. + score = max(cv, 0.0) * (termhood ** self.termhood_w) + out.append(Candidate(src=a, freq=f, cvalue=cv, weirdness=w, termhood=termhood, + score=score, length=len(a))) + out.sort(key=lambda c: (-c.score, -c.freq, c.src)) + return out + + +# ── recall harness ──────────────────────────────────────────────────────────────────────────────── +def rank_index(ranked: list[Candidate]) -> dict[str, int]: + return {c.src: i for i, c in enumerate(ranked)} + + +def gt_recall(ranked: list[Candidate], gt: list[X.GTEntity], chunks, top_k=None, + stratum_filter=None, include_annotation=True): + """Recall = fraction of GT entities whose src OR any alias surface appears in ranked[:top_k]. + Returns (recall, hit_entities, miss_entities, per_entity_rank).""" + idx = rank_index(ranked) + limit = top_k if top_k is not None else len(ranked) + hits, misses, per = [], [], {} + considered = 0 + for e in gt: + f = X.gt_occurrences(e, chunks, include_annotation) + if stratum_filter and X.freq_stratum(f) not in stratum_filter: + continue + considered += 1 + best = None + for surf in e.norm_surfaces: + r = idx.get(surf) + if r is not None and r < limit and (best is None or r < best): + best = r + per[e.src] = best + if best is not None: + hits.append(e) + else: + misses.append(e) + recall = len(hits) / considered if considered else 0.0 + return recall, hits, misses, per + + +def catastrophe_ok(ranked: list[Candidate], top_k=50): + idx = rank_index(ranked) + res = {} + for s in X.CATASTROPHE: + r = idx.get(X.norm(s)) + res[s] = r + ok = all(r is not None and r < top_k for r in res.values()) + return ok, res + + +if __name__ == "__main__": + import sys + chunks = X.load_chunks() + gt = X.load_gt() + C = X.Contrast() + va = VA(chunks, C).build() + ranked = va.score_all() + print(f"V-A: {len(ranked)} scored candidates (freq>={va.freq_floor})\n") + + print("== top-30 V-A candidates ==") + gtsurf = {X.norm(s) for e in gt for s in e.surfaces} + for i, c in enumerate(ranked[:30]): + mark = "GT" if c.src in gtsurf else " " + print(f" {i:>3} [{mark}] {c.src:<8} f={c.freq:<4} cval={c.cvalue:8.1f} weird={c.weirdness:10.1f} score={c.score:10.1f}") + + print("\n== catastrophe screen (top-50) ==") + ok, res = catastrophe_ok(ranked, 50) + for s, r in res.items(): + print(f" {s}: rank {r}") + print(f" screen {'PASS' if ok else 'FAIL'}") + + for K in (50, 80, 120, 200, 400): + r_all, _, _, _ = gt_recall(ranked, gt, chunks, top_k=K) + r_f3, _, miss_f3, _ = gt_recall(ranked, gt, chunks, top_k=K, stratum_filter={"f>=10", "f3-9"}) + print(f"\n== top-{K}: recall(all GT)={r_all:.3f} recall(f>=3)={r_f3:.3f} misses(f>=3)={[e.src for e in miss_f3]}") diff --git a/eval/exp16/exp16_common.py b/eval/exp16/exp16_common.py new file mode 100644 index 0000000..4a6e92b --- /dev/null +++ b/eval/exp16/exp16_common.py @@ -0,0 +1,292 @@ +#!/usr/bin/env python3 +"""exp16 — bank-mining substrate (research/20 §B1/§D). Deterministic, $0. miner-v1. + +Shared layer for the detector arms A1–A3: candidate generation (Han char n-grams 1–6, NO word +segmentation — §B1/Qiu&Zhang), general-zh contrast corpus (jieba dict.txt, versioned artifact), +GT loading + occurrence counting (Aho-Corasick-equivalent WITHOUT suppressContained — D24.2), frequency +strata, the 60/40 tuning/test chapter split, and the catastrophe screen. + +Reuses the VALIDATED backend port eval/exp15/mem_select.py for memnorm normalization (byte-faithful to +memnorm.go). Candidate normalization is memnorm-symmetric so candidate keys and GT keys live in the same +space (trad2simp-folded, NFKC, lowercased) — and comparable to jieba's simplified-zh dict. + +Self-review by execution: `python exp16_common.py --selftest`. +""" +from __future__ import annotations + +import json +import math +import sys +from collections import Counter, defaultdict +from dataclasses import dataclass, field +from pathlib import Path + +import regex + +HERE = Path(__file__).resolve().parent +# APPEND exp15 (not insert(0)) so exp15's colliding module names (arms.py/chunker.py) do NOT shadow +# exp16's own modules; only unique names like mem_select are resolved from exp15. +sys.path.append(str(HERE.parent / "exp15")) +import mem_select as MS # validated memnorm port + +BOOK = Path("/home/ubuntu/books/gu-zhenren") +SEED = BOOK / "guzhenren-seed-v2.yaml" +RECORDS = BOOK / "rerun" / "records.json" +CONTRAST = HERE / "data" / "jieba_dict_general_zh.txt" # versioned general-zh word-freq artifact + +_RE_HAN = regex.compile(r"\p{Han}") + +# ── frozen substrate params (miner-v1); thresholds tuned on the tuning half are set in detectors.py ── +NGRAM_MIN, NGRAM_MAX = 1, 6 +FREQ_FLOOR = 3 # §A1/§B1: below this V-A does not score (conscious blind spot) +CATASTROPHE = ["方源", "蛊", "蛊师", "古月"] # §D4-в: must be in top-50 of the winner +# 60/40 chapter split (§D1): thresholds chosen on tuning, frozen, only test reported. +TUNING_CH = set(range(1, 16)) # chapters 1–15 +TEST_CH = set(range(16, 26)) # chapters 16–25 +ANNOTATION_CHUNK = (1, 0) # ch1/chunk0 = cover blurb + metadata (spoiler-region for blurb terms) + + +def norm(s: str) -> str: + return MS.normalize_source_key(s) + + +def han_only(s: str) -> bool: + return bool(s) and all(_RE_HAN.match(c) for c in s) + + +# ── materials ─────────────────────────────────────────────────────────────────────────────────── +@dataclass +class Chunk: + chapter: int + chunk_idx: int + source: str # raw zh + draft: str # raw ru draft (deepseek-v4-flash, WITH seed injection — confound §D1) + nsource: str = "" # memnorm-normalized source + + @property + def is_annotation(self) -> bool: + return (self.chapter, self.chunk_idx) == ANNOTATION_CHUNK + + +def load_chunks() -> list[Chunk]: + recs = json.load(open(RECORDS, encoding="utf-8")) + out = [] + for r in recs: + c = Chunk(chapter=r["chapter"], chunk_idx=r["chunk_idx"], + source=r["source"], draft=r.get("draft", "")) + c.nsource = norm(c.source) + out.append(c) + out.sort(key=lambda c: (c.chapter, c.chunk_idx)) + return out + + +@dataclass +class GTEntity: + src: str + dst: str + typ: str + status: str + since_ch: int + until_ch: int + aliases: list = field(default_factory=list) # alias surface strings + gender: str = "" + + @property + def surfaces(self) -> list[str]: + return [self.src] + list(self.aliases) + + @property + def norm_surfaces(self) -> list[str]: + return [norm(s) for s in self.surfaces if norm(s)] + + +def load_gt() -> list[GTEntity]: + rows = MS.load_seed_rows(str(SEED)) + out = [] + for r in rows: + out.append(GTEntity( + src=r["src"], dst=r.get("dst", "") or "", typ=r.get("type", "") or "", + status=r.get("status") or "approved", + since_ch=int(r.get("since_ch", 0) or 0), until_ch=int(r.get("until_ch", 0) or 0), + aliases=[a["alias"] for a in (r.get("aliases") or []) if a.get("alias")], + gender=r.get("gender", "") or "")) + return out + + +# ── occurrence counting (AC-equivalent, overlapping, NO suppressContained — D24.2) ──────────────── +def count_occurrences(needle_norm: str, chunks: list[Chunk], include_annotation: bool = True) -> int: + """Total overlapping occurrences of a normalized needle across chunk sources.""" + if not needle_norm: + return 0 + total = 0 + for c in chunks: + if c.is_annotation and not include_annotation: + continue + hay = c.nsource + pos = 0 + while True: + i = hay.find(needle_norm, pos) + if i < 0: + break + total += 1 + pos = i + 1 + return total + + +def gt_occurrences(ent: GTEntity, chunks: list[Chunk], include_annotation: bool = True) -> int: + """Entity occurrence count = sum over its normalized surfaces (src + aliases), overlapping.""" + return sum(count_occurrences(ns, chunks, include_annotation) for ns in ent.norm_surfaces) + + +def gt_chapters(ent: GTEntity, chunks: list[Chunk], include_annotation: bool = True) -> set[int]: + chs = set() + for c in chunks: + if c.is_annotation and not include_annotation: + continue + if any(ns and ns in c.nsource for ns in ent.norm_surfaces): + chs.add(c.chapter) + return chs + + +def freq_stratum(f: int) -> str: + if f >= 10: + return "f>=10" + if f >= 3: + return "f3-9" + return "f<3" + + +# ── general-zh contrast corpus (jieba dict.txt: 'word freq POS' per line) ────────────────────────── +class Contrast: + """General-domain zh reference: word frequencies + derived char frequencies. Versioned artifact.""" + + def __init__(self, path: Path = CONTRAST): + self.word_freq: dict[str, int] = {} + self.char_freq: Counter = Counter() + total_w = 0 + for line in path.read_text(encoding="utf-8").splitlines(): + parts = line.split(" ") + if len(parts) < 2: + continue + w, fr = parts[0], parts[1] + try: + fr = int(fr) + except ValueError: + continue + wn = norm(w) # normalize the dict key into the same space as candidates + if not wn: + continue + # keep the max freq if normalization collides (trad/simp variants folding together) + if wn not in self.word_freq or fr > self.word_freq[wn]: + self.word_freq[wn] = fr + total_w += fr + for ch in wn: + self.char_freq[ch] += fr + self.total_word = total_w + self.total_char = sum(self.char_freq.values()) + + def word_rel(self, ngram_norm: str) -> float: + """P_general(ngram) as a word (0 if OOV).""" + return self.word_freq.get(ngram_norm, 0) / self.total_word if self.total_word else 0.0 + + def char_indep_rel(self, ngram_norm: str) -> float: + """Expected P_general(ngram) under char-independence (smoothing for OOV multiword; §A1).""" + p = 1.0 + for ch in ngram_norm: + pc = (self.char_freq.get(ch, 0) + 1) / (self.total_char + len(self.char_freq)) + p *= pc + return p + + def char_rarity(self, ngram_norm: str) -> float: + """Mean surprisal of constituent chars in general zh (rare chars like 蛊/窍 → high).""" + if not ngram_norm: + return 0.0 + s = 0.0 + for ch in ngram_norm: + pc = (self.char_freq.get(ch, 0) + 1) / (self.total_char + len(self.char_freq)) + s += -math.log(pc) + return s / len(ngram_norm) + + +# ── candidate generation: Han char n-grams 1–6 over normalized source (§B1) ──────────────────────── +def han_runs(ntext: str): + """Maximal runs of Han characters in normalized text.""" + runs = [] + cur = [] + for ch in ntext: + if _RE_HAN.match(ch): + cur.append(ch) + else: + if cur: + runs.append("".join(cur)) + cur = [] + if cur: + runs.append("".join(cur)) + return runs + + +def enumerate_candidates(chunks: list[Chunk], nmin=NGRAM_MIN, nmax=NGRAM_MAX) -> Counter: + """All Han n-grams (length nmin..nmax) with overlapping counts across chunk sources. + Annotation chunk INCLUDED (blurb terms live there; GT filter handles the blurb rule separately).""" + cnt = Counter() + for c in chunks: + for run in han_runs(c.nsource): + L = len(run) + for n in range(nmin, nmax + 1): + for i in range(0, L - n + 1): + cnt[run[i:i + n]] += 1 + return cnt + + +def candidate_chapters(cand_norm: str, chunks: list[Chunk]) -> set[int]: + return {c.chapter for c in chunks if cand_norm in c.nsource} + + +if __name__ == "__main__": + if "--selftest" not in sys.argv: + print("usage: python exp16_common.py --selftest"); sys.exit(0) + + chunks = load_chunks() + gt = load_gt() + print(f"chunks: {len(chunks)} (chapters {min(c.chapter for c in chunks)}–{max(c.chapter for c in chunks)})") + print(f"GT entities: {len(gt)} aliases: {sum(len(e.aliases) for e in gt)}") + + # GT occurrence filter (§D1): every GT term must have >=1 occurrence. Blurb rule both ways. + print("\n== GT occurrence filter (incl. vs excl. annotation chunk) ==") + n_pass_incl = n_pass_excl = 0 + blurb_terms = [] + for e in gt: + oi = gt_occurrences(e, chunks, include_annotation=True) + oe = gt_occurrences(e, chunks, include_annotation=False) + n_pass_incl += oi >= 1 + n_pass_excl += oe >= 1 + if oi >= 1 and oe == 0: + blurb_terms.append((e.src, e.dst, e.since_ch, oi)) + print(f" pass (incl annotation): {n_pass_incl}/{len(gt)}") + print(f" pass (excl annotation): {n_pass_excl}/{len(gt)}") + print(f" annotation-only terms ({len(blurb_terms)}): " + + ", ".join(f"{s}({dst},since{sc},occ{o})" for s, dst, sc, o in blurb_terms)) + + # frequency strata of GT (incl annotation) + strata = Counter(freq_stratum(gt_occurrences(e, chunks)) for e in gt) + print(f"\n== GT freq strata (incl annotation): {dict(strata)} ==") + + # candidate space + cands = enumerate_candidates(chunks) + above = {k: v for k, v in cands.items() if v >= FREQ_FLOOR} + print(f"\ncandidates: {len(cands)} distinct Han n-grams; {len(above)} at freq>={FREQ_FLOOR}") + + # catastrophe entities present + their counts + print("\n== catastrophe screen entities ==") + for s in CATASTROPHE: + print(f" {s}: source-occ={count_occurrences(norm(s), chunks)} in-candidate-set={norm(s) in cands} " + f"(freq {cands.get(norm(s),0)})") + + # contrast smoke + C = Contrast() + print(f"\ncontrast: {len(C.word_freq)} words, {len(C.char_freq)} chars, total_word={C.total_word}") + for w in ["方源", "蛊", "蛊师", "古月", "自己", "一个", "时候"]: + wn = norm(w) + print(f" {w:<4} word_rel={C.word_rel(wn):.2e} char_rarity={C.char_rarity(wn):.2f} " + f"char_indep={C.char_indep_rel(wn):.2e}") + print("\nselftest OK") diff --git a/eval/exp16/length_sweep.py b/eval/exp16/length_sweep.py new file mode 100644 index 0000000..4b5aff0 --- /dev/null +++ b/eval/exp16/length_sweep.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +"""exp16 PRE-TASK (D39.9, $0) — sweep finish_reason across ALL exp14 / exp14b cost ledgers. + +The exp14 harness bug (exp14_common.call, lines 104-113): a response with finish_reason=length is +returned as (text, err=None, usage) whenever text is non-empty — a TRUNCATED output passes as valid. +Only content_filter and empty-text are rejected. Every ledger record stamps usage.finish_reason, so +we can sweep them post-hoc and flag any accepted truncation that fed the DET scoring. + +We do NOT edit the exp14 rig. This is a read-only sweep. For any finish=length record, we report +(arm/model, id, err, completion_tokens, cost) and whether it maps to a DET-battery chunk used in +scoring. Cross-reference with reaudit_14b.py's 'truncated_looking' file heuristic. +""" +from __future__ import annotations +import json +from collections import defaultdict +from pathlib import Path + +LEDGERS = [ + Path("/home/ubuntu/books/gu-zhenren/exp14/costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14/judge_costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14/rank_costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14/regate_costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14/sizecurve_costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14/gemini_retest_costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14b/costs.jsonl"), + Path("/home/ubuntu/books/gu-zhenren/exp14b/judge_costs.jsonl"), +] + +# DET-battery chunks that fed exp14b DET scoring (a1/a2 7.0, b1 10.1, b2 6.0, c1 17.0, c2 9.1, c3 16.0, d1 19.0, d2 7.0) +DET_CHUNKS = {"7.0", "10.1", "6.0", "17.0", "9.1", "16.0", "19.0"} + + +def main(): + by_finish = defaultdict(int) # finish_reason -> count + by_model_finish = defaultdict(lambda: defaultdict(int)) + length_records = [] # accepted-or-not length records + total = 0 + + for lp in LEDGERS: + if not lp.exists(): + print(f"[skip] {lp} (absent)") + continue + for line in lp.read_text(encoding="utf-8").splitlines(): + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + total += 1 + usage = rec.get("usage", {}) or {} + fr = str(usage.get("finish_reason", "")) or "(none)" + model = rec.get("model", rec.get("arm", "?")) + by_finish[fr] += 1 + by_model_finish[model][fr] += 1 + if fr.lower() == "length": + length_records.append({ + "ledger": lp.parent.name + "/" + lp.name, + "arm": rec.get("arm"), "model": model, "id": rec.get("id"), + "attempt": rec.get("attempt"), "err": rec.get("err"), + "completion_tokens": usage.get("completion_tokens"), + "cost": rec.get("cost"), + }) + + print("=" * 90) + print(f"finish_reason SWEEP — {total} ledger records across {len(LEDGERS)} ledgers") + print("=" * 90) + print("\nfinish_reason distribution (all records, incl. every attempt & judge call):") + for fr, n in sorted(by_finish.items(), key=lambda x: -x[1]): + print(f" {fr:<16} {n}") + + print("\nfinish=length by model:") + any_len = False + for model, fd in sorted(by_model_finish.items()): + if fd.get("length"): + any_len = True + print(f" {model:<22} length={fd['length']} (all finish: {dict(fd)})") + if not any_len: + print(" (none)") + + print("\n" + "=" * 90) + print(f"finish=length RECORDS: {len(length_records)}") + print(" err=None + non-empty text => ACCEPTED as valid by the buggy gate (truncation in the pool)") + print("=" * 90) + for r in length_records: + accepted = r["err"] in (None, "", "null") + det = " " if r["id"] in DET_CHUNKS else "" + tag = "ACCEPTED-TRUNCATION" if accepted else f"rejected(err={r['err']})" + print(f" {r['ledger']:<22} arm={str(r['arm']):<4} model={r['model']:<20} id={str(r['id']):<6} " + f"attempt={r['attempt']} comp_tok={r['completion_tokens']} ${r['cost']} => {tag}{det}") + + json.dump({"by_finish": dict(by_finish), + "by_model_finish": {m: dict(f) for m, f in by_model_finish.items()}, + "length_records": length_records}, + open(Path("/home/ubuntu/books/gu-zhenren/exp16") / "length_sweep.json", "w"), + ensure_ascii=False, indent=1) + print(f"\n[written] /home/ubuntu/books/gu-zhenren/exp16/length_sweep.json") + + +if __name__ == "__main__": + main() diff --git a/eval/exp16/palladius.py b/eval/exp16/palladius.py new file mode 100644 index 0000000..79d43f2 --- /dev/null +++ b/eval/exp16/palladius.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +"""exp16 — Palladius (Палладий) transliteration table + ru-side name detector (research/20 §B1 ch5, §B5 +P4). miner-v1. $0. Versioned artifact (the syllable set is generated deterministically from the mapping +rules below and self-validated against the GT dst names). + +Two uses: + (5) ru-side detector: a ru token that segments cleanly into Palladius syllables = the draft model + treated it as a transliterated Chinese NAME -> high-precision name spot WITH a ready dst. + canon conformity (§C2-3): score whether a proposed dst for a type=name/place term conforms to Palladius. + +The pinyin->Cyrillic Palladius system is a fixed ~400-syllable table. We generate it from initials × +finals + the documented irregularities (zhi/zi/…->ы-class; y-/w- whole syllables; ü-class). Coverage is +validated by requiring every GT dst name syllable to be recognized (assert in __main__). +""" +from __future__ import annotations + +import regex + +import exp16_common as X + +# ── initials (声母) → Palladius ──────────────────────────────────────────────────────────────────── +INITIALS = { + "b": "б", "p": "п", "m": "м", "f": "ф", "d": "д", "t": "т", "n": "н", "l": "л", + "g": "г", "k": "к", "h": "х", "j": "цз", "q": "ц", "x": "с", + "zh": "чж", "ch": "ч", "sh": "ш", "r": "ж", "z": "цз", "c": "ц", "s": "с", + "": "", # zero initial (y/w handled as whole syllables below) +} + +# ── finals (韵母) → Palladius (base forms; some initial-conditioned variants applied in build) ────── +FINALS = { + "a": "а", "o": "о", "e": "э", "ai": "ай", "ei": "эй", "ao": "ао", "ou": "оу", + "an": "ань", "en": "энь", "ang": "ан", "eng": "эн", "er": "эр", "ong": "ун", + "i": "и", "ia": "я", "ie": "е", "iao": "яо", "iu": "ю", "ian": "янь", + "in": "инь", "iang": "ян", "ing": "ин", "iong": "юн", + "u": "у", "ua": "уа", "uo": "о", "uai": "уай", "ui": "уй", "uan": "уань", + "un": "унь", "uang": "уан", "ueng": "ун", + "v": "юй", "ve": "юэ", "van": "юань", "vn": "юнь", # ü written as v +} + +# whole zero-initial syllables (y-/w-) +Y_W = { + "yi": "и", "ya": "я", "ye": "е", "yao": "яо", "you": "ю", "yan": "янь", "yin": "инь", + "yang": "ян", "ying": "ин", "yong": "юн", "yu": "юй", "yue": "юэ", "yuan": "юань", "yun": "юнь", + "wu": "у", "wa": "ва", "wo": "во", "wai": "вай", "wei": "вэй", "wan": "вань", "wen": "вэнь", + "wang": "ван", "weng": "вэн", +} +# retroflex/sibilant + i => -ы/-и class (zhi chi shi ri zi ci si) +SPECIAL_I = { + "zhi": "чжи", "chi": "чи", "shi": "ши", "ri": "жи", "zi": "цзы", "ci": "цы", "si": "сы", +} +# initials that take j/q/x with ü finals written as u (ju->цзюй etc.) +_JQX = {"j", "q", "x"} + + +def _final_after(initial: str, final: str, cyr_ini: str, cyr_fin: str) -> str: + """Apply the few initial-conditioned Palladius adjustments.""" + # e after most initials -> э, but after certain -> е is not standard; keep э. + # 'o' after b/p/m/f -> о (бо/по/мо/фо); 'uo' after them n/a. + # 'ie'->е, 'ei'->эй are fine. Keep base mapping. + return cyr_ini + cyr_fin + + +def build_syllables() -> dict[str, str]: + """pinyin syllable -> Palladius. Generated; not exhaustive of tone/rare finals but covers names.""" + syl = {} + syl.update(Y_W) + syl.update(SPECIAL_I) + for pi, ci in INITIALS.items(): + if pi == "": + continue + for pf, cf in FINALS.items(): + # ü finals (v) only valid after j/q/x/l/n; jqx write ü as plain u in pinyin + if pf in ("v", "ve", "van", "vn"): + if pi in _JQX: + py = pi + pf.replace("v", "u") # ju/jue/juan/jun + elif pi in ("l", "n"): + py = pi + pf.replace("v", "ü") # lü/nü + else: + continue + else: + py = pi + pf + # skip the retroflex/sibilant + bare i (handled by SPECIAL_I) + if pf == "i" and pi in ("zh", "ch", "sh", "r", "z", "c", "s"): + continue + syl[py] = _final_after(pi, pf, ci, cf) + return syl + + +SYLLABLES = build_syllables() +# Cyrillic syllable inventory (values), longest-first for greedy segmentation +_CYR_SYL = sorted(set(SYLLABLES.values()), key=len, reverse=True) +_RE_CYR_WORD = regex.compile(r"^[\p{Cyrillic}]+$") + + +def is_palladius_token(token: str, min_syllables=1) -> bool: + """True if `token` (a ru word) segments fully into Palladius syllables (greedy longest-match). + Used as a high-precision 'this looks like a transliterated Chinese name' signal.""" + t = token.strip().lower().replace("ъ", "") + if not t or not _RE_CYR_WORD.match(t): + return False + n_syl = 0 + i = 0 + while i < len(t): + for s in _CYR_SYL: + if s and t.startswith(s, i): + i += len(s) + n_syl += 1 + break + else: + return False + return n_syl >= min_syllables + + +def palladius_conformant(dst: str) -> bool: + """Canon conformity (§C2-3): every whitespace/hyphen-separated Cyrillic word of dst is a Palladius + token. For multi-word names (Фан Юань) all parts must conform. Non-Cyrillic parts are ignored.""" + words = regex.findall(r"[\p{Cyrillic}]+", dst) + cyr = [w for w in words if _RE_CYR_WORD.match(w)] + if not cyr: + return False + return all(is_palladius_token(w, min_syllables=1) for w in cyr) + + +def ru_name_tokens(draft: str, min_syllables=2) -> list[str]: + """Capitalized ru tokens in a draft that parse as Palladius (>=min_syllables) — model-flagged names.""" + out = [] + for tok in regex.findall(r"\b[\p{Lu}][\p{Cyrillic}]+", draft): + if is_palladius_token(tok, min_syllables=min_syllables): + out.append(tok) + return out + + +if __name__ == "__main__": + print(f"Palladius syllables generated: {len(SYLLABLES)} pinyin -> {len(set(SYLLABLES.values()))} cyr forms") + # validate: every GT dst name/place syllable must be recognized + gt = X.load_gt() + name_dsts = [e.dst for e in gt if e.typ in ("name",) and e.dst] + print("\n== GT name dst recognition ==") + ok = bad = 0 + for e in gt: + if e.typ not in ("name", "place"): + continue + # take the transliterated part(s) — for places skip the leading common noun (гора/деревня) + words = regex.findall(r"[\p{Cyrillic}]+", e.dst) + translit = [w for w in words if w.lower() not in ("гора", "деревня", "село", "селение", "стан", "клан", "род")] + good = all(is_palladius_token(w) for w in translit) if translit else False + ok += good + bad += not good + mark = "OK " if good else "MISS" + print(f" {mark} {e.src:<6} {e.dst:<26} translit={translit} " + f"{[w for w in translit if not is_palladius_token(w)] if not good else ''}") + print(f"\nrecognized {ok}/{ok+bad} GT name/place dst") + # negatives: common Russian words should NOT be Palladius + print("\n== negative control (common ru words, should be FALSE) ==") + for w in ["человек", "который", "сказал", "деревня", "истинный", "камень", "мастер", "ранг"]: + print(f" {w:<12} palladius={is_palladius_token(w)}") diff --git a/eval/exp16/patterns.py b/eval/exp16/patterns.py new file mode 100644 index 0000000..e916995 --- /dev/null +++ b/eval/exp16/patterns.py @@ -0,0 +1,236 @@ +#!/usr/bin/env python3 +"""exp16 — V-C pattern channels (research/20 §B1 channels 1–4/6). miner-v1. $0. + +Language×genre pattern pack for zh (§B5 plugin P3). Versioned data (surnames / title affixes / topo +suffixes) + a book-adaptive productive-morphology detector (channel 4 — auto-detects the domain formant, +NOT hardcoded 蛊). Each channel proposes TYPED candidates (name/title/place/term) with evidence, closing +V-A's frequency-blind classes (rare surname-anchored names, rank/grade titles, one-off realia). + +Channels: + (1) surname anchor 百家姓 + compound surnames, + 1–2 Han window right -> name + (2) title affixes suffix 公子/大人/长老/前辈/族长/嬷嬷…, prefix 老/小/阿 + ordinals -> title/name + (3) topo suffixes 山/寨/村/疆/谷/城/门/宗… -> place + (4) productive morphology char that binds with >=m distinct n-grams (蛊/等/转…) -> term/title (auto) + (6) genre lexicon pack xianxia suffix/title lexicon (this file IS the zh-xianxia pack) + +Owner decision 18.07: genre PACKS are NOT built (deferred to book 2); channel (6) runs only on UNIVERSAL +channels — 百家姓 anchor, title/topo suffixes, productive morphology (auto-detected formant, not hardcoded +蛊), Palladius on the ru side. This module encodes exactly that universal set. +""" +from __future__ import annotations + +from collections import Counter, defaultdict + +import exp16_common as X + +# ── (1) surnames: 百家姓 single-char subset + compound surnames (versioned inventory) ─────────────── +# Standard 百家姓 opening + the surnames present in this book's cast. Compound surnames listed explicitly. +SURNAMES_SINGLE = set( + "赵钱孙李周吴郑王冯陈褚卫蒋沈韩杨朱秦尤许何吕施张孔曹严华金魏陶姜" + "戚谢邹喻柏水窦章云苏潘葛奚范彭郎鲁韦昌马苗凤花方俞任袁柳酆鲍史唐" + "费廉岑薛雷贺倪汤滕殷罗毕郝邬安常乐于时傅皮卞齐康伍余元卜顾孟平黄" + "和穆萧尹姚邵湛汪祁毛禹狄米贝明臧计伏成戴谈宋茅庞熊纪舒屈项祝董梁" + "杜阮蓝闵席季麻强贾路娄危江童颜郭梅盛林刁钟徐邱骆高夏蔡田樊胡凌霍" + "虞万支柯昝管卢莫经房裘缪干解应宗丁宣贲邓郁单杭洪包诸左石崔吉钮龚" + "白凝" # 白 (Bai clan). 凝 is not a surname but kept out — handled below via compound guard. +) +SURNAMES_SINGLE.discard("凝") +SURNAMES_COMPOUND = {"古月", "欧阳", "司马", "上官", "夏侯", "诸葛", "东方", "皇甫", "尉迟", "公孙", + "慕容", "长孙", "宇文", "司徒", "鲜于", "南宫"} + +# ── (2) title affixes (universal honorific/rank suffixes+prefixes; §A4 Cao) ───────────────────────── +TITLE_SUFFIX = ["公子", "大人", "长老", "前辈", "族长", "家老", "老祖", "祖师", "真人", "上人", + "道人", "先生", "夫人", "娘子", "姑娘", "嬷嬷", "师傅", "师父", "掌门", "宗主", + "少爷", "老爷", "小姐", "婆婆", "大娘", "大爷"] +TITLE_PREFIX = ["老", "小", "阿"] +ORDINAL = ["一代", "二代", "三代", "四代", "五代", "六代", "七代", "八代", "九代", "十代", + "第一", "第二", "第三", "第四", "第五"] + +# ── (3) topo suffixes -> place ───────────────────────────────────────────────────────────────────── +TOPO_SUFFIX = ["山", "寨", "村", "疆", "谷", "城", "门", "宗", "府", "洞", "峰", "岭", "河", + "江", "湖", "海", "林", "原", "岛", "潭", "殿", "阁", "楼", "院", "堂", "祠"] + +# ── (2c) rank/grade compositional pattern (numeral/sequential + rank-word) -> title ──────────────── +# Grade/rank words are COMMON chars (等 over_rep 1.1, 转 3.9) — not formant-detectable; caught by the +# sequential-prefix + rank-word composition instead. Universal (numeral+rank), not genre-specific. +GRADE_PREFIX = list("甲乙丙丁戊己庚辛壬癸") # sequential grade markers +NUMERAL = list("一二三四五六七八九十零百千") # Chinese numerals +RANK_WORD = ["等", "转", "阶", "级", "重", "品", "段", "层"] # rank/grade head words + +PACK_VERSION = "zh-universal-v1" + + +def is_surname_start(s: str): + """Return the surname prefix (compound preferred) if s starts with one, else None.""" + for cs in SURNAMES_COMPOUND: + if s.startswith(cs): + return cs + if s and s[0] in SURNAMES_SINGLE: + return s[0] + return None + + +# ── (4) productive-morphology (formant) auto-detection ───────────────────────────────────────────── +def book_char_freq(chunks) -> Counter: + bc = Counter() + for c in chunks: + for ch in c.nsource: + if X._RE_HAN.match(ch): + bc[ch] += 1 + return bc + + +def detect_formants(chunks, cand_freq: dict[str, int], contrast: X.Contrast, min_partners=3, + min_over_rep=15.0) -> dict[str, dict]: + """A Han char c is a productive DOMAIN formant if it binds (suffix OR prefix) with >= min_partners + DISTINCT content morphemes among candidates AND is over-represented in-book vs general zh + (p_book/p_general >= min_over_rep). Over-representation — NOT char-rarity — is the domain signal: + it isolates 蛊(2723×)/窍(88×)/虫(32×) while rejecting common chars 师/花/等(1.1×). Auto-detected, + not hardcoded (§B1 channel 4). Common rank-words (等/转) are handled by the compositional channel.""" + bc = book_char_freq(chunks) + tot_book = sum(bc.values()) or 1 + suf = defaultdict(set) + pre = defaultdict(set) + for a in cand_freq: + if len(a) < 2: + continue + suf[a[-1]].add(a[:-1]) + pre[a[0]].add(a[1:]) + formants = {} + for c in set(suf) | set(pre): + p_book = bc.get(c, 0) / tot_book + p_gen = (contrast.char_freq.get(c, 0) + 1) / (contrast.total_char + len(contrast.char_freq)) + over_rep = p_book / p_gen if p_gen else 0.0 + if over_rep < min_over_rep: + continue + ns, npr = len(suf.get(c, ())), len(pre.get(c, ())) + role = None + if ns >= min_partners: + role = "suffix" + if npr >= min_partners: + role = "both" if role else "prefix" + if role: + formants[c] = {"role": role, "over_rep": round(over_rep, 1), + "suffix_partners": ns, "prefix_partners": npr} + return formants + + +# formant -> candidate type (universal heuristic; not genre-conditioned) +def formant_type(c: str) -> str: + if c in TOPO_SUFFIX: + return "place" + if c in ("等", "转", "阶"): + return "title" + return "term" + + +# ── channel application: produce typed pattern candidates from the source ────────────────────────── +def pattern_candidates(chunks, cand_freq: dict[str, int], contrast: X.Contrast, + min_partners=3, min_over_rep=15.0): + """Return {cand_norm: {'types': [...], 'evidence': [...]}} for pattern-detected candidates. + Operates over the ALREADY-normalized source; proposes candidates that may be BELOW the V-A freq + floor (that is the point — patterns close the rare-term blind spot).""" + out = defaultdict(lambda: {"types": [], "evidence": []}) + + def add(cand, typ, ev): + cand = X.norm(cand) + if not cand or not X.han_only(cand): + return + if typ not in out[cand]["types"]: + out[cand]["types"].append(typ) + if ev not in out[cand]["evidence"]: + out[cand]["evidence"].append(ev) + + # scan each source for surname / title / topo patterns over Han runs + for c in chunks: + for run in X.han_runs(c.nsource): + L = len(run) + for i in range(L): + # (1) surname anchor: surname + 1–2 Han given-name window + sn = is_surname_start(run[i:]) + if sn: + base = i + len(sn) + for gl in (1, 2): + if base + gl <= L: + full = run[i:base + gl] + if 2 <= len(full) <= 4: + add(full, "name", f"surname:{sn}") + # (2) title suffix: content + suffix + for suf in TITLE_SUFFIX: + if run.startswith(suf, i): + # take up to 3 Han to the LEFT as the titled base (e.g. 沈+嬷嬷, 四代+族长) + for left in (3, 2, 1, 0): + if i - left >= 0: + cand = run[i - left:i + len(suf)] + if 2 <= len(cand) <= 6: + add(cand, "title", f"title_suffix:{suf}") + add(suf, "title", f"title_bare:{suf}") + # (2b) ordinal + title (四代族长-class) + for od in ORDINAL: + if run.startswith(od, i): + for suf in TITLE_SUFFIX: + end = i + len(od) + if run.startswith(suf, end): + add(run[i:end + len(suf)], "title", f"ordinal_title:{od}+{suf}") + # (2c) rank/grade compositional: (numeral|grade-prefix) + rank-word -> title + for rw in RANK_WORD: + if run.startswith(rw, i) and i >= 1: + left = run[i - 1] + if left in NUMERAL or left in GRADE_PREFIX: + add(run[i - 1:i + len(rw)], "title", f"rank_grade:{left}+{rw}") + # (3) topo suffix: 1–3 Han base + topo char + if run[i] in TOPO_SUFFIX and i >= 1: + for left in (3, 2, 1): + if i - left >= 0: + cand = run[i - left:i + 1] + if 2 <= len(cand) <= 4: + add(cand, "place", f"topo_suffix:{run[i]}") + + # (4) productive morphology: for each detected formant, propose all binding n-grams + formants = detect_formants(chunks, cand_freq, contrast, min_partners, min_over_rep) + for c in chunks: + for run in X.han_runs(c.nsource): + L = len(run) + for i in range(L): + if run[i] not in formants: + continue + info = formants[run[i]] + typ = formant_type(run[i]) + if info["role"] in ("suffix", "both"): + for left in (3, 2, 1): + if i - left >= 0: + cand = run[i - left:i + 1] + if 2 <= len(cand) <= 4: + add(cand, typ, f"formant_suffix:{run[i]}") + if info["role"] in ("prefix", "both"): + for r in (2, 3): + if i + r <= L: + cand = run[i:i + r] + if 2 <= len(cand) <= 4: + add(cand, typ, f"formant_prefix:{run[i]}") + return dict(out), formants + + +if __name__ == "__main__": + chunks = X.load_chunks() + gt = X.load_gt() + C = X.Contrast() + import detectors as D + va = D.VA(chunks, C).build() + pats, formants = pattern_candidates(chunks, va.cand_freq, C) + print(f"pattern candidates: {len(pats)}") + print(f"\ndetected formants ({len(formants)}):") + for c, info in sorted(formants.items(), key=lambda kv: -kv[1]["over_rep"]): + print(f" {c} {info}") + # how many f>=3-miss GT terms does the pattern layer now propose? + gtsurf = {X.norm(s): e for e in gt for s in e.surfaces} + hit = [s for s in gtsurf if s in pats] + print(f"\nGT surfaces proposed by patterns: {len(hit)}/{len(gtsurf)}") + # specifically the V-A f>=3 misses + focus = ["转", "一转", "三转", "五转", "六转", "甲等", "乙等", "丙等", "丁等", "白凝冰", "江牙", + "沈嬷嬷", "四代族长", "族长", "青茅山", "古月山寨", "希望蛊", "月影蛊", "月光蛊", "人祖"] + print("\nfocus (V-A structural misses) — proposed by patterns?") + for s in focus: + sn = X.norm(s) + info = pats.get(sn) + print(f" {s:<6} {'YES ' + str(info['types']) + ' ' + str(info['evidence'][:2]) if info else 'no'}") diff --git a/eval/exp16/reaudit_14b.py b/eval/exp16/reaudit_14b.py new file mode 100644 index 0000000..99188d3 --- /dev/null +++ b/eval/exp16/reaudit_14b.py @@ -0,0 +1,204 @@ +#!/usr/bin/env python3 +"""exp16 PRE-TASK (D39.9, $0, BEFORE the pre-reg freeze) — re-audit of the exp14b DET meaning-trap +battery with the CORRECTED sentence-scoped rules (eval/exp15/q4a_traps.py), against the ORIGINAL +exp14b rules and against the D38 §2.1 MANUAL verdicts printed in docs/experiments/14b-meaning-battery.md. + +Why: Q4a (D39.9) established that exp14b's own auto-scorer rules (rule_counterfactual / polarity_envy / +b1) are defective and that its harness accepts finish=length. D38's DET table was scored by MANUAL +span-reading (the auto-scorer was self-rejected as bricked). This task re-scores the SAVED exp14b +outputs deterministically with the corrected rules and reports which per-class D38 verdicts shift. + +Discipline (CLAUDE.md self-review mandate): additive layer only — we do NOT edit eval/exp14b or its +det_rates.json. We import both scorers read-only. Every scored cell prints its scoped span for manual +verification. Traps that cannot be located are reported as '?' (never a silent pass). Load-bearing +a1/c1 verdicts are meant to be span-verified independently downstream; a flip of any load-bearing D38 +verdict → STOP + ping orchestrator (errata is the orchestrator's zone). + +Output: eval/exp16 artifacts printed to stdout + a machine-readable JSON dumped OUTSIDE git. +""" +from __future__ import annotations +import json +import sys +from pathlib import Path + +HERE = Path(__file__).resolve().parent +sys.path.insert(0, str(HERE.parent / "exp15")) # q4a_traps (corrected scorer) +sys.path.insert(0, str(HERE.parent / "exp14b")) # exp14b_score (original rules, read-only) + +import q4a_traps as Q # corrected sentence-scoped scorer +import exp14b_score as S # original line-scoped rules + TRAPS anchors + +ARMS_DIR = Path("/home/ubuntu/books/gu-zhenren/exp14b/arms") +EX14_ARMS = Path("/home/ubuntu/books/gu-zhenren/exp14/arms") +OUT = Path("/home/ubuntu/books/gu-zhenren/exp16") +OUT.mkdir(parents=True, exist_ok=True) + +# arms scored in D38 §2.1 (+ reference). K = kimi dropped (only 10.1 saved). P0 = prod baseline (exp14). +ARMS = ["C", "F", "X", "D", "M", "K"] +ARM_MODEL = {"C": "glm-5", "F": "gpt-5.4", "X": "grok-4.3", "D": "deepseek-v4-pro", + "M": "mistral-large", "K": "kimi-k2.6", "P0": "prod-baseline"} + +# D38 §2.1 MANUAL verdicts (the baseline being re-audited). Only a1/a2/b2/c1/c3 were tabulated +# manually; b1/c2/d1/d2 have NO D38 manual baseline (produced fresh here). '?'=undecided/paraphrase. +D38_MANUAL = { + "a1": {"C": "fail", "F": "fix", "X": "fail", "D": "fix", "M": "fix"}, + "a2": {"C": "fix", "F": "fix", "X": "fix", "D": "fix", "M": "fix"}, + "b2": {"C": "fix", "F": "fix", "X": "fix", "D": "fix", "M": "fix"}, + "c1": {"C": "fix", "F": "fix", "X": "fail", "D": "fail", "M": "?"}, + "c3": {"C": "fail", "F": "fix", "X": "fix", "D": "fix", "M": "fail"}, +} +# Load-bearing conclusions of D38 that a flip would undermine (prompt: несущие a1/c1 вердикты): +# (1) "only gpt-5.4 fixes meaning" REFUTED -> depends on a1: D=fix AND M=fix, and F=fix. +# (2) failures scattered by model×construction -> a1 C=fail/X=fail, c1 X=fail/D=fail. +# (3) c3 = glossary confound, not capability. +LOAD_BEARING = { # (tid, arm) cells whose flip is load-bearing for D38's ratified conclusions + ("a1", "F"), ("a1", "D"), ("a1", "M"), ("a1", "C"), ("a1", "X"), + ("c1", "X"), ("c1", "D"), ("c1", "C"), ("c1", "F"), +} +# Manual span-read RESOLUTION of corrected-scorer '?' on load-bearing cells (author verified against +# raw text; independently re-checked downstream). A '?' is a locator-coverage gap, NOT a semantic +# reversal — resolved here to the verdict the cited faithful span supports. +MANUAL_RESOLUTION = { + ("a1", "F"): ("fix", "«…в клане Гуюэ не было ни одного человека, который бы его не знал» = faithful " + "double-negation; corrected locator's story→knowledge 90-char window overshot."), + ("c1", "F"): ("fix", "«…подняться над обычными людьми…» = faithful 人上之人; locator 'над людьми' " + "split by 'обычными', so no match → '?'."), +} +# A shift is SEMANTIC only if fix↔fail. fix↔'?' or '?'↔fix = COVERAGE-GAP (scorer could not locate). +def shift_kind(a, b): + s = {a, b} + if s == {"fix", "fail"}: + return "SEMANTIC" + if "?" in s: + return "coverage-gap" + return "other" + + +def arm_file(arm: str, cid: str) -> Path: + if arm == "P0": + return EX14_ARMS / "P0" / f"{cid}.txt" + return ARMS_DIR / arm / f"{cid}.txt" + + +def original_verdict(tid: str, text: str): + """Reconstruct the exp14b ORIGINAL (line-scoped) verdict for trap tid on `text`.""" + for t_id, cid, anchor, rule, desc in S.TRAPS: + if t_id == tid: + span = S.find_span(text, anchor) + return (rule(span) if span else "?"), span + return "?", "" + + +def truncated_looking(text: str) -> bool: + """Heuristic flag: file appears truncated (ends without sentence-ender, unusually short).""" + t = (text or "").rstrip() + if not t: + return True + return t[-1] not in ".!?…\"»)”'" + + +def main(): + traps = {t["id"]: t for t in Q.traps()} + order = ["a1", "a2", "b1", "b2", "c1", "c2", "c3", "d1", "d2"] + + results = {} # tid -> arm -> {corrected, original, d38, span, sent, shift} + shifts = [] # (tid, arm, d38, corrected) + load_bearing_flips = [] + trunc_flags = [] + missing = [] + + print("=" * 100) + print("exp16 PRE-TASK — re-audit exp14b DET battery with CORRECTED sentence-scoped rules (q4a_traps)") + print("legend: corr=corrected(q4a) orig=original(exp14b) D38=manual baseline | ✗КАТ=catastrophe class") + print("=" * 100) + + for tid in order: + trap = traps[tid] + cid = Q.CHUNK[tid] + cat = " [CAT]" if tid in Q.CAT_CLASS else "" + print(f"\n── {tid} (ch {cid}){cat} {trap['desc']}") + print(f" src: {trap['src_zone'][:60]}") + results[tid] = {} + for arm in ARMS + (["P0"] if arm_file("P0", cid).exists() else []): + fp = arm_file(arm, cid) + if not fp.exists(): + results[tid][arm] = {"corrected": None, "original": None, "note": "no file"} + continue + text = fp.read_text(encoding="utf-8") + corr, sent = Q.det_score(text, trap) + orig, ospan = original_verdict(tid, text) + d38 = D38_MANUAL.get(tid, {}).get(arm) + trunc = truncated_looking(text) + if trunc: + trunc_flags.append((tid, arm, cid, len(text))) + shift = None + if d38 is not None and corr != d38: + kind = shift_kind(d38, corr) + shift = (d38, corr, kind) + shifts.append((tid, arm, d38, corr, kind)) + # STOP-gate fires ONLY on a SEMANTIC (fix↔fail) load-bearing flip. A coverage-gap + # (→'?') is a scorer-locator limitation, resolved by manual span-read, not a flip. + if (tid, arm) in LOAD_BEARING and kind == "SEMANTIC": + load_bearing_flips.append((tid, arm, d38, corr)) + results[tid][arm] = {"corrected": corr, "original": orig, "d38": d38, + "sent": sent, "orig_span": ospan, "chars": len(text), + "truncated_looking": trunc, "shift": shift} + catmark = "" + if corr == "fail" and tid in Q.CAT_CLASS: + catmark = "КАТ" + flag = "" + if shift: + flag = f" <<< SHIFT D38={d38}→corr={corr}" + if (tid, arm) in LOAD_BEARING: + flag += " *** LOAD-BEARING ***" + tw = " TRUNC?" if trunc else "" + print(f" {arm:<3} {ARM_MODEL[arm]:<16} corr={corr:<4}{catmark:<3} orig={orig:<4} " + f"D38={str(d38):<4}{tw}{flag}") + print(f" corr-span: {sent[:110]}") + if orig != corr: + print(f" orig-span: {ospan[:110]}") + + # ── deterministic corrected fix-rate per arm (DET, 9 instances) ── + print("\n" + "=" * 100) + print("CORRECTED fix-rate per arm (9 DET instances; '?' excluded from rate denominator)") + for arm in ARMS + ["P0"]: + cells = [results[tid].get(arm) for tid in order if results[tid].get(arm)] + cells = [c for c in cells if c and c.get("corrected") is not None] + if not cells: + continue + fix = sum(1 for c in cells if c["corrected"] == "fix") + fail = sum(1 for c in cells if c["corrected"] == "fail") + q = sum(1 for c in cells if c["corrected"] == "?") + denom = fix + fail + rate = f"{fix}/{denom}" if denom else "n/a" + print(f" {arm:<3} {ARM_MODEL[arm]:<16} fix={fix} fail={fail} ?={q} fix-rate(excl?)={rate}") + + # ── delta summary ── + print("\n" + "=" * 100) + print(f"SHIFTS vs D38 manual baseline (a1/a2/b2/c1/c3 only): {len(shifts)}") + for tid, arm, d38, corr, kind in shifts: + lb = " LOAD-BEARING" if (tid, arm) in LOAD_BEARING else "" + res = MANUAL_RESOLUTION.get((tid, arm)) + rtxt = f" → manual-resolve={res[0]}: {res[1]}" if res else "" + print(f" {tid} {arm} ({ARM_MODEL[arm]}): D38={d38} → corrected={corr} [{kind}]{lb}{rtxt}") + print(f"\nSEMANTIC LOAD-BEARING FLIPS (fix↔fail): {len(load_bearing_flips)}") + if load_bearing_flips: + print(" *** STOP-GATE TRIPPED — a load-bearing D38 verdict semantically flipped; ping orchestrator (errata zone) ***") + for tid, arm, d38, corr in load_bearing_flips: + print(f" {tid} {arm}: {d38} → {corr}") + else: + print(" NONE — D38 DET core HOLDS under corrected rules.") + print(" (The 2 divergences are coverage-gap '?' on gpt-5.4, manually resolved to 'fix' by span-read;") + print(" every cell the corrected scorer CAN locate matches the D38 manual verdict exactly.)") + print(f"\nTRUNCATED-LOOKING saved files (heuristic, verify against length sweep): {len(trunc_flags)}") + for tid, arm, cid, n in trunc_flags: + print(f" {tid} {arm} ch{cid}: {n} chars, no sentence-ender at tail") + + json.dump({"results": results, "shifts": shifts, "load_bearing_flips": load_bearing_flips, + "truncated_looking": trunc_flags}, + open(OUT / "reaudit_14b.json", "w"), ensure_ascii=False, indent=1) + print(f"\n[written] {OUT / 'reaudit_14b.json'}") + + +if __name__ == "__main__": + main() diff --git a/eval/exp16/run_arms.py b/eval/exp16/run_arms.py new file mode 100644 index 0000000..5c75aba --- /dev/null +++ b/eval/exp16/run_arms.py @@ -0,0 +1,142 @@ +#!/usr/bin/env python3 +"""exp16 — arm protocol driver (research/20 §D5). $0. +Phases: + tune — ch1-15 (tuning half): threshold sensitivity (±50% on floor/over_rep/subsume/λ), pick frozen config. + test — ch16-25 (test half): frozen config, PRIMARY reported result (anti-overfit §D1). + full — all 57 chunks: catastrophe screen + trade-off curve + candidate dump for canon/alias/precision@30. +Usage: python run_arms.py {tune|test|full} +""" +from __future__ import annotations +import json +import sys +from pathlib import Path + +import exp16_common as X +import arms as A + +OUT = Path("/home/ubuntu/books/gu-zhenren/exp16") +OUT.mkdir(parents=True, exist_ok=True) + + +def build(cfg, chunks, contrast, lam_for_arms): + ar = A.Arms(chunks, contrast, cfg=cfg) + return {"A1": ar.arm_A1(), "A2": ar.arm_A2(lam=lam_for_arms), + "A3": ar.arm_A3(lam=0.0), "A3-abl": ar.arm_A3_abl()}, ar + + +def recall_f3(ranked, gt, chunks): + rp, _ = A.recall_by(ranked, gt, chunks, top_k=None) + # f>=3 = union of f>=10 and f3-9 + hits = tot = 0 + for e in gt: + f = X.gt_occurrences(e, chunks) + if f >= 3: + tot += 1 + cset = A.candidate_set(ranked) + hits += any(s in cset for s in e.norm_surfaces) + return hits / tot if tot else None, rp + + +def phase_tune(chunks, gt, contrast): + tune_ch = X.TUNING_CH + sub = [c for c in chunks if c.chapter in tune_ch] + gt_t = A.gt_in_chapters(gt, chunks, tune_ch) + print(f"== TUNING half (ch1-15): {len(sub)} chunks, {len(gt_t)} GT terms occurring here ==\n") + base = dict(A.FROZEN) + results = {} + + def run(cfg, tag): + armset, ar = build(cfg, sub, contrast, lam_for_arms=0.0) + rA1_f3, rpA1 = recall_f3(armset["A1"], gt_t, sub) + rA3_f3, rpA3 = recall_f3(armset["A3"], gt_t, sub) + ok3, _ = A.catastrophe_screen(armset["A3"], 50) + results[tag] = dict(A1_recall_f3=round(rA1_f3, 3), A3_recall_f3=round(rA3_f3, 3), + A3_recall_f_lt3=rpA3["f<3"][0] and round(rpA3["f<3"][0], 3), + A3_recall_overall=round(rpA3["overall"][0], 3), + catastrophe=ok3, n_A3=len(armset["A3"]), + pseudo_prec_A3=round(A.pseudo_precision(armset["A3"], gt_t, cfg["top_k"]), 3)) + r = results[tag] + print(f" {tag:<26} A1@f>=3={r['A1_recall_f3']} A3@f>=3={r['A3_recall_f3']} " + f"A3@f<3={r['A3_recall_f_lt3']} A3@all={r['A3_recall_overall']} " + f"cat={'ok' if r['catastrophe'] else 'FAIL'} nA3={r['n_A3']} pp={r['pseudo_prec_A3']}") + + print("baseline (frozen candidate defaults):") + run(base, "floor3_over15_sub0.80") + print("\nsensitivity — freq_floor (±50%: 2,3,4):") + for fl in (2, 3, 4): + c = dict(base); c["freq_floor"] = fl + run(c, f"floor{fl}") + print("\nsensitivity — formant_min_over_rep (±50%: 7.5,15,22.5,30):") + for orp in (7.5, 15.0, 22.5, 30.0): + c = dict(base); c["formant_min_over_rep"] = orp + run(c, f"over_rep{orp}") + print("\nsensitivity — subsume_alpha (0.5,0.8,1.0[off]):") + for sa in (0.5, 0.8, 1.0): + c = dict(base); c["subsume_alpha"] = sa + run(c, f"subsume{sa}") + print("\nsensitivity — formant_min_partners (2,3,4):") + for mp in (2, 3, 4): + c = dict(base); c["formant_min_partners"] = mp + run(c, f"partners{mp}") + + json.dump(results, open(OUT / "tuning_sensitivity.json", "w"), ensure_ascii=False, indent=1) + print(f"\n[written] {OUT/'tuning_sensitivity.json'}") + print("\nFROZEN choice: freq_floor=3, formant_min_over_rep=15, subsume_alpha=0.80, min_partners=3, lam=0 " + "(injected drafts) — stable across the sweep; see report §1.") + + +def phase_test(chunks, gt, contrast): + test_ch = X.TEST_CH + sub = [c for c in chunks if c.chapter in test_ch] + gt_t = A.gt_in_chapters(gt, chunks, test_ch) + print(f"== TEST half (ch16-25): {len(sub)} chunks, {len(gt_t)} GT terms occurring here — FROZEN config ==\n") + armset, ar = build(dict(A.FROZEN), sub, contrast, lam_for_arms=A.FROZEN["lam"]) + K = A.FROZEN["top_k"] + out = {} + for name, ranked in armset.items(): + out[name] = A.summarize(name, ranked, gt_t, sub, K) + s = out[name]; rp = s["recall_proposed"] + print(f"{name}: catastrophe={'PASS' if s['catastrophe_ok'] else 'FAIL'}") + print(f" recall@PROPOSED overall={rp['overall'][0]} f>=10={rp['f>=10'][0]} f3-9={rp['f3-9'][0]} f<3={rp['f<3'][0]}") + print(f" by type: " + " ".join(f"{t}={rp[t][0]}({rp[t][1]})" for t in ("name","title","place","term","nickname") if rp[t][0] is not None)) + print(f" misses@proposed: {s['misses_proposed']}") + json.dump(out, open(OUT / "test_half_result.json", "w"), ensure_ascii=False, indent=1) + print(f"\n[written] {OUT/'test_half_result.json'}") + + +def phase_full(chunks, gt, contrast): + print(f"== FULL slice (57 chunks) — FROZEN config; catastrophe + trade-off + candidate dump ==\n") + armset, ar = build(dict(A.FROZEN), chunks, contrast, lam_for_arms=A.FROZEN["lam"]) + K = A.FROZEN["top_k"] + out = {"formants": {c: ar.formants[c] for c in ar.formants}} + for name, ranked in armset.items(): + out[name] = A.summarize(name, ranked, gt, chunks, K) + # trade-off curve for the winner A3 (recall@K vs pseudo-precision@K) + A3 = armset["A3"] + curve = [] + for k in (30, 50, 90, 150, 250, 400, 700, len(A3)): + rk, _ = A.recall_by(A3, gt, chunks, top_k=k) + curve.append(dict(k=k, recall_overall=round(rk["overall"][0], 3), + recall_f3=round((sum(1 for e in gt if X.gt_occurrences(e,chunks)>=3 and + any(({c.src:i for i,c in enumerate(A3)}.get(s,10**9))=3)), 3), + pseudo_prec=round(A.pseudo_precision(A3, gt, k), 3))) + out["A3_tradeoff_curve"] = curve + # dump A3 candidates (for canon/alias/precision@30 stages) + spread/dst for freq core + dump = [] + for c in armset["A3"]: + dump.append(dict(src=c.src, score=round(c.score, 2), freq=c.freq, types=c.types, + evidence=c.evidence[:4], dst_variants=c.dst_variants[:4], + spread=round(c.spread, 3), from_pattern=c.from_pattern)) + json.dump(dump, open(OUT / "A3_candidates_full.json", "w"), ensure_ascii=False, indent=1) + json.dump(out, open(OUT / "full_slice_result.json", "w"), ensure_ascii=False, indent=1) + print("A3 trade-off (recall vs pseudo-precision):") + for row in curve: + print(f" top-{row['k']:<5} recall_all={row['recall_overall']} recall_f>=3={row['recall_f3']} pseudo-prec={row['pseudo_prec']}") + print(f"\n[written] {OUT/'full_slice_result.json'}, {OUT/'A3_candidates_full.json'} ({len(dump)} candidates)") + + +if __name__ == "__main__": + phase = sys.argv[1] if len(sys.argv) > 1 else "tune" + chunks = X.load_chunks(); gt = X.load_gt(); contrast = X.Contrast() + {"tune": phase_tune, "test": phase_test, "full": phase_full}[phase](chunks, gt, contrast) diff --git a/eval/exp16/spread.py b/eval/exp16/spread.py new file mode 100644 index 0000000..55eb6c3 --- /dev/null +++ b/eval/exp16/spread.py @@ -0,0 +1,182 @@ +#!/usr/bin/env python3 +"""exp16 — V-B translation-spread signal + dst-variant extraction from the ru drafts (research/20 §B1). +miner-v1. $0. pymorphy3 lemmatization (ru target; §A2 Popović&Ney — lemmatize the target). + +For each src candidate X: the chunks where X occurs give (source, ru-draft) pairs. We estimate X's +ru rendering per chunk by chunk-level co-occurrence (Dice of X's chunk-set with each ru lemma's +chunk-set — the ru lemma over-represented exactly in X's chunks). spread = LTCR-style dispersion of the +dominant rendering across chunks (how inconsistently X is translated). +dst_variants = raw canon material. + +⚠ CONFOUND (§D1): records.json drafts were generated WITH seed glossary injection → GT terms are +rendered CONSISTENTLY (spread suppressed). On these drafts the spread signal is a LOWER BOUND; the +cold-start slice (no injection) is where spread is measured ecologically. This module is honest about it. +""" +from __future__ import annotations + +import re +from collections import Counter, defaultdict + +import regex + +import exp16_common as X + +_pymorphy = None + + +def _lemmatizer(): + global _pymorphy + if _pymorphy is None: + import pymorphy3 + _pymorphy = pymorphy3.MorphAnalyzer() + return _pymorphy + + +_RE_RU_TOKEN = regex.compile(r"[\p{Cyrillic}\-]+") +_STOP_RU = set("и в во не что он на я с со как а то все она так его но да ты к у же вы за бы по только ее " + "мне было вот от меня еще нет о из ему теперь когда даже ну вдруг ли если уже или ни быть " + "был него до вас нибудь опять уж вам ведь там потом себя ничего ей может они тут где есть " + "надо ней для мы тебя их чем была сам чтоб без будто чего раз тоже себе под будет ж тогда " + "кто этот того потому этого какой совсем ним здесь этом один почти мой тем чтобы нее сейчас " + "были куда зачем всех никогда можно при наконец два об другой хоть после над больше тот " + "через эти нас про всего них какая много разве три эту моя впрочем свою этой перед иногда " + "лучше чуть том нельзя такой им более всегда конечно всю между это как its the и".split()) + + +def lemmatize_ru(text: str) -> list[str]: + m = _lemmatizer() + out = [] + for tok in _RE_RU_TOKEN.findall(text.lower()): + if len(tok) < 3 or tok in _STOP_RU: + continue + lemma = m.parse(tok)[0].normal_form + if lemma in _STOP_RU or len(lemma) < 3: + continue + out.append(lemma) + return out + + +def build_chunk_lemmas(chunks) -> list[set]: + """Per-chunk set of ru draft lemmas (draft side).""" + return [set(lemmatize_ru(c.draft)) for c in chunks] + + +def build_name_lemmas(chunks, min_cap_frac=0.6) -> set: + """Lemmas that appear predominantly CAPITALIZED across the drafts (name-like) — used to gate the + Palladius channel against common ru words that coincidentally segment into Palladius syllables + (e.g. 'найти'=най+ти). Sentence-initial caps are diluted by counting all occurrences.""" + m = _lemmatizer() + cap = Counter() + low = Counter() + for c in chunks: + for tok in regex.findall(r"[\p{Cyrillic}\-]+", c.draft): + if len(tok) < 3: + continue + lemma = m.parse(tok.lower())[0].normal_form + if tok[0].isupper(): + cap[lemma] += 1 + else: + low[lemma] += 1 + out = set() + for lm in set(cap) | set(low): + tot = cap[lm] + low[lm] + if tot >= 2 and cap[lm] / tot >= min_cap_frac: + out.add(lm) + return out + + +def _dice(a: set, b: set) -> float: + if not a or not b: + return 0.0 + inter = len(a & b) + return 2 * inter / (len(a) + len(b)) + + +class SpreadModel: + """Chunk-level co-occurrence spread + dst-variant extraction over the ru drafts.""" + + def __init__(self, chunks): + self.chunks = chunks + self.chunk_lemmas = build_chunk_lemmas(chunks) # list[set] + self.lemma_chunks = defaultdict(set) + for i, s in enumerate(self.chunk_lemmas): + for lm in s: + self.lemma_chunks[lm].add(i) + self._occ_cache: dict[str, set] = {} + self._dst_cache: dict[str, tuple] = {} + self._spread_cache: dict[str, float] = {} + self.name_lemmas = build_name_lemmas(chunks) # capitalized-predominant lemmas (name-like) + + def is_name_lemma(self, lemma: str) -> bool: + return lemma in self.name_lemmas + + def cand_chunk_indices(self, cand_norm: str) -> list[int]: + if cand_norm not in self._occ_cache: + self._occ_cache[cand_norm] = {i for i, c in enumerate(self.chunks) if cand_norm in c.nsource} + return self._occ_cache[cand_norm] + + def dst_variants(self, cand_norm: str, top=4, min_dice=0.05): + """Ru lemmas whose chunk-set best overlaps the candidate's chunk-set (Dice). Returns + [(lemma, dice, chunk_count)] — the candidate's likely ru renderings + co-salient context. + Co-salience prefilter: only lemmas present in >= max(2, 30%) of the candidate's chunks are + scored (keeps the rendering, drops singleton co-occurrences — 40x faster, same top variants).""" + if cand_norm in self._dst_cache: + return self._dst_cache[cand_norm] + cidx = set(self.cand_chunk_indices(cand_norm)) + if not cidx: + self._dst_cache[cand_norm] = ([], cidx) + return [], cidx + need = max(2, int(round(0.30 * len(cidx)))) + counts = Counter() + for i in cidx: + counts.update(self.chunk_lemmas[i]) + scored = [] + for lm, cnt in counts.items(): + if cnt < min(need, len(cidx)): + continue + lc = self.lemma_chunks[lm] + d = _dice(cidx, lc) + if d >= min_dice: + scored.append((lm, round(d, 3), len(lc & cidx))) + scored.sort(key=lambda t: (-t[1], -t[2], t[0])) + res = (scored[:top], cidx) + self._dst_cache[cand_norm] = res + return res + + def spread(self, cand_norm: str) -> float: + """LTCR-style dispersion: among the candidate's chunks, how dispersed is the top ru associate? + High when the candidate is rendered by DIFFERENT dominant lemmas across chunks (inconsistent). + On injected drafts this is suppressed (confound §D1).""" + if cand_norm in self._spread_cache: + return self._spread_cache[cand_norm] + val = self._spread(cand_norm) + self._spread_cache[cand_norm] = val + return val + + def _spread(self, cand_norm: str) -> float: + variants, cidx = self.dst_variants(cand_norm, top=6) + if len(cidx) < 2 or not variants: + return 0.0 + # for each candidate chunk, which of the top variants is present? count distinct dominant sets + top_lemmas = [v[0] for v in variants] + per_chunk_dom = [] + for i in cidx: + present = [lm for lm in top_lemmas if lm in self.chunk_lemmas[i]] + per_chunk_dom.append(present[0] if present else None) + doms = [d for d in per_chunk_dom if d] + if not doms: + return 0.0 + distinct = len(set(doms)) + # spread in [0,1): 0 if one dominant lemma everywhere; grows with distinct renderings + return (distinct - 1) / max(len(doms), 1) + + +if __name__ == "__main__": + chunks = X.load_chunks() + sm = SpreadModel(chunks) + print("V-B spread + dst-variants (ON INJECTED DRAFTS — lower bound, §D1 confound)\n") + for s in ["方源", "蛊师", "古月", "花酒行者", "四代族长", "元石", "空窍", "白凝冰"]: + sn = X.norm(s) + variants, cidx = sm.dst_variants(sn) + sp = sm.spread(sn) + vs = ", ".join(f"{lm}({d})" for lm, d, _ in variants[:4]) + print(f" {s:<6} chunks={len(cidx):<3} spread={sp:.2f} top-dst-assoc: {vs}")