textmachine/eval/exp16/spread.py

182 lines
8.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""exp16 — V-B translation-spread signal + dst-variant extraction from the ru drafts (research/20 §B1).
miner-v1. $0. pymorphy3 lemmatization (ru target; §A2 Popović&Ney — lemmatize the target).
For each src candidate X: the chunks where X occurs give (source, ru-draft) pairs. We estimate X's
ru rendering per chunk by chunk-level co-occurrence (Dice of X's chunk-set with each ru lemma's
chunk-set — the ru lemma over-represented exactly in X's chunks). spread = LTCR-style dispersion of the
dominant rendering across chunks (how inconsistently X is translated). +dst_variants = raw canon material.
⚠ CONFOUND (§D1): records.json drafts were generated WITH seed glossary injection → GT terms are
rendered CONSISTENTLY (spread suppressed). On these drafts the spread signal is a LOWER BOUND; the
cold-start slice (no injection) is where spread is measured ecologically. This module is honest about it.
"""
from __future__ import annotations
import re
from collections import Counter, defaultdict
import regex
import exp16_common as X
_pymorphy = None
def _lemmatizer():
global _pymorphy
if _pymorphy is None:
import pymorphy3
_pymorphy = pymorphy3.MorphAnalyzer()
return _pymorphy
_RE_RU_TOKEN = regex.compile(r"[\p{Cyrillic}\-]+")
_STOP_RU = set("и в во не что он на я с со как а то все она так его но да ты к у же вы за бы по только ее "
"мне было вот от меня еще нет о из ему теперь когда даже ну вдруг ли если уже или ни быть "
"был него до вас нибудь опять уж вам ведь там потом себя ничего ей может они тут где есть "
"надо ней для мы тебя их чем была сам чтоб без будто чего раз тоже себе под будет ж тогда "
"кто этот того потому этого какой совсем ним здесь этом один почти мой тем чтобы нее сейчас "
"были куда зачем всех никогда можно при наконец два об другой хоть после над больше тот "
"через эти нас про всего них какая много разве три эту моя впрочем свою этой перед иногда "
"лучше чуть том нельзя такой им более всегда конечно всю между это как its the и".split())
def lemmatize_ru(text: str) -> list[str]:
m = _lemmatizer()
out = []
for tok in _RE_RU_TOKEN.findall(text.lower()):
if len(tok) < 3 or tok in _STOP_RU:
continue
lemma = m.parse(tok)[0].normal_form
if lemma in _STOP_RU or len(lemma) < 3:
continue
out.append(lemma)
return out
def build_chunk_lemmas(chunks) -> list[set]:
"""Per-chunk set of ru draft lemmas (draft side)."""
return [set(lemmatize_ru(c.draft)) for c in chunks]
def build_name_lemmas(chunks, min_cap_frac=0.6) -> set:
"""Lemmas that appear predominantly CAPITALIZED across the drafts (name-like) — used to gate the
Palladius channel against common ru words that coincidentally segment into Palladius syllables
(e.g. 'найти'=най+ти). Sentence-initial caps are diluted by counting all occurrences."""
m = _lemmatizer()
cap = Counter()
low = Counter()
for c in chunks:
for tok in regex.findall(r"[\p{Cyrillic}\-]+", c.draft):
if len(tok) < 3:
continue
lemma = m.parse(tok.lower())[0].normal_form
if tok[0].isupper():
cap[lemma] += 1
else:
low[lemma] += 1
out = set()
for lm in set(cap) | set(low):
tot = cap[lm] + low[lm]
if tot >= 2 and cap[lm] / tot >= min_cap_frac:
out.add(lm)
return out
def _dice(a: set, b: set) -> float:
if not a or not b:
return 0.0
inter = len(a & b)
return 2 * inter / (len(a) + len(b))
class SpreadModel:
"""Chunk-level co-occurrence spread + dst-variant extraction over the ru drafts."""
def __init__(self, chunks):
self.chunks = chunks
self.chunk_lemmas = build_chunk_lemmas(chunks) # list[set]
self.lemma_chunks = defaultdict(set)
for i, s in enumerate(self.chunk_lemmas):
for lm in s:
self.lemma_chunks[lm].add(i)
self._occ_cache: dict[str, set] = {}
self._dst_cache: dict[str, tuple] = {}
self._spread_cache: dict[str, float] = {}
self.name_lemmas = build_name_lemmas(chunks) # capitalized-predominant lemmas (name-like)
def is_name_lemma(self, lemma: str) -> bool:
return lemma in self.name_lemmas
def cand_chunk_indices(self, cand_norm: str) -> list[int]:
if cand_norm not in self._occ_cache:
self._occ_cache[cand_norm] = {i for i, c in enumerate(self.chunks) if cand_norm in c.nsource}
return self._occ_cache[cand_norm]
def dst_variants(self, cand_norm: str, top=4, min_dice=0.05):
"""Ru lemmas whose chunk-set best overlaps the candidate's chunk-set (Dice). Returns
[(lemma, dice, chunk_count)] — the candidate's likely ru renderings + co-salient context.
Co-salience prefilter: only lemmas present in >= max(2, 30%) of the candidate's chunks are
scored (keeps the rendering, drops singleton co-occurrences — 40x faster, same top variants)."""
if cand_norm in self._dst_cache:
return self._dst_cache[cand_norm]
cidx = set(self.cand_chunk_indices(cand_norm))
if not cidx:
self._dst_cache[cand_norm] = ([], cidx)
return [], cidx
need = max(2, int(round(0.30 * len(cidx))))
counts = Counter()
for i in cidx:
counts.update(self.chunk_lemmas[i])
scored = []
for lm, cnt in counts.items():
if cnt < min(need, len(cidx)):
continue
lc = self.lemma_chunks[lm]
d = _dice(cidx, lc)
if d >= min_dice:
scored.append((lm, round(d, 3), len(lc & cidx)))
scored.sort(key=lambda t: (-t[1], -t[2], t[0]))
res = (scored[:top], cidx)
self._dst_cache[cand_norm] = res
return res
def spread(self, cand_norm: str) -> float:
"""LTCR-style dispersion: among the candidate's chunks, how dispersed is the top ru associate?
High when the candidate is rendered by DIFFERENT dominant lemmas across chunks (inconsistent).
On injected drafts this is suppressed (confound §D1)."""
if cand_norm in self._spread_cache:
return self._spread_cache[cand_norm]
val = self._spread(cand_norm)
self._spread_cache[cand_norm] = val
return val
def _spread(self, cand_norm: str) -> float:
variants, cidx = self.dst_variants(cand_norm, top=6)
if len(cidx) < 2 or not variants:
return 0.0
# for each candidate chunk, which of the top variants is present? count distinct dominant sets
top_lemmas = [v[0] for v in variants]
per_chunk_dom = []
for i in cidx:
present = [lm for lm in top_lemmas if lm in self.chunk_lemmas[i]]
per_chunk_dom.append(present[0] if present else None)
doms = [d for d in per_chunk_dom if d]
if not doms:
return 0.0
distinct = len(set(doms))
# spread in [0,1): 0 if one dominant lemma everywhere; grows with distinct renderings
return (distinct - 1) / max(len(doms), 1)
if __name__ == "__main__":
chunks = X.load_chunks()
sm = SpreadModel(chunks)
print("V-B spread + dst-variants (ON INJECTED DRAFTS — lower bound, §D1 confound)\n")
for s in ["方源", "蛊师", "古月", "花酒行者", "四代族长", "元石", "空窍", "白凝冰"]:
sn = X.norm(s)
variants, cidx = sm.dst_variants(sn)
sp = sm.spread(sn)
vs = ", ".join(f"{lm}({d})" for lm, d, _ in variants[:4])
print(f" {s:<6} chunks={len(cidx):<3} spread={sp:.2f} top-dst-assoc: {vs}")