textmachine/eval/exp16/alias.py

188 lines
9.8 KiB
Python

#!/usr/bin/env python3
"""exp16 — alias tier-1 clustering (research/20 §B2). miner-v1. $0.
PROP-layer, precision-safe rules over a set of src surfaces + their dominant ru rendering:
R1 surface containment X ⊂ Y, |X|>=2 -> 'extension' edge (方源 ⊂ 古月方源)
R2 surname anchor+compose shared surname -> FAMILY supercluster edge (NOT identity)
R3 shared ru rendering same dominant ru lemma-set -> strong IDENTITY edge (古月方源,方源 -> Фан Юань)
R4 NEGATIVE constraints (block a merge):
(i) same surname + DIFFERENT given names (方源 vs 方正) -> not identity
(ii) different confirmed gender
(iii)different approved dst
(iv) co-presence in one source sentence/dialogue turn -> not the same entity
(v) title-nesting with different seed dst (族长 ⊂ 四代族长) -> different entities
Output: identity clusters (R1/R3 minus R4), family superclusters (R2), and tier-2 weak-link proposals.
Precision measured entity-level (B³ style, §A5 — NOT MUC link metric). Recall needs the owner mini-gold
(§E-1): this module also emits a candidate-edge sheet for the ~20-30 min owner touchpoint.
"""
from __future__ import annotations
import json
from collections import defaultdict
from dataclasses import dataclass, field
import exp16_common as X
import spread as SP
import palladius as PAL
SURNAMES = None # lazy import to avoid cycle
def _surname_of(src_norm: str):
import patterns as P
return P.is_surname_start(src_norm)
@dataclass
class Surface:
src: str # normalized src surface
dst_lemmas: list = field(default_factory=list) # dominant ru rendering lemma(s)
gender: str = ""
approved_dst: str = ""
typ: str = ""
def dominant_dst(sm: SP.SpreadModel, src_norm: str, top=3):
variants, _ = sm.dst_variants(src_norm, top=top)
# keep the ru lemmas that look like a NAME rendering (Palladius) OR the very top associates
return [lm for lm, d, _ in variants]
def cooccur_same_sentence(chunks, a_norm: str, b_norm: str) -> bool:
"""R4-iv: do a and b co-occur within one source sentence anywhere?"""
import regex
for c in chunks:
for sent in regex.split(r"[。!?\n]", c.nsource):
if a_norm in sent and b_norm in sent:
return True
return False
def build_surfaces(surface_srcs, sm, gt_by_norm) -> dict[str, Surface]:
out = {}
for s in surface_srcs:
sn = X.norm(s)
if not sn:
continue
ent = gt_by_norm.get(sn)
out[sn] = Surface(src=sn, dst_lemmas=dominant_dst(sm, sn),
gender=(ent.gender if ent else ""),
approved_dst=(ent.dst if ent and ent.status == "approved" else ""),
typ=(ent.typ if ent else ""))
return out
def propose_edges(surfaces: dict[str, Surface], chunks, sm, gt_norm_surfaces=frozenset()):
_SM = sm
ident, family, weak = [], [], []
keys = list(surfaces)
for i in range(len(keys)):
for j in range(i + 1, len(keys)):
a, b = surfaces[keys[i]], surfaces[keys[j]]
# ── R4 HARD BLOCKS (checked first; no exceptions) ──
# R4-ii different confirmed gender
if a.gender and b.gender and a.gender != b.gender and "hidden" not in (a.gender, b.gender):
continue
# R4-iii/v different approved dst -> different entities, EVEN IF one contains the other
# (this is the 族长⊂四代族长 lesson D38 §3 and the clan⊂person case 古月⊂古月方源).
if a.approved_dst and b.approved_dst and X.norm(a.approved_dst) != X.norm(b.approved_dst):
continue
sur_a, sur_b = _surname_of(a.src), _surname_of(b.src)
# R1 containment -> extension edge (a and b are the SAME entity, fuller vs shorter surface).
# Guard against boundary fragments: the SHORTER surface must itself be a GT surface or a
# multi-char pattern-typed entity (not a longer±particle artifact).
if len(a.src) >= 2 and len(b.src) >= 2 and (a.src in b.src or b.src in a.src) and a.src != b.src:
short, long_ = (a, b) if len(a.src) < len(b.src) else (b, a)
short_is_entity = bool(short.typ) or short.src in gt_norm_surfaces
# COMPOSITIONAL guard: long = short + Y (or Y + short) where Y is ANOTHER known entity of
# type title/place/term => long is a PHRASE (古月+族长 = clan-head phrase), not an alias.
# clan/surname + NAME stays a fullname alias (古月+方源). Blocks the 古月↔族长 chaining.
rest = long_.src[len(short.src):] if long_.src.startswith(short.src) else long_.src[:-len(short.src)]
rest_ent = surfaces.get(rest)
compositional = bool(rest) and rest in gt_norm_surfaces and \
(rest_ent is None or rest_ent.typ in ("title", "place", "term"))
if compositional:
weak.append(dict(a=a.src, b=b.src, rule="R1-compositional", kind="phrase_not_alias", part=rest))
elif short_is_entity:
ident.append(dict(a=a.src, b=b.src, rule="R1", kind="extension"))
else:
weak.append(dict(a=a.src, b=b.src, rule="R1-fragment?", kind="containment_weak"))
continue
# R2 surname anchor -> family (NOT identity)
if sur_a and sur_b and sur_a == sur_b and a.src != b.src:
# R4-i same surname + different given names -> NOT identity, family only
given_a, given_b = a.src[len(sur_a):], b.src[len(sur_b):]
if given_a != given_b and given_a and given_b:
if cooccur_same_sentence(chunks, a.src, b.src):
family.append(dict(a=a.src, b=b.src, rule="R2+R4iv", kind="family_copresent"))
else:
family.append(dict(a=a.src, b=b.src, rule="R2", kind="family"))
continue
# R3 shared ru rendering -> identity (strong), unless R4-iv co-presence.
# Palladius tokens must ALSO be capitalized-predominant name lemmas (blocks 'найти'=най+ти FP).
palla = [lm for lm in a.dst_lemmas[:2] if PAL.is_palladius_token(lm, 2) and _SM.is_name_lemma(lm)]
pallb = [lm for lm in b.dst_lemmas[:2] if PAL.is_palladius_token(lm, 2) and _SM.is_name_lemma(lm)]
if palla and pallb and set(palla) & set(pallb):
if cooccur_same_sentence(chunks, a.src, b.src):
weak.append(dict(a=a.src, b=b.src, rule="R3-blockedR4iv", kind="shared_render_copresent"))
else:
ident.append(dict(a=a.src, b=b.src, rule="R3", kind="shared_render", render=list(set(palla) & set(pallb))))
return ident, family, weak
def cluster(ident_edges, all_surfaces):
"""Union-find over identity edges -> entity clusters."""
parent = {s: s for s in all_surfaces}
def find(x):
while parent[x] != x:
parent[x] = parent[parent[x]]; x = parent[x]
return x
for e in ident_edges:
if e["a"] in parent and e["b"] in parent:
parent[find(e["a"])] = find(e["b"])
clusters = defaultdict(list)
for s in all_surfaces:
clusters[find(s)].append(s)
return [sorted(v) for v in clusters.values() if len(v) > 1]
if __name__ == "__main__":
chunks = X.load_chunks(); gt = X.load_gt()
sm = SP.SpreadModel(chunks)
gt_by_norm = {X.norm(s): e for e in gt for s in e.surfaces}
# surface set: GT surfaces (measure precision vs the 2 known aliases) + high-conf name candidates
import arms as A
ar = A.Arms(chunks, X.Contrast())
a3 = ar.arm_A3()
# input = GT surfaces + genuine name/place/title candidates (freq>=5, typed, NOT boundary fragments).
# Exclude GT-surface+trailing-particle artifacts (方源在/方源心 from the surname window).
subsumed = ar.va.subsumed
gt_surf_norm = {X.norm(s) for e in gt for s in e.surfaces}
PARTICLE = set("的了在是和就也都不心面前后上下今少多大小和与之其")
def frag(src):
return len(src) >= 2 and src[:-1] in gt_surf_norm and src[-1] in PARTICLE
name_cands = [c.src for c in a3[:200]
if any(t in c.types for t in ("name", "place", "title"))
and c.freq >= 5 and c.src not in subsumed and len(c.src) >= 2 and not frag(c.src)]
gt_norm_surfaces = frozenset(X.norm(s) for e in gt for s in e.surfaces)
surface_srcs = list(gt_norm_surfaces | set(name_cands))
surfaces = build_surfaces(surface_srcs, sm, gt_by_norm)
ident, family, weak = propose_edges(surfaces, chunks, sm, gt_norm_surfaces)
clusters = cluster(ident, list(surfaces))
print(f"surfaces: {len(surfaces)} identity-edges: {len(ident)} family-edges: {len(family)} weak: {len(weak)}")
print("\n== identity edges (R1/R3) ==")
for e in ident:
print(f" {e['a']} = {e['b']} [{e['rule']}/{e['kind']}] {e.get('render','')}")
print("\n== family superclusters (R2, NOT identity) ==")
for e in family[:20]:
print(f" {e['a']} ~ {e['b']} [{e['rule']}/{e['kind']}]")
print(f"\n== identity clusters (>1) ==")
for cl in clusters:
print(f" {{ {', '.join(cl)} }}")
# precision vs known seed aliases (古月方源=方源, 古月方正=方正)
known = {frozenset([X.norm("古月方源"), X.norm("方源")]), frozenset([X.norm("古月方正"), X.norm("方正")])}
id_pairs = {frozenset([e["a"], e["b"]]) for e in ident if e["kind"] in ("extension", "shared_render")}
tp = len(id_pairs & known)
print(f"\nknown-alias recall: {tp}/{len(known)}; identity-edge count {len(id_pairs)} "
f"(precision vs seed not meaningful — seed alias set is thin, needs owner mini-gold §E-1)")