textmachine/eval/exp16/patterns.py

236 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""exp16 — V-C pattern channels (research/20 §B1 channels 14/6). miner-v1. $0.
Language×genre pattern pack for zh (§B5 plugin P3). Versioned data (surnames / title affixes / topo
suffixes) + a book-adaptive productive-morphology detector (channel 4 — auto-detects the domain formant,
NOT hardcoded 蛊). Each channel proposes TYPED candidates (name/title/place/term) with evidence, closing
V-A's frequency-blind classes (rare surname-anchored names, rank/grade titles, one-off realia).
Channels:
(1) surname anchor 百家姓 + compound surnames, + 12 Han window right -> name
(2) title affixes suffix 公子/大人/长老/前辈/族长/嬷嬷…, prefix 老/小/阿 + ordinals -> title/name
(3) topo suffixes 山/寨/村/疆/谷/城/门/宗… -> place
(4) productive morphology char that binds with >=m distinct n-grams (蛊/等/转…) -> term/title (auto)
(6) genre lexicon pack xianxia suffix/title lexicon (this file IS the zh-xianxia pack)
Owner decision 18.07: genre PACKS are NOT built (deferred to book 2); channel (6) runs only on UNIVERSAL
channels — 百家姓 anchor, title/topo suffixes, productive morphology (auto-detected formant, not hardcoded
蛊), Palladius on the ru side. This module encodes exactly that universal set.
"""
from __future__ import annotations
from collections import Counter, defaultdict
import exp16_common as X
# ── (1) surnames: 百家姓 single-char subset + compound surnames (versioned inventory) ───────────────
# Standard 百家姓 opening + the surnames present in this book's cast. Compound surnames listed explicitly.
SURNAMES_SINGLE = set(
"赵钱孙李周吴郑王冯陈褚卫蒋沈韩杨朱秦尤许何吕施张孔曹严华金魏陶姜"
"戚谢邹喻柏水窦章云苏潘葛奚范彭郎鲁韦昌马苗凤花方俞任袁柳酆鲍史唐"
"费廉岑薛雷贺倪汤滕殷罗毕郝邬安常乐于时傅皮卞齐康伍余元卜顾孟平黄"
"和穆萧尹姚邵湛汪祁毛禹狄米贝明臧计伏成戴谈宋茅庞熊纪舒屈项祝董梁"
"杜阮蓝闵席季麻强贾路娄危江童颜郭梅盛林刁钟徐邱骆高夏蔡田樊胡凌霍"
"虞万支柯昝管卢莫经房裘缪干解应宗丁宣贲邓郁单杭洪包诸左石崔吉钮龚"
"白凝" # 白 (Bai clan). 凝 is not a surname but kept out — handled below via compound guard.
)
SURNAMES_SINGLE.discard("")
SURNAMES_COMPOUND = {"古月", "欧阳", "司马", "上官", "夏侯", "诸葛", "东方", "皇甫", "尉迟", "公孙",
"慕容", "长孙", "宇文", "司徒", "鲜于", "南宫"}
# ── (2) title affixes (universal honorific/rank suffixes+prefixes; §A4 Cao) ─────────────────────────
TITLE_SUFFIX = ["公子", "大人", "长老", "前辈", "族长", "家老", "老祖", "祖师", "真人", "上人",
"道人", "先生", "夫人", "娘子", "姑娘", "嬷嬷", "师傅", "师父", "掌门", "宗主",
"少爷", "老爷", "小姐", "婆婆", "大娘", "大爷"]
TITLE_PREFIX = ["", "", ""]
ORDINAL = ["一代", "二代", "三代", "四代", "五代", "六代", "七代", "八代", "九代", "十代",
"第一", "第二", "第三", "第四", "第五"]
# ── (3) topo suffixes -> place ─────────────────────────────────────────────────────────────────────
TOPO_SUFFIX = ["", "", "", "", "", "", "", "", "", "", "", "", "",
"", "", "", "", "", "", "", "殿", "", "", "", "", ""]
# ── (2c) rank/grade compositional pattern (numeral/sequential + rank-word) -> title ────────────────
# Grade/rank words are COMMON chars (等 over_rep 1.1, 转 3.9) — not formant-detectable; caught by the
# sequential-prefix + rank-word composition instead. Universal (numeral+rank), not genre-specific.
GRADE_PREFIX = list("甲乙丙丁戊己庚辛壬癸") # sequential grade markers
NUMERAL = list("一二三四五六七八九十零百千") # Chinese numerals
RANK_WORD = ["", "", "", "", "", "", "", ""] # rank/grade head words
PACK_VERSION = "zh-universal-v1"
def is_surname_start(s: str):
"""Return the surname prefix (compound preferred) if s starts with one, else None."""
for cs in SURNAMES_COMPOUND:
if s.startswith(cs):
return cs
if s and s[0] in SURNAMES_SINGLE:
return s[0]
return None
# ── (4) productive-morphology (formant) auto-detection ─────────────────────────────────────────────
def book_char_freq(chunks) -> Counter:
bc = Counter()
for c in chunks:
for ch in c.nsource:
if X._RE_HAN.match(ch):
bc[ch] += 1
return bc
def detect_formants(chunks, cand_freq: dict[str, int], contrast: X.Contrast, min_partners=3,
min_over_rep=15.0) -> dict[str, dict]:
"""A Han char c is a productive DOMAIN formant if it binds (suffix OR prefix) with >= min_partners
DISTINCT content morphemes among candidates AND is over-represented in-book vs general zh
(p_book/p_general >= min_over_rep). Over-representation — NOT char-rarity — is the domain signal:
it isolates 蛊(2723×)/窍(88×)/虫(32×) while rejecting common chars 师/花/等(1.1×). Auto-detected,
not hardcoded (§B1 channel 4). Common rank-words (等/转) are handled by the compositional channel."""
bc = book_char_freq(chunks)
tot_book = sum(bc.values()) or 1
suf = defaultdict(set)
pre = defaultdict(set)
for a in cand_freq:
if len(a) < 2:
continue
suf[a[-1]].add(a[:-1])
pre[a[0]].add(a[1:])
formants = {}
for c in set(suf) | set(pre):
p_book = bc.get(c, 0) / tot_book
p_gen = (contrast.char_freq.get(c, 0) + 1) / (contrast.total_char + len(contrast.char_freq))
over_rep = p_book / p_gen if p_gen else 0.0
if over_rep < min_over_rep:
continue
ns, npr = len(suf.get(c, ())), len(pre.get(c, ()))
role = None
if ns >= min_partners:
role = "suffix"
if npr >= min_partners:
role = "both" if role else "prefix"
if role:
formants[c] = {"role": role, "over_rep": round(over_rep, 1),
"suffix_partners": ns, "prefix_partners": npr}
return formants
# formant -> candidate type (universal heuristic; not genre-conditioned)
def formant_type(c: str) -> str:
if c in TOPO_SUFFIX:
return "place"
if c in ("", "", ""):
return "title"
return "term"
# ── channel application: produce typed pattern candidates from the source ──────────────────────────
def pattern_candidates(chunks, cand_freq: dict[str, int], contrast: X.Contrast,
min_partners=3, min_over_rep=15.0):
"""Return {cand_norm: {'types': [...], 'evidence': [...]}} for pattern-detected candidates.
Operates over the ALREADY-normalized source; proposes candidates that may be BELOW the V-A freq
floor (that is the point — patterns close the rare-term blind spot)."""
out = defaultdict(lambda: {"types": [], "evidence": []})
def add(cand, typ, ev):
cand = X.norm(cand)
if not cand or not X.han_only(cand):
return
if typ not in out[cand]["types"]:
out[cand]["types"].append(typ)
if ev not in out[cand]["evidence"]:
out[cand]["evidence"].append(ev)
# scan each source for surname / title / topo patterns over Han runs
for c in chunks:
for run in X.han_runs(c.nsource):
L = len(run)
for i in range(L):
# (1) surname anchor: surname + 12 Han given-name window
sn = is_surname_start(run[i:])
if sn:
base = i + len(sn)
for gl in (1, 2):
if base + gl <= L:
full = run[i:base + gl]
if 2 <= len(full) <= 4:
add(full, "name", f"surname:{sn}")
# (2) title suffix: content + suffix
for suf in TITLE_SUFFIX:
if run.startswith(suf, i):
# take up to 3 Han to the LEFT as the titled base (e.g. 沈+嬷嬷, 四代+族长)
for left in (3, 2, 1, 0):
if i - left >= 0:
cand = run[i - left:i + len(suf)]
if 2 <= len(cand) <= 6:
add(cand, "title", f"title_suffix:{suf}")
add(suf, "title", f"title_bare:{suf}")
# (2b) ordinal + title (四代族长-class)
for od in ORDINAL:
if run.startswith(od, i):
for suf in TITLE_SUFFIX:
end = i + len(od)
if run.startswith(suf, end):
add(run[i:end + len(suf)], "title", f"ordinal_title:{od}+{suf}")
# (2c) rank/grade compositional: (numeral|grade-prefix) + rank-word -> title
for rw in RANK_WORD:
if run.startswith(rw, i) and i >= 1:
left = run[i - 1]
if left in NUMERAL or left in GRADE_PREFIX:
add(run[i - 1:i + len(rw)], "title", f"rank_grade:{left}+{rw}")
# (3) topo suffix: 13 Han base + topo char
if run[i] in TOPO_SUFFIX and i >= 1:
for left in (3, 2, 1):
if i - left >= 0:
cand = run[i - left:i + 1]
if 2 <= len(cand) <= 4:
add(cand, "place", f"topo_suffix:{run[i]}")
# (4) productive morphology: for each detected formant, propose all binding n-grams
formants = detect_formants(chunks, cand_freq, contrast, min_partners, min_over_rep)
for c in chunks:
for run in X.han_runs(c.nsource):
L = len(run)
for i in range(L):
if run[i] not in formants:
continue
info = formants[run[i]]
typ = formant_type(run[i])
if info["role"] in ("suffix", "both"):
for left in (3, 2, 1):
if i - left >= 0:
cand = run[i - left:i + 1]
if 2 <= len(cand) <= 4:
add(cand, typ, f"formant_suffix:{run[i]}")
if info["role"] in ("prefix", "both"):
for r in (2, 3):
if i + r <= L:
cand = run[i:i + r]
if 2 <= len(cand) <= 4:
add(cand, typ, f"formant_prefix:{run[i]}")
return dict(out), formants
if __name__ == "__main__":
chunks = X.load_chunks()
gt = X.load_gt()
C = X.Contrast()
import detectors as D
va = D.VA(chunks, C).build()
pats, formants = pattern_candidates(chunks, va.cand_freq, C)
print(f"pattern candidates: {len(pats)}")
print(f"\ndetected formants ({len(formants)}):")
for c, info in sorted(formants.items(), key=lambda kv: -kv[1]["over_rep"]):
print(f" {c} {info}")
# how many f>=3-miss GT terms does the pattern layer now propose?
gtsurf = {X.norm(s): e for e in gt for s in e.surfaces}
hit = [s for s in gtsurf if s in pats]
print(f"\nGT surfaces proposed by patterns: {len(hit)}/{len(gtsurf)}")
# specifically the V-A f>=3 misses
focus = ["", "一转", "三转", "五转", "六转", "甲等", "乙等", "丙等", "丁等", "白凝冰", "江牙",
"沈嬷嬷", "四代族长", "族长", "青茅山", "古月山寨", "希望蛊", "月影蛊", "月光蛊", "人祖"]
print("\nfocus (V-A structural misses) — proposed by patterns?")
for s in focus:
sn = X.norm(s)
info = pats.get(sn)
print(f" {s:<6} {'YES ' + str(info['types']) + ' ' + str(info['evidence'][:2]) if info else 'no'}")