textmachine/backend/internal/pipeline/miner_patterns.go

354 lines
11 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

package pipeline
// miner_patterns.go: the V-C pattern channels (WS3 — a Go port of exp16 patterns.py). A language×genre
// pattern pack for zh (§B5 plugin P3, universal set only — owner decision 18.07: genre packs are NOT
// built). Each channel proposes TYPED candidates (name/title/place/term) with evidence, closing V-A's
// frequency-blind classes (rare surname-anchored names, rank/grade titles, one-off realia). Versioned
// data (surnames / title affixes / topo suffixes) + a book-adaptive productive-morphology detector
// (channel 4 — auto-detects the domain formant, NOT hardcoded 蛊). Pure and deterministic.
// minerPackVersion is the pattern-inventory version (patterns.PACK_VERSION).
const minerPackVersion = "zh-universal-v1"
// surnamesSingle is the 百家姓 single-char subset + this book's cast (patterns.SURNAMES_SINGLE, with 凝
// discarded — it is not a surname). surnamesCompound are the explicit two-char compound surnames.
var (
surnamesSingle = buildRuneSet(surnamesSingleRaw)
surnamesCompound = map[string]bool{
"古月": true, "欧阳": true, "司马": true, "上官": true, "夏侯": true, "诸葛": true, "东方": true,
"皇甫": true, "尉迟": true, "公孙": true, "慕容": true, "长孙": true, "宇文": true, "司徒": true,
"鲜于": true, "南宫": true,
}
)
const surnamesSingleRaw = "赵钱孙李周吴郑王冯陈褚卫蒋沈韩杨朱秦尤许何吕施张孔曹严华金魏陶姜" +
"戚谢邹喻柏水窦章云苏潘葛奚范彭郎鲁韦昌马苗凤花方俞任袁柳酆鲍史唐" +
"费廉岑薛雷贺倪汤滕殷罗毕郝邬安常乐于时傅皮卞齐康伍余元卜顾孟平黄" +
"和穆萧尹姚邵湛汪祁毛禹狄米贝明臧计伏成戴谈宋茅庞熊纪舒屈项祝董梁" +
"杜阮蓝闵席季麻强贾路娄危江童颜郭梅盛林刁钟徐邱骆高夏蔡田樊胡凌霍" +
"虞万支柯昝管卢莫经房裘缪干解应宗丁宣贲邓郁单杭洪包诸左石崔吉钮龚" +
"白凝" // 凝 removed below via buildRuneSet's caller — kept literal for provenance parity
// buildRuneSet builds a rune set, then removes 凝 (patterns.SURNAMES_SINGLE.discard("凝")).
func buildRuneSet(s string) map[rune]bool {
m := map[rune]bool{}
for _, r := range s {
m[r] = true
}
delete(m, '凝')
return m
}
// Title / ordinal / topo / rank inventories (patterns.TITLE_SUFFIX/ORDINAL/TOPO_SUFFIX/GRADE_PREFIX/
// NUMERAL/RANK_WORD). TITLE_PREFIX is defined in the reference but unused by pattern_candidates, so it
// is omitted here.
var (
titleSuffix = []string{"公子", "大人", "长老", "前辈", "族长", "家老", "老祖", "祖师", "真人", "上人",
"道人", "先生", "夫人", "娘子", "姑娘", "嬷嬷", "师傅", "师父", "掌门", "宗主",
"少爷", "老爷", "小姐", "婆婆", "大娘", "大爷"}
ordinalTitle = []string{"一代", "二代", "三代", "四代", "五代", "六代", "七代", "八代", "九代", "十代",
"第一", "第二", "第三", "第四", "第五"}
topoSuffix = buildRuneSet2("山寨村疆谷城门宗府洞峰岭河江湖海林原岛潭殿阁楼院堂祠")
gradePrefix = buildRuneSet2("甲乙丙丁戊己庚辛壬癸")
numeral = buildRuneSet2("一二三四五六七八九十零百千")
rankWord = []string{"等", "转", "阶", "级", "重", "品", "段", "层"}
)
// buildRuneSet2 builds a rune set with NO discard (topo/grade/numeral inventories).
func buildRuneSet2(s string) map[rune]bool {
m := map[rune]bool{}
for _, r := range s {
m[r] = true
}
return m
}
// isSurnameStart returns the surname prefix (compound preferred) if s starts with one, else "" (patterns.
// is_surname_start). s is a []rune (a Han run suffix).
func isSurnameStart(s []rune) string {
if len(s) >= 2 {
if pre := string(s[:2]); surnamesCompound[pre] {
return pre
}
}
if len(s) >= 1 && surnamesSingle[s[0]] {
return string(s[0])
}
return ""
}
// patternInfo is one candidate's channel-proposed types + evidence, in first-seen (deterministic) order.
type patternInfo struct {
types []string
evidence []string
}
// formantInfo describes a detected productive-morphology formant (patterns.detect_formants entry).
type formantInfo struct {
role string // suffix|prefix|both
overRep float64
nSuffix int
nPrefix int
}
// runeHasPrefixAt reports whether run[i:] starts with p (patterns' run.startswith(suf, i)).
func runeHasPrefixAt(run []rune, i int, p []rune) bool {
if i+len(p) > len(run) {
return false
}
for k := range p {
if run[i+k] != p[k] {
return false
}
}
return true
}
// formantType maps a formant char to a candidate type (patterns.formant_type): topo→place, 等/转/阶→
// title, else term.
func formantType(c rune) string {
if topoSuffix[c] {
return "place"
}
if c == '等' || c == '转' || c == '阶' {
return "title"
}
return "term"
}
// bookCharFreq counts Han chars over the normalized sources (patterns.book_char_freq).
func bookCharFreq(chunks []MinerChunk) map[rune]int {
bc := map[rune]int{}
for _, c := range chunks {
for _, r := range c.NSource {
if isMinerHan(r) {
bc[r]++
}
}
}
return bc
}
// detectFormants auto-detects the productive DOMAIN formants (patterns.detect_formants): a Han char that
// binds (suffix OR prefix) with ≥ minPartners DISTINCT content morphemes among candidates AND is
// over-represented in-book vs general zh (over_rep ≥ minOverRep). Over-representation — NOT char-rarity
// — isolates 蛊/窍/虫 while rejecting common chars 师/花/等. Deterministic (per-char, order-independent).
func detectFormants(chunks []MinerChunk, candFreq map[string]int, contrast *Contrast, minPartners int, minOverRep float64) map[rune]formantInfo {
bc := bookCharFreq(chunks)
totBook := 0
for _, v := range bc {
totBook += v
}
if totBook == 0 {
totBook = 1
}
suf := map[rune]map[string]bool{}
pre := map[rune]map[string]bool{}
addPartner := func(m map[rune]map[string]bool, key rune, part string) {
if m[key] == nil {
m[key] = map[string]bool{}
}
m[key][part] = true
}
for a := range candFreq {
ar := []rune(a)
if len(ar) < 2 {
continue
}
addPartner(suf, ar[len(ar)-1], string(ar[:len(ar)-1]))
addPartner(pre, ar[0], string(ar[1:]))
}
chars := map[rune]bool{}
for c := range suf {
chars[c] = true
}
for c := range pre {
chars[c] = true
}
formants := map[rune]formantInfo{}
for c := range chars {
pBook := float64(bc[c]) / float64(totBook)
overRep := contrast.charOverRep(c, pBook)
if overRep < minOverRep {
continue
}
ns, npr := len(suf[c]), len(pre[c])
role := ""
if ns >= minPartners {
role = "suffix"
}
if npr >= minPartners {
if role != "" {
role = "both"
} else {
role = "prefix"
}
}
if role != "" {
formants[c] = formantInfo{role: role, overRep: overRep, nSuffix: ns, nPrefix: npr}
}
}
return formants
}
// patternCandidates produces the typed pattern candidates from the source (patterns.pattern_candidates).
// It operates over the ALREADY-normalized source and may propose candidates BELOW the V-A freq floor
// (that is the point — patterns close the rare-term blind spot). Returns the candidate→info map and the
// detected formants. Deterministic (fixed chunk/run/position iteration; add() preserves first-seen order).
func patternCandidates(chunks []MinerChunk, candFreq map[string]int, contrast *Contrast, minPartners int, minOverRep float64) (map[string]*patternInfo, map[rune]formantInfo) {
out := map[string]*patternInfo{}
add := func(cand []rune, typ, ev string) {
cn := normalizeSourceKey(string(cand))
if cn == "" || !minerHanOnly(cn) {
return
}
info := out[cn]
if info == nil {
info = &patternInfo{}
out[cn] = info
}
if !containsStr(info.types, typ) {
info.types = append(info.types, typ)
}
if !containsStr(info.evidence, ev) {
info.evidence = append(info.evidence, ev)
}
}
titleSufR := toRuneSlices(titleSuffix)
ordinalR := toRuneSlices(ordinalTitle)
rankR := toRuneSlices(rankWord)
for _, c := range chunks {
for _, run := range hanRuns(c.NSource) {
L := len(run)
for i := 0; i < L; i++ {
// (1) surname anchor: surname + 12 Han given-name window → name
if sn := isSurnameStart(run[i:]); sn != "" {
base := i + len([]rune(sn))
for _, gl := range [2]int{1, 2} {
if base+gl <= L {
full := run[i : base+gl]
if len(full) >= 2 && len(full) <= 4 {
add(full, "name", "surname:"+sn)
}
}
}
}
// (2) title suffix: content + suffix → title
for si, suf := range titleSufR {
if runeHasPrefixAt(run, i, suf) {
for _, left := range [4]int{3, 2, 1, 0} {
if i-left >= 0 {
cand := run[i-left : i+len(suf)]
if len(cand) >= 2 && len(cand) <= 6 {
add(cand, "title", "title_suffix:"+titleSuffix[si])
}
}
}
add([]rune(titleSuffix[si]), "title", "title_bare:"+titleSuffix[si])
}
}
// (2b) ordinal + title (四代族长-class) → title
for oi, od := range ordinalR {
if runeHasPrefixAt(run, i, od) {
for si, suf := range titleSufR {
end := i + len(od)
if runeHasPrefixAt(run, end, suf) {
add(run[i:end+len(suf)], "title", "ordinal_title:"+ordinalTitle[oi]+"+"+titleSuffix[si])
}
}
}
}
// (2c) rank/grade compositional: (numeral|grade-prefix) + rank-word → title
for ri, rw := range rankR {
if runeHasPrefixAt(run, i, rw) && i >= 1 {
left := run[i-1]
if numeral[left] || gradePrefix[left] {
add(run[i-1:i+len(rw)], "title", "rank_grade:"+string(left)+"+"+rankWord[ri])
}
}
}
// (3) topo suffix: 13 Han base + topo char → place
if topoSuffix[run[i]] && i >= 1 {
for _, left := range [3]int{3, 2, 1} {
if i-left >= 0 {
cand := run[i-left : i+1]
if len(cand) >= 2 && len(cand) <= 4 {
add(cand, "place", "topo_suffix:"+string(run[i]))
}
}
}
}
}
}
}
// (4) productive morphology: for each detected formant, propose all binding n-grams.
formants := detectFormants(chunks, candFreq, contrast, minPartners, minOverRep)
for _, c := range chunks {
for _, run := range hanRuns(c.NSource) {
L := len(run)
for i := 0; i < L; i++ {
info, ok := formants[run[i]]
if !ok {
continue
}
typ := formantType(run[i])
if info.role == "suffix" || info.role == "both" {
for _, left := range [3]int{3, 2, 1} {
if i-left >= 0 {
cand := run[i-left : i+1]
if len(cand) >= 2 && len(cand) <= 4 {
add(cand, typ, "formant_suffix:"+string(run[i]))
}
}
}
}
if info.role == "prefix" || info.role == "both" {
for _, r := range [2]int{2, 3} {
if i+r <= L {
cand := run[i : i+r]
if len(cand) >= 2 && len(cand) <= 4 {
add(cand, typ, "formant_prefix:"+string(run[i]))
}
}
}
}
}
}
}
return out, formants
}
// minerHanOnly reports whether s is non-empty and all Han (exp16_common.han_only).
func minerHanOnly(s string) bool {
if s == "" {
return false
}
for _, r := range s {
if !isMinerHan(r) {
return false
}
}
return true
}
// toRuneSlices converts a []string inventory to []([]rune) once (avoids re-decoding per scan position).
func toRuneSlices(ss []string) [][]rune {
out := make([][]rune, len(ss))
for i, s := range ss {
out[i] = []rune(s)
}
return out
}
// containsStr reports whether ss contains s.
func containsStr(ss []string, s string) bool {
for _, x := range ss {
if x == s {
return true
}
}
return false
}