textmachine/backend/internal/lang/langpack.go

852 lines
38 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

// Package lang holds the language-specific DATA the translation engine reads, isolated OUT of
// internal/pipeline/ as versioned files under configs/langpacks/ (owner directive D39.15: language data
// must not live as Go constants inside the engine; horizon = hundreds of languages, data-as-files, add a
// pair = drop a directory, no recompile, no pipeline/ edits).
//
// This package carries NO behaviour. The miner / checker ALGORITHMS stay in internal/pipeline (parity-
// locked, owner: "from the engine, DATA leaves; the ALGORITHM stays") and READ these tables. The one-way
// dependency (pipeline → lang, never the reverse) is compile-enforced: nothing here imports pipeline, so
// the data/algorithm boundary is real, not a convention.
//
// A Pack is resolved by (source, target) language and content-hashed at load, mirroring the two in-repo
// precedents: the pair-keyed prompt seam (config.Stage.Prompts + PromptPathFor + PromptSHA256, resolved
// by Book.LangPair(), fail-loud on a missing pair) and the memnorm trad→simp table (a data file whose
// bytes ARE its version via a content hash). Editing a pack file changes its Version() → the pipeline
// folds that into the snapshot → a loud --resnapshot, drift-proof by mechanism, not discipline.
package lang
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"os"
"path/filepath"
"sort"
"strconv"
"strings"
)
// packAlgoVersion tags the PARSE/layout of a pack (the file manifest + how each file is read). Bump it on
// a schema change (a new file, a format change). The DATA content is versioned separately by hashing the
// files into Version(), so editing a table also invalidates — you cannot forget to bump a version when
// you change the data, because the data's bytes ARE the version (the text/norm.go drift-proofing).
// v2 (pair-14): new pair/source files (title-formant, sentence-terminator, palladius-phonotactics, dc-checkers)
// + new formats (the pattern/category rows, the generic Palladius parser) — a schema change, so the tag bumps.
const packAlgoVersion = "langpack-v2"
// Pack is a loaded, versioned language-data pack for one source→target pair. Fields are the DATA the
// pipeline algorithms read; the zero value is unusable (load via Load). Maps are membership sets / lookup
// tables (order-free); slices preserve their authored order.
type Pack struct {
Pair string // "<src>-<tgt>", e.g. "zh-ru" (== config.Book.LangPair())
// <src> source morphology (configs/langpacks/<src>/) — the miner's typed-candidate channels.
SurnamesSingle map[rune]bool // 百家姓 single-char surnames
SurnamesCompound map[string]bool // two-char compound surnames
TitleSuffix []string // title suffixes → title (ordered)
OrdinalTitle []string // ordinal titles 一代/第一… (ordered)
RankWord []string // rank/measure words 等/转… (ordered)
TopoSuffix map[rune]bool // topographic suffix chars → place
GradePrefix map[rune]bool // grade/stem prefix chars 甲乙丙…
Numeral map[rune]bool // CJK numeral chars
AliasParticle map[rune]bool // trailing-particle set marking a boundary fragment
// TitleFormant: the rank/measure formant chars a miner classifies as a TITLE (patterns.formant_type
// 等/转/阶 → title). Pair-14: moved out of the pipeline formant switch, sitting beside TopoSuffix (its
// place-side neighbour). A rune SET.
TitleFormant map[rune]bool
// SentenceTerminator: the source sentence-ending marks the alias miner splits on (。!?, alias.
// cooccur_same_sentence). Pair-14: moved out of the pipeline literal. A rune SET; the miner adds "\n"
// (a structural newline, not a language mark) itself.
SentenceTerminator map[rune]bool
// Palladius is the <src>-<tgt> transliteration table (configs/langpacks/<pair>/palladius.txt +
// palladius-phonotactics.txt) — the pinyin→Cyrillic maps and the syllable-generator's phonotactic
// constraints, as ONE typed value (owner addendum 24.07: a struct, not four+three parallel Pack fields).
Palladius Palladius
// DCCheckers is the OPTIONAL pair data for the WS5 defect-class checkers (configs/langpacks/<pair>/
// dc-checkers.txt, checkers_zh_ru.go). nil when the pair ships no file — the checker ALGORITHMS then read
// empty tables and fire 0 (a pair that does not opt in is never flagged). DATA only (pair-14 §6: the
// LOOKUP TABLES move out of the pipeline; the regex DETECTION patterns stay as the checker algorithm).
DCCheckers *DCCheckerData
// GenreGlossary is the OPTIONAL pair/genre reference glossary (configs/langpacks/<pair>/genre-glossary.txt,
// D39.42 п.1): the renderings the INDUSTRY has settled on for this pair's genre (修炼 → «культивация», not
// «совершенствование»). It is PAIR/GENRE data, never book canon — a book's private names live in its own
// CANON-NOTES/seed and never in the shared pair layer. The terminologist reads it as an anchor so an
// unsigned book does not re-invent a term the whole market already renders one way. nil when the pair
// ships no file. Its bytes are NOT part of Version() — see Load: this is the one optional pair file that
// shapes no snapshot-folded stage, so curating it must not re-bill books.
GenreGlossary []GenreTerm
// Heading is the OPTIONAL chapter-heading rule (configs/langpacks/<pair>/heading.txt). nil when the pair
// carries no heading.txt — the chapter-title feature is then inert (the chunker keeps the source header
// as-is), so a pair that does not opt in is never re-billed for it. It is DATA only: the detect/strip/
// render ALGORITHM lives in the chunker (internal/pipeline), which reads this table — the same
// data/algorithm boundary the miner tables keep.
Heading *HeadingRule
version string
}
// Palladius is the pair transliteration table: the pinyin→Cyrillic maps and the syllable-generator's
// phonotactic constraint sets (owner addendum 24.07 — one typed struct instead of four+three parallel Pack
// fields). Populated from palladius.txt (Initials/Finals/YW/SpecialI) + palladius-phonotactics.txt
// (Retroflex/VFinal/VFinalInitial) by the generic category parser; which categories are REQUIRED is enforced
// by validate(), not the parser (an unknown category is collected, never a parse error).
type Palladius struct {
Initials, Finals, YW, SpecialI map[string]string // pinyin→Cyrillic
Retroflex, VFinal, VFinalInitial map[string]bool // phonotactic constraint sets (pinyin tokens)
}
// newPalladius returns a Palladius with all tables allocated (so merge/union is nil-safe).
func newPalladius() Palladius {
return Palladius{
Initials: map[string]string{}, Finals: map[string]string{}, YW: map[string]string{}, SpecialI: map[string]string{},
Retroflex: map[string]bool{}, VFinal: map[string]bool{}, VFinalInitial: map[string]bool{},
}
}
// DCCheckerData is the per-pair lookup data the WS5 defect-class checkers read (checkers_zh_ru.go). The
// checker DETECTION regexes stay in the pipeline as the pair-scoped algorithm (§12.2); only the LOOKUP
// TABLES live here as data. Parsed from dc-checkers.txt.
type DCCheckerData struct {
Numeral map[rune]int // DC1: small CJK count → value (一→1 … 十→10, incl. 两→2)
RuHours map[string]int // DC1: Russian hours-count word → value (один→1 … шесть→6)
RegisterNeg []string // DC6: register negative-list lexemes (терем + case forms), authored order
// Patterns are the DETECTION patterns as DATA (pair-14 data-out): key → verbatim regex or literal probe.
// The `*_re` keys are compiled by the consumer; the rest are literal strings.Contains probes. A pair that
// ships no pattern for a key runs that sub-checker inert.
Patterns map[string]string
// Ratios are the pair's numeric UNIT RELATIONS as data (nit of generality, pre-run hygiene 25.07): the
// checker algorithm compares counts, but "one source time-unit equals N target hours" is a fact ABOUT THE
// PAIR, not about the algorithm — 时辰=2h for zh, and a pair using another unit (刻=¼h) must not need a Go
// edit to say so. key → integer.
Ratios map[string]int
// Messages are the checkers' human-readable DETAIL templates as data (same nit): a detail line naming
// 时辰/千万/成 is pair SUBSTANCE, and a second pair shipping its own patterns would otherwise inherit a
// message about Chinese double-hours. `{name}` placeholders are substituted by the consumer; the template
// is verbatim. Required whenever the corresponding detection data is present (a detector that fires with
// no way to say what it found is a corrupt pack).
Messages map[string]string
}
// GenreTerm is one row of the pair's genre reference glossary. Genre is an OPTIONAL free label matched
// against the book's `genre` field (case-folded, trimmed); an empty Genre applies to every book of the
// pair. Keeping the label in the DATA is what lets a second genre (or a second pair) ship its own
// conventions without a Go edit — the engine only ever compares two strings it was handed.
type GenreTerm struct {
Src string
Dst string
Genre string
}
// HeadingRule is the per-pair data for the chapter-title policy: instead of letting the model render a
// chapter heading (which drifted to «Раздел 2» / «Первая глава» / an orphaned « :» across models), the
// chunker detects a source header (Marker + a numeral + a Unit rune), strips it from the model input, and
// the read-models render Template deterministically instead. Data only; parsed from heading.txt.
type HeadingRule struct {
Marker string // the prefix rune(s) that open a numbered heading, e.g. "第"
Units map[rune]bool // the section-unit runes accepted right after the numeral (章 节 節 回)
Template string // the target rendering; the literal "{n}" is replaced by the parsed Arabic number
}
// Version is the content hash of the pack (packAlgoVersion + a sha256 of the authored file bytes). A pack
// edit changes it, so the pipeline can fold it into the snapshot (a loud --resnapshot on any data edit).
func (p *Pack) Version() string { return p.version }
// srcFiles are the source-morphology files (under configs/langpacks/<src>/), read in this fixed order.
// SCOPE (honest): this manifest is the zh-family NAME-MINER's morphology schema (百家姓 surnames, 甲乙丙 grade
// prefixes, 等/转/阶 title-formants, CJK numerals, …), and validate() forbids an empty table. So "add a pair =
// drop a directory, no recompile" holds for a source in THIS family (a second CJK source drops its files); a
// different source family that wants mining needs new Pack fields + parse + miner channels, not just a dir.
var srcFiles = []string{
"surnames-single.txt", "surnames-compound.txt", "title-suffix.txt", "ordinal-title.txt",
"rank-word.txt", "topo-suffix.txt", "grade-prefix.txt", "numeral.txt", "alias-particle.txt",
"title-formant.txt", "sentence-terminator.txt",
}
// pairFiles are the pair-transliteration files (under configs/langpacks/<pair>/).
var pairFiles = []string{"palladius.txt", "palladius-phonotactics.txt"}
// Load resolves and reads the pack for (sourceLang, targetLang) from root (e.g. "configs/langpacks"): the
// source-morphology files under root/<src>/ and the pair-transliteration files under root/<src>-<tgt>/.
// EVERY declared file must be present and well-formed — a missing/corrupt file fails LOUD (the caller runs
// this at load, before any billing, mirroring the prompt-pack os.Stat loop). The content hash covers all
// files in a fixed order, so it is deterministic and drift-proof.
func Load(root, sourceLang, targetLang string) (*Pack, error) {
pair := sourceLang + "-" + targetLang
p := &Pack{Pair: pair, Palladius: newPalladius()}
h := sha256.New()
h.Write([]byte(packAlgoVersion))
read := func(dir, name string) ([]byte, error) {
path := filepath.Join(root, dir, name)
b, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("langpack %q: %w (add the file or fix the book's source_lang/target_lang; D39.15: language data is required, never silently empty)", pair, err)
}
// Fold the RELATIVE path + bytes so a rename or a moved byte both shift the hash.
h.Write([]byte("\x00" + dir + "/" + name + "\x00"))
h.Write(b)
return b, nil
}
for _, name := range srcFiles {
b, err := read(sourceLang, name)
if err != nil {
return nil, err
}
if err := p.assignSrc(name, b); err != nil {
return nil, fmt.Errorf("langpack %q %s: %w", pair, name, err)
}
}
for _, name := range pairFiles {
b, err := read(pair, name)
if err != nil {
return nil, err
}
if err := p.assignPair(name, b); err != nil {
return nil, fmt.Errorf("langpack %q %s: %w", pair, name, err)
}
}
// Optional per-pair heading rule (pack-13 title policy). ABSENT → nil, the title feature is inert and
// the pack's Version() is byte-stable (a pair that does not opt in is never re-billed); PRESENT → its
// bytes fold into the content hash (a loud --resnapshot on any edit) and it is parsed; CORRUPT → fail
// loud, like the required-file path. Distinct from read(): a missing heading.txt is NOT an error.
if hb, ok, herr := readOptional(root, pair, "heading.txt"); herr != nil {
return nil, fmt.Errorf("langpack %q heading.txt: %w", pair, herr)
} else if ok {
h.Write([]byte("\x00" + pair + "/heading.txt\x00"))
h.Write(hb)
hr, perr := parseHeading(hb)
if perr != nil {
return nil, fmt.Errorf("langpack %q heading.txt: %w", pair, perr)
}
p.Heading = hr
}
// Optional per-pair DC-checker tables (pair-14 §6). ABSENT → nil, the checkers read empty tables and fire
// 0 (a pair that does not opt in is never re-billed / flagged); PRESENT → bytes fold into the hash and it
// is parsed; CORRUPT → fail loud. Same optional contract as heading.txt.
if db, ok, derr := readOptional(root, pair, "dc-checkers.txt"); derr != nil {
return nil, fmt.Errorf("langpack %q dc-checkers.txt: %w", pair, derr)
} else if ok {
h.Write([]byte("\x00" + pair + "/dc-checkers.txt\x00"))
h.Write(db)
dc, perr := parseDCCheckers(db)
if perr != nil {
return nil, fmt.Errorf("langpack %q dc-checkers.txt: %w", pair, perr)
}
p.DCCheckers = dc
}
// Optional per-pair GENRE glossary (pack-20 / D39.42 п.1). Same optional contract as heading.txt: ABSENT
// → nil; CORRUPT → loud. DELIBERATELY NOT FOLDED into the content hash, unlike every other optional file
// here. The rule the fold encodes is "this data shapes the WIRE or the VERDICT of a snapshot-folded
// stage": the miner tables, the heading rule and the DC checkers all do. This one does not — it is read
// by exactly one consumer, the TERMINOLOGIST's request, and that call class is itself deliberately
// outside the snapshot (config.TerminologyGate). Folding it would make an EDITORIAL edit — the curator
// signing one industry rendering — move every book's snapshot, and a langpack move is not a bank-only
// move, so not one chunk of those books could re-pin at $0: a one-line data edit would re-buy whole
// waves. The edit is still not free where it should not be: the terminologist's own request hash covers
// the rendered anchor, so its batches are re-bought and nothing else is.
if gb, ok, gerr := readOptional(root, pair, "genre-glossary.txt"); gerr != nil {
return nil, fmt.Errorf("langpack %q genre-glossary.txt: %w", pair, gerr)
} else if ok {
gg, perr := parseGenreGlossary(gb)
if perr != nil {
return nil, fmt.Errorf("langpack %q genre-glossary.txt: %w", pair, perr)
}
p.GenreGlossary = gg
}
if err := p.validate(); err != nil {
return nil, fmt.Errorf("langpack %q: %w", pair, err)
}
p.version = packAlgoVersion + "-" + hex.EncodeToString(h.Sum(nil))[:12]
return p, nil
}
// LoadWithOverlay loads the shared pair pack (Load) and then UNIONS a book-scoped OVERLAY on top: a book's
// PRIVATE canon (a clan surname 古月, a sect term) that belongs to ONE book, not the shared pair langpack
// (pair-14 §1 — a book term in the shared pair layer is a leak). overlayRoot has the same layout as root
// (<src>/… + <pair>/…); ONLY the files a book chooses to ship are present, each read OPTIONALLY and UNIONED
// (additive — sets gain members, ordered slices append; an overlay never removes a base entry). The overlay
// bytes fold into Version(), so a book-canon edit is a loud --resnapshot for THAT book while the shared pack
// stays byte-stable. overlayRoot == "" ⇒ identical to Load (same *Pack, same Version).
func LoadWithOverlay(root, src, tgt, overlayRoot string) (*Pack, error) {
p, err := Load(root, src, tgt)
if err != nil {
return nil, err
}
if overlayRoot == "" {
return p, nil
}
pair := src + "-" + tgt
// FAIL LOUD one level UP first (pack-15): an overlay root that does not exist, or that holds anything
// besides the two expected subdirectories, is an operator mistake — and the fold below would swallow it
// in silence. langpack_extend is an EXPLICIT opt-in in book.yaml: if it is set, the caller asserts that a
// private canon exists, so a missing/typo'd path must stop the load, never degrade recall quietly.
if fi, serr := os.Stat(overlayRoot); serr != nil || !fi.IsDir() {
return nil, fmt.Errorf("langpack %q overlay root %s is not a readable directory (langpack_extend is an explicit opt-in — a missing or mistyped path would silently drop the book's private canon): %v", pair, overlayRoot, serr)
}
rootEnts, rerr := os.ReadDir(overlayRoot)
if rerr != nil {
return nil, fmt.Errorf("langpack %q overlay root %s: %w", pair, overlayRoot, rerr)
}
for _, e := range rootEnts {
if e.IsDir() && (e.Name() == src || e.Name() == pair) {
continue
}
return nil, fmt.Errorf("langpack %q overlay root %s: unexpected entry %q — an overlay holds exactly the %q/ and %q/ subdirectories (a data file dropped at the root, or a mistyped pair directory, would be silently ignored)", pair, overlayRoot, e.Name(), src, pair)
}
// FAIL LOUD on a misnamed overlay file (review finding, pair-14 scale lens): the fold loop below reads
// ONLY the manifest names, so a typo (surname-compound.txt) or a non-mergeable file (heading.txt) shipped
// in an overlay would be SILENTLY ignored — the book's private canon never reaches the miner, recall
// degrades with no load-time signal. Enumerate the overlay dirs and refuse any unexpected file, keeping
// the "never silently empty / drift-proof" guarantee the shared loader makes.
mergeable := map[string]bool{}
for _, n := range srcFiles {
mergeable[src+"/"+n] = true
}
for _, n := range pairFiles {
mergeable[pair+"/"+n] = true
}
for _, dir := range []string{src, pair} {
names, derr := overlayDirFiles(filepath.Join(overlayRoot, dir))
if derr != nil {
return nil, fmt.Errorf("langpack %q overlay %s: %w", pair, dir, derr)
}
for _, n := range names {
if !mergeable[dir+"/"+n] {
return nil, fmt.Errorf("langpack %q overlay: unexpected file %s/%s — an overlay merges only the source/pair manifest files (a typo, or a non-mergeable file like heading.txt/dc-checkers.txt, would be silently ignored)", pair, dir, n)
}
}
}
// Seed a fresh hash with the base version (which already uniquely encodes every base byte), then fold the
// overlay files in a fixed order — deterministic and drift-proof (edit the overlay → Version() moves).
h := sha256.New()
h.Write([]byte(p.version))
merged := false
fold := func(dir, name string, pairFile bool) error {
b, ok, rerr := readOptional(overlayRoot, dir, name)
if rerr != nil {
return fmt.Errorf("langpack %q overlay %s/%s: %w", pair, dir, name, rerr)
}
if !ok {
return nil
}
h.Write([]byte("\x00" + dir + "/" + name + "\x00"))
h.Write(b)
merged = true
if pairFile {
return p.mergePair(name, b)
}
return p.mergeSrc(name, b)
}
for _, name := range srcFiles {
if err := fold(src, name, false); err != nil {
return nil, err
}
}
for _, name := range pairFiles {
if err := fold(pair, name, true); err != nil {
return nil, err
}
}
if merged {
p.version = packAlgoVersion + "-x" + hex.EncodeToString(h.Sum(nil))[:12]
}
return p, nil
}
// mergeSrc unions an overlay source file into the loaded pack (additive; see LoadWithOverlay). Rune/string
// SETS gain members; ordered slices append (a book's extra title/rank tokens follow the shared ones).
func (p *Pack) mergeSrc(name string, b []byte) error {
switch name {
case "surnames-single.txt":
unionRuneSet(p.SurnamesSingle, runeSet(b))
case "surnames-compound.txt":
unionStringSet(p.SurnamesCompound, stringSet(b))
case "title-suffix.txt":
p.TitleSuffix = append(p.TitleSuffix, lines(b)...)
case "ordinal-title.txt":
p.OrdinalTitle = append(p.OrdinalTitle, lines(b)...)
case "rank-word.txt":
p.RankWord = append(p.RankWord, lines(b)...)
case "topo-suffix.txt":
unionRuneSet(p.TopoSuffix, runeSet(b))
case "grade-prefix.txt":
unionRuneSet(p.GradePrefix, runeSet(b))
case "numeral.txt":
unionRuneSet(p.Numeral, runeSet(b))
case "alias-particle.txt":
unionRuneSet(p.AliasParticle, runeSet(b))
case "title-formant.txt":
unionRuneSet(p.TitleFormant, runeSet(b))
case "sentence-terminator.txt":
unionRuneSet(p.SentenceTerminator, runeSet(b))
default:
return fmt.Errorf("unknown source file")
}
return nil
}
// mergePair unions an overlay pair file into the loaded pack (additive). Both pair files carry Palladius
// categories, parsed generically and unioned into p.Palladius (same path as assignPair).
func (p *Pack) mergePair(name string, b []byte) error {
return p.assignPair(name, b)
}
func unionRuneSet(dst, src map[rune]bool) {
for k := range src {
dst[k] = true
}
}
func unionStringSet(dst, src map[string]bool) {
for k := range src {
dst[k] = true
}
}
// unionStringMap merges src into dst ADDITIVELY and refuses a COLLISION: a key already present in dst
// (the shared pack) that an overlay also carries would have SILENTLY overridden the base value. Override
// semantics are deliberately NOT ratified (D39.23) — an overlay exists to ADD a book's private canon, and
// a book that needs to change a shared pair mapping is a signal about the shared pack, not something to
// absorb quietly. Revisit if a book ever genuinely needs it; until then the collision is loud and names
// the key and both values. Sets/slices need no such rule: adding a member twice is idempotent.
func unionStringMap(category string, dst, src map[string]string) error {
for k, v := range src {
if old, ok := dst[k]; ok {
return fmt.Errorf("category %q key %q is already defined as %q and the overlay redefines it as %q — an overlay may only ADD (override is not supported: change the shared pair pack instead)", category, k, old, v)
}
dst[k] = v
}
return nil
}
// validate makes the "never silently empty" contract real: a present-but-empty or comment-only data file
// (a fat-fingered edit once packs are hand-authored at R1) parses to an empty table with no error and would
// silently disable a miner channel — recall degradation with no load-time signal. Every required table must
// be non-empty. The content hash catches an EDIT (a --resnapshot signal), but "edited to empty" is a corrupt
// pack, not an intended change, so it is refused at load, before any consumer, like the missing-file path.
func (p *Pack) validate() error {
var empty []string
req := func(name string, n int) {
if n == 0 {
empty = append(empty, name)
}
}
req("surnames-single", len(p.SurnamesSingle))
req("surnames-compound", len(p.SurnamesCompound))
req("title-suffix", len(p.TitleSuffix))
req("ordinal-title", len(p.OrdinalTitle))
req("rank-word", len(p.RankWord))
req("topo-suffix", len(p.TopoSuffix))
req("grade-prefix", len(p.GradePrefix))
req("numeral", len(p.Numeral))
req("alias-particle", len(p.AliasParticle))
req("title-formant", len(p.TitleFormant))
req("sentence-terminator", len(p.SentenceTerminator))
// The Palladius REQUIRED-category list lives HERE (the consumer), not in the generic parser (addendum).
req("palladius/initials", len(p.Palladius.Initials))
req("palladius/finals", len(p.Palladius.Finals))
req("palladius/yw", len(p.Palladius.YW))
req("palladius/special_i", len(p.Palladius.SpecialI))
req("palladius-phonotactics/retroflex", len(p.Palladius.Retroflex))
req("palladius-phonotactics/vfinal", len(p.Palladius.VFinal))
req("palladius-phonotactics/vfinal_initial", len(p.Palladius.VFinalInitial))
if len(empty) > 0 {
return fmt.Errorf("empty required table(s) %s — a present-but-empty/comment-only data file is a corrupt pack, not a valid one", strings.Join(empty, ", "))
}
return nil
}
func (p *Pack) assignSrc(name string, b []byte) error {
switch name {
case "surnames-single.txt":
p.SurnamesSingle = runeSet(b)
case "surnames-compound.txt":
p.SurnamesCompound = stringSet(b)
case "title-suffix.txt":
p.TitleSuffix = lines(b)
case "ordinal-title.txt":
p.OrdinalTitle = lines(b)
case "rank-word.txt":
p.RankWord = lines(b)
case "topo-suffix.txt":
p.TopoSuffix = runeSet(b)
case "grade-prefix.txt":
p.GradePrefix = runeSet(b)
case "numeral.txt":
p.Numeral = runeSet(b)
case "alias-particle.txt":
p.AliasParticle = runeSet(b)
case "title-formant.txt":
p.TitleFormant = runeSet(b)
case "sentence-terminator.txt":
p.SentenceTerminator = runeSet(b)
default:
return fmt.Errorf("unknown source file")
}
return nil
}
// assignPair reads a pair file into p.Palladius. Both pair files (palladius.txt, palladius-phonotactics.txt)
// carry `category<TAB>…` rows, parsed by ONE generic category parser (owner addendum 24.07); the known
// Palladius categories are UNIONED into p.Palladius and an unknown category is simply ignored, NEVER a parse
// error — which categories are REQUIRED is enforced downstream by validate(), not here.
func (p *Pack) assignPair(name string, b []byte) error {
cats, err := parseCategoryRows(b)
if err != nil {
return err
}
if merr := p.Palladius.merge(cats); merr != nil {
return merr
}
return nil
}
// overlayDirFiles lists the regular-file names directly in dir (an overlay's <src> or <pair> subdir). A
// MISSING dir is fine (returns nil — a book may overlay only sources or only the pair). Any other read
// error is loud. Nested dirs are ignored (only top-level manifest files are mergeable).
func overlayDirFiles(dir string) ([]string, error) {
ents, err := os.ReadDir(dir)
if err != nil {
if os.IsNotExist(err) {
return nil, nil
}
return nil, err
}
var out []string
for _, e := range ents {
if !e.IsDir() {
out = append(out, e.Name())
}
}
return out, nil
}
// readOptional reads an OPTIONAL pack file. A missing file returns (nil, false, nil) — the feature it
// backs is simply inert — while any OTHER read error (permission, a directory) is a loud failure; a
// present file returns (bytes, true, nil). Used for the pack-13 heading rule, which a pair opts into.
func readOptional(root, dir, name string) ([]byte, bool, error) {
b, err := os.ReadFile(filepath.Join(root, dir, name))
if err != nil {
if os.IsNotExist(err) {
return nil, false, nil
}
return nil, false, err
}
return b, true, nil
}
// parseHeading reads heading.txt into a HeadingRule. Format: `key<TAB>value` per non-comment line, keys
// marker | units | template (all three required). `units` is a rune SET (each rune a member); `template`
// must contain the literal "{n}" placeholder (else it could never render a number). Fail-loud on a missing
// key / unknown key / empty value / a template without {n} — a malformed rule is a corrupt pack, not an
// intended silent no-op (mirrors the required-file "never silently empty" contract).
func parseHeading(b []byte) (*HeadingRule, error) {
hr := &HeadingRule{Units: map[rune]bool{}}
seen := map[string]bool{}
for i, raw := range strings.Split(string(b), "\n") {
t := strings.TrimRight(raw, "\r")
if strings.TrimSpace(t) == "" || strings.HasPrefix(strings.TrimSpace(t), "#") {
continue
}
f := strings.SplitN(t, "\t", 2)
if len(f) != 2 || strings.TrimSpace(f[1]) == "" {
return nil, fmt.Errorf("line %d: want `key<TAB>value` with a non-empty value (%q)", i+1, t)
}
key, val := strings.TrimSpace(f[0]), strings.TrimSpace(f[1])
seen[key] = true
switch key {
case "marker":
hr.Marker = val
case "units":
for _, r := range val {
if !isSpace(r) {
hr.Units[r] = true
}
}
case "template":
hr.Template = val
default:
return nil, fmt.Errorf("line %d: unknown key %q (want marker|units|template)", i+1, key)
}
}
if !seen["marker"] || !seen["units"] || !seen["template"] || len(hr.Units) == 0 {
return nil, fmt.Errorf("heading rule needs non-empty marker, units and template")
}
if !strings.Contains(hr.Template, "{n}") {
return nil, fmt.Errorf("template %q must contain the {n} number placeholder", hr.Template)
}
return hr, nil
}
// parseGenreGlossary reads genre-glossary.txt into the ordered GenreTerm rows. Format per non-comment
// line: `src<TAB>dst[<TAB>genre]`. AUTHORED ORDER is preserved (the anchor block a model reads is a list,
// and its order is the curator's). Fail-loud on a malformed row (fewer than two fields, an empty src or
// dst): a silently-dropped convention is worse than a refused pack — the whole point of the file is that
// the engine does not re-invent a rendering the market already agreed on.
func parseGenreGlossary(b []byte) ([]GenreTerm, error) {
var out []GenreTerm
seen := map[string]int{} // src|genre → line, so a duplicated key cannot silently shadow itself
for i, raw := range strings.Split(string(b), "\n") {
t := strings.TrimRight(raw, "\r")
if strings.TrimSpace(t) == "" || strings.HasPrefix(strings.TrimSpace(t), "#") {
continue
}
f := strings.Split(t, "\t")
if len(f) < 2 {
return nil, fmt.Errorf("line %d: want `src<TAB>dst[<TAB>genre]` (%q)", i+1, t)
}
g := GenreTerm{Src: strings.TrimSpace(f[0]), Dst: strings.TrimSpace(f[1])}
if len(f) >= 3 {
g.Genre = strings.TrimSpace(f[2])
}
if g.Src == "" || g.Dst == "" {
return nil, fmt.Errorf("line %d: src and dst must both be non-empty (%q)", i+1, t)
}
key := g.Src + "\x00" + strings.ToLower(g.Genre)
if prev, dup := seen[key]; dup {
return nil, fmt.Errorf("line %d: duplicate src %q for genre %q (already on line %d) — two conventions for one term is a curator decision, not a silent last-wins", i+1, g.Src, g.Genre, prev)
}
seen[key] = i + 1
out = append(out, g)
}
return out, nil
}
// GenreGlossaryFor returns the pack's genre-glossary rows that apply to a book of the given genre: rows
// with no genre label (pair-wide conventions) plus rows whose label matches, case-folded and trimmed.
// Authored order is preserved. A nil pack or an empty glossary yields nil — the caller renders no anchor.
func (p *Pack) GenreGlossaryFor(genre string) []GenreTerm {
if p == nil || len(p.GenreGlossary) == 0 {
return nil
}
want := strings.ToLower(strings.TrimSpace(genre))
var out []GenreTerm
for _, g := range p.GenreGlossary {
if g.Genre == "" || strings.ToLower(strings.TrimSpace(g.Genre)) == want {
out = append(out, g)
}
}
return out
}
// runeSet reads a rune SET: every non-whitespace rune of every non-comment line is a member (order-free).
func runeSet(b []byte) map[rune]bool {
m := map[rune]bool{}
for _, ln := range contentLines(b) {
for _, r := range ln {
if !isSpace(r) {
m[r] = true
}
}
}
return m
}
// stringSet reads a set of whole tokens, one per non-comment line (trimmed).
func stringSet(b []byte) map[string]bool {
m := map[string]bool{}
for _, ln := range contentLines(b) {
if t := strings.TrimSpace(ln); t != "" {
m[t] = true
}
}
return m
}
// lines reads an ORDERED slice of whole tokens, one per non-comment line (trimmed), in file order.
func lines(b []byte) []string {
var out []string
for _, ln := range contentLines(b) {
if t := strings.TrimSpace(ln); t != "" {
out = append(out, t)
}
}
return out
}
// parseCategoryRows is the GENERIC Palladius category parser (owner addendum 24.07): it reads a pair file's
// `category<TAB>key[<TAB>value]` rows into category → key → value, WITHOUT a per-category switch and WITHOUT
// treating an unknown category as an error (which categories are required is the CONSUMER's call — validate()).
// A 3-field row (initials/finals/yw/special_i) stores key→cyrillic; a 2-field row (retroflex/vfinal/…) stores
// key→"" (a set member). Scans raw lines so an error names the PHYSICAL file line; the only parse errors are a
// bad field count / an empty key.
func parseCategoryRows(b []byte) (map[string]map[string]string, error) {
cats := map[string]map[string]string{}
for i, raw := range strings.Split(string(b), "\n") {
t := strings.TrimSpace(strings.TrimRight(raw, "\r"))
if t == "" || strings.HasPrefix(t, "#") {
continue
}
f := strings.Split(t, "\t")
if len(f) < 2 || len(f) > 3 || strings.TrimSpace(f[0]) == "" || strings.TrimSpace(f[1]) == "" {
return nil, fmt.Errorf("line %d: want `category<TAB>key[<TAB>value]`, got %q", i+1, t)
}
if cats[f[0]] == nil {
cats[f[0]] = map[string]string{}
}
val := "" // a 2-field row is a set member (value "")
if len(f) == 3 {
val = f[2]
}
cats[f[0]][f[1]] = val
}
return cats, nil
}
// palladiusCategories are the categories this CONSUMER reads. The generic parser stays category-agnostic
// (an unknown category is not a parse error — owner addendum 24.07); the CONSUMER is where a category that
// nothing consumes is caught, because only the consumer knows what "consumed" means.
var palladiusCategories = map[string]bool{
"initials": true, "finals": true, "yw": true, "special_i": true,
"retroflex": true, "vfinal": true, "vfinal_initial": true,
}
// merge unions parsed categories into the Palladius table. KNOWN categories populate the typed fields (a
// map category takes key→cyrillic, a set category takes its keys). A category NOTHING consumes is an
// ERROR here (pack-15): a typo (`retroflexx`, `initals`) parses fine and used to vanish — the file looked
// authored, the table stayed empty for that class, and the miner quietly lost a channel. A map-category
// key COLLISION between the base pack and an overlay is loud for the same reason (see unionStringMap).
func (pal *Palladius) merge(cats map[string]map[string]string) error {
var unknown []string
for c := range cats {
if !palladiusCategories[c] {
unknown = append(unknown, c)
}
}
if len(unknown) > 0 {
sort.Strings(unknown)
known := make([]string, 0, len(palladiusCategories))
for c := range palladiusCategories {
known = append(known, c)
}
sort.Strings(known)
return fmt.Errorf("category %s is not consumed by anything (a typo would parse fine and silently leave its table empty); known categories: %s",
strings.Join(unknown, ", "), strings.Join(known, ", "))
}
for _, m := range []struct {
name string
dst map[string]string
}{
{"initials", pal.Initials}, {"finals", pal.Finals}, {"yw", pal.YW}, {"special_i", pal.SpecialI},
} {
if err := unionStringMap(m.name, m.dst, cats[m.name]); err != nil {
return err
}
}
unionKeysAsSet(pal.Retroflex, cats["retroflex"])
unionKeysAsSet(pal.VFinal, cats["vfinal"])
unionKeysAsSet(pal.VFinalInitial, cats["vfinal_initial"])
return nil
}
// unionKeysAsSet adds the KEYS of a parsed category (a 2-field set) to dst.
func unionKeysAsSet(dst map[string]bool, src map[string]string) {
for k := range src {
dst[k] = true
}
}
// parseDCCheckers reads the DC-checker pair tables from category-keyed lines (checkers_zh_ru.go data):
//
// cjk_numeral<TAB>rune<TAB>value | ru_hour<TAB>word<TAB>value | register_neg<TAB>lexeme
// pattern<TAB>key<TAB>value (VERBATIM) | msg<TAB>key<TAB>template (VERBATIM) | dc_ratio<TAB>key<TAB>int
//
// Scans raw lines so an error names the PHYSICAL file line. Fail-loud on a bad field count / non-integer
// value / unknown category (a malformed table is a corrupt pack). RegisterNeg keeps its authored order.
func parseDCCheckers(b []byte) (*DCCheckerData, error) {
d := &DCCheckerData{
Numeral: map[rune]int{}, RuHours: map[string]int{},
Patterns: map[string]string{}, Ratios: map[string]int{}, Messages: map[string]string{},
}
for i, raw := range strings.Split(string(b), "\n") {
// A `pattern` / `msg` line's VALUE is verbatim (a regex may carry trailing metachars, a message its
// own spacing), so only \r is stripped from the whole line, not the value; other categories tolerate
// the trimmed form.
line := strings.TrimRight(raw, "\r")
t := strings.TrimSpace(line)
if t == "" || strings.HasPrefix(t, "#") {
continue
}
if strings.HasPrefix(line, "pattern\t") || strings.HasPrefix(line, "msg\t") {
f := strings.SplitN(line, "\t", 3) // value (f[2]) VERBATIM
if len(f) != 3 || strings.TrimSpace(f[1]) == "" || f[2] == "" {
return nil, fmt.Errorf("line %d: %s wants `%s<TAB>key<TAB>value`, got %q", i+1, f[0], f[0], line)
}
if f[0] == "msg" {
d.Messages[f[1]] = f[2]
} else {
d.Patterns[f[1]] = f[2]
}
continue
}
f := strings.Split(t, "\t")
switch f[0] {
case "cjk_numeral":
if len(f) != 3 {
return nil, fmt.Errorf("line %d: cjk_numeral wants `cjk_numeral<TAB>rune<TAB>value`, got %q", i+1, t)
}
r := []rune(f[1])
if len(r) != 1 {
return nil, fmt.Errorf("line %d: cjk_numeral key %q must be a single rune", i+1, f[1])
}
v, err := strconv.Atoi(strings.TrimSpace(f[2]))
if err != nil {
return nil, fmt.Errorf("line %d: cjk_numeral value %q: %w", i+1, f[2], err)
}
d.Numeral[r[0]] = v
case "ru_hour":
if len(f) != 3 {
return nil, fmt.Errorf("line %d: ru_hour wants `ru_hour<TAB>word<TAB>value`, got %q", i+1, t)
}
v, err := strconv.Atoi(strings.TrimSpace(f[2]))
if err != nil {
return nil, fmt.Errorf("line %d: ru_hour value %q: %w", i+1, f[2], err)
}
d.RuHours[f[1]] = v
case "register_neg":
if len(f) != 2 || strings.TrimSpace(f[1]) == "" {
return nil, fmt.Errorf("line %d: register_neg wants `register_neg<TAB>lexeme`, got %q", i+1, t)
}
d.RegisterNeg = append(d.RegisterNeg, f[1])
case "dc_ratio":
if len(f) != 3 {
return nil, fmt.Errorf("line %d: dc_ratio wants `dc_ratio<TAB>key<TAB>value`, got %q", i+1, t)
}
v, err := strconv.Atoi(strings.TrimSpace(f[2]))
if err != nil {
return nil, fmt.Errorf("line %d: dc_ratio value %q: %w", i+1, f[2], err)
}
d.Ratios[f[1]] = v
default:
return nil, fmt.Errorf("line %d: unknown category %q (want cjk_numeral|ru_hour|register_neg|dc_ratio|pattern|msg)", i+1, f[0])
}
}
if len(d.Numeral) == 0 || len(d.RuHours) == 0 || len(d.RegisterNeg) == 0 {
return nil, fmt.Errorf("dc-checkers needs non-empty cjk_numeral, ru_hour and register_neg sections")
}
return d, nil
}
// contentLines splits into lines, dropping '#'-comment and blank lines.
func contentLines(b []byte) []string {
var out []string
for _, ln := range strings.Split(string(b), "\n") {
s := strings.TrimRight(ln, "\r")
if strings.HasPrefix(strings.TrimSpace(s), "#") || strings.TrimSpace(s) == "" {
continue
}
out = append(out, s)
}
return out
}
func isSpace(r rune) bool { return r == ' ' || r == '\t' || r == '\n' || r == '\r' }