package checks import ( "fmt" "regexp" "sort" "strconv" "strings" "unicode" "textmachine/backend/internal/lang" "textmachine/backend/internal/text" ) // checkers.go: the WS5 defect-class checkers (DC1 时辰 double-hour units, DC2 千万/数十万 magnitude scale, DC6 // register negative-list) + the pack-13 general checkers (percent scale, Latin residue, broken word) — // deterministic, $0 OBSERVABILITY flaggers on the source↔FINAL text, ported from ws5_checkers_verify.py. // Like the four cheap style gates they are NEVER a disposition (a hit is recorded, never drops a chunk), // tuned PRECISION over recall. // // PAIR-AGNOSTIC (pair-14 data-out): this file no longer holds any language literal. Every DETECTION pattern, // lookup table and wordlist is DATA — the SOURCE-gated pair checkers (DC1/DC2/percent: they need a zh source // token and compare src↔tgt) read the pair pack configs/langpacks//dc-checkers.txt (lang.DCCheckerData); // the TARGET-general ones (register, broken word: they run on ANY →target output) read the embedded target // data (lang.TargetChecks). The ALGORITHM (compare counts, ×2 hours, suppress-if-ok, whole-word match) stays // here. A pair/target that ships no data runs the relevant sub-checker inert (empty → 0, the no-pack golden // path). Version rides the langpack Version() (data) + CheapGateVersion (algorithm) — a data or rule edit is // a loud --resnapshot. // Checkers is the compiled, per-run checker spec: the pair's DETECTION patterns compiled ONCE + its lookup // tables + the target-general lists, resolved from the langpack. A nil receiver, a nil pattern or an empty // table leaves that sub-checker inert. Built once per run (CompileCheckers), carried in CheapGateConfig. type Checkers struct { numeral map[rune]int ruHours map[string]int registerNeg []string brokenSuffix []string // target-general (any →ru output) yoHomograph map[string]bool // target-general: ё↔е homograph whitelist (cheapgates yofikator) magnitudeStem map[string]int // target-general: ru magnitude word stem → base-10 exponent (cheapgates) shichenRE, ruHoursRE, chengRE, decimalFractionRE *regexp.Regexp qianwanOKRE, shushiwanOKRE, shushiwanFireRE *regexp.Regexp qianwanSrc, qianwanFireWord, qianwanVetoWord string shushiwanSrc, percentWord string // halfShichenRE / halfShichenFireWord are the FRACTIONAL-unit probe of the DC1 family: the count group of // shichenRE accepts a numeral only, so a fractional unit word (半 «half») never reaches the count parser and // the most frequent form of the unit is invisible to the whole class. The probe is a plain src-pattern + // target fire-word pair (the DC2 idiom), so the pair supplies BOTH or the sub-check stays inert — a pack // without these keys behaves exactly as before this field existed. halfShichenRE *regexp.Regexp halfShichenFireWord string // hourWordRE is the target's BARE hour word (both boundaries explicit). It is not a detection pattern: // it is the positive post-condition a fractional-unit REPAIR must satisfy, and it lives with its // siblings because "how this target renders the double-hour" is one fact with one home. hourWordRE *regexp.Regexp } // CompileCheckers resolves the checker spec from the pair pack (dc) and the target data (tc). A malformed // regex in the pack is a corrupt pack → panic (deterministic, caught by the checker/golden tests). dc==nil // (a no-pack book) → the pair sub-checkers are inert; tc still supplies the target-general lists. func CompileCheckers(dc *lang.DCCheckerData, tc lang.TargetChecks) *Checkers { c := &Checkers{ brokenSuffix: tc.List("broken_suffix"), yoHomograph: listToSet(tc.List("yo_homograph")), magnitudeStem: listToStemExp(tc.List("magnitude_stem")), } if dc != nil { c.numeral, c.ruHours, c.registerNeg = dc.Numeral, dc.RuHours, dc.RegisterNeg p := dc.Patterns c.shichenRE = mustPairRE(p, "shichen_re") c.ruHoursRE = mustPairRE(p, "ru_hours_re") c.chengRE = mustPairRE(p, "cheng_re") c.decimalFractionRE = mustPairRE(p, "decimal_fraction_re") c.qianwanOKRE = mustPairRE(p, "qianwan_ok_re") c.shushiwanOKRE = mustPairRE(p, "shushiwan_ok_re") c.shushiwanFireRE = mustPairRE(p, "shushiwan_fire_re") c.qianwanSrc, c.qianwanFireWord, c.qianwanVetoWord = p["qianwan_src"], p["qianwan_fire_word"], p["qianwan_veto_word"] c.shushiwanSrc, c.percentWord = p["shushiwan_src"], p["percent_word"] c.halfShichenRE = mustPairRE(p, "halfshichen_re") c.halfShichenFireWord = p["halfshichen_fire_word"] c.hourWordRE = mustPairRE(p, "hour_word_re") } return c } // listToSet turns an ordered value list into a membership set (target wordlists). func listToSet(xs []string) map[string]bool { m := make(map[string]bool, len(xs)) for _, x := range xs { m[x] = true } return m } // listToStemExp parses `stemexp` values (the magnitude_stem list carries a second tab-field) into a // stem→exponent map. A malformed value is a corrupt embed → panic (deterministic, caught by tests). func listToStemExp(xs []string) map[string]int { m := make(map[string]int, len(xs)) for _, x := range xs { f := strings.SplitN(x, "\t", 2) if len(f) != 2 { panic(fmt.Sprintf("checks: magnitude_stem wants `stemexp`, got %q", x)) } v, err := strconv.Atoi(strings.TrimSpace(f[1])) if err != nil { panic(fmt.Sprintf("checks: magnitude_stem exponent %q: %v", f[1], err)) } m[f[0]] = v } return m } // DCCheckerData returns the pack's checker data, or nil when the book has no pack (nil-safe helper for the // compile step, which runs even for a no-langpack book). func DCCheckerData(p *lang.Pack) *lang.DCCheckerData { if p != nil { return p.DCCheckers } return nil } // mustPairRE compiles a pair detection pattern by key; a missing key → nil (inert sub-checker), a malformed // regex → panic (a corrupt pack, not a silent no-op — the same fail-loud discipline the loader keeps). func mustPairRE(p map[string]string, key string) *regexp.Regexp { s := p[key] if s == "" { return nil } re, err := regexp.Compile(s) if err != nil { panic(fmt.Sprintf("checks: langpack checker pattern %q is not a valid regex: %v", key, err)) } return re } // --- DC-1: 时辰 (double-hour) unit checker (ws5.shichen_checker) ----------------------------------- // lintTimeUnits flags a 时辰 (=2h) unit error: N个时辰 rendered as N часов (the count copied as hours) // instead of ~2N hours (三个时辰 → «три часа» should be ~6h). It fires ONLY on an explicit mismatch — a // paraphrase with no hours count is a valid rendering, not a defect. Pure and deterministic. Inert when the // pair ships no DC1 pattern (the shichen_re / ru_hours_re detection patterns are pair langpack DATA). func (c *Checkers) lintTimeUnits(source, final string) (int, []string) { if c == nil { return 0, nil } // Fractional-unit probe FIRST: 半个时辰 carries no numeral, so the counted branch below can never see it // (its count group matches a numeral only) — yet 半 is the most frequent 时辰 form in a real corpus. // Rendering it as «полчаса» halves the duration. Data-gated on BOTH keys (empty-probe guard, pack-15): // a pair that ships neither key runs this inert, exactly as before the probe existed. if c.halfShichenRE != nil && c.halfShichenFireWord != "" && c.halfShichenRE.MatchString(source) && strings.Contains(final, c.halfShichenFireWord) { return 1, []string{fmt.Sprintf("DC1 时辰 (fractional): 半个时辰 ≈ 1 h rendered as %q (half of it)", c.halfShichenFireWord)} } if c.shichenRE == nil || c.ruHoursRE == nil { return 0, nil } m := c.shichenRE.FindStringSubmatch(source) if m == nil { return 0, nil } n, ok := dcParseCount(m[1], c.numeral) if !ok { return 0, nil } hm := c.ruHoursRE.FindStringSubmatch(final) if hm == nil { return 0, nil // no explicit hours rendering → a valid paraphrase, not a defect } ruNum, ok := dcParseRuHours(hm[1], c.ruHours) if !ok { return 0, nil } expectedHours := n * 2 if ruNum == n && ruNum != expectedHours { return 1, []string{fmt.Sprintf("DC1 时辰: %d个时辰 rendered as «%d час…» (counted as hours) instead of ~%d h (1 时辰 = 2 h)", n, ruNum, expectedHours)} } return 0, nil } // dcParseCount parses the DC1 count group: an Arabic digit string or a single small CJK numeral (looked up // in the pair's DCCheckerData.Numeral, empty for a no-pack book → CJK counts don't resolve, the check is inert). func dcParseCount(s string, dcNum map[rune]int) (int, bool) { if v, err := strconv.Atoi(s); err == nil { return v, true } r := []rune(s) if len(r) == 1 { if v, ok := dcNum[r[0]]; ok { return v, true } } return 0, false } // dcParseRuHours parses the DC1 hours group: an Arabic digit string or a Russian count word (pair data). func dcParseRuHours(s string, dcRu map[string]int) (int, bool) { if v, err := strconv.Atoi(s); err == nil { return v, true } if v, ok := dcRu[s]; ok { return v, true } return 0, false } // --- DC-2: number-scale magnitude checker (ws5.magnitude_checker, 千万 / 数十万) -------------------- // lintMagnitudeScale flags a 千万 (10^7) / 数十万 (~several×10^5) magnitude rendered at a WRONG smaller // scale. A CORRECT rendering anywhere in the chunk (the ok-suppressor) suppresses the flag (ws5 reference // parity). The ok-suppressors are case-INsensitive (their data carries the (?i)); the FIRE predicates are // case-SENSITIVE literal Contains (the reference does not pass re.I to the inner searches). ⚠ 千万 is also // stock HYPERBOLE whose «тысячи» rendering is in-register (§5 A4). Observability only. All probes/patterns // are pair langpack DATA — inert when the pair ships no DC2. func (c *Checkers) lintMagnitudeScale(source, final string) (int, []string) { if c == nil { return 0, nil } var flags []string // EMPTY-PROBE GUARD (pack-15): a probe word missing from the pair data must leave the sub-check // INERT, never spurious. strings.Contains(x, "") is TRUE for every x, so an empty fire word would fire // this flag on every chunk whose source carries 千万, and an empty veto word would (silently) disable it // — a data slip that reads as a checker bug. Both words are required for the probe to run at all. if c.qianwanOKRE != nil && c.qianwanSrc != "" && c.qianwanFireWord != "" && c.qianwanVetoWord != "" && strings.Contains(source, c.qianwanSrc) && !c.qianwanOKRE.MatchString(final) { if strings.Contains(final, c.qianwanFireWord) && !strings.Contains(final, c.qianwanVetoWord) { // case-sensitive, per reference flags = append(flags, "DC2 千万=10^7 rendered as «тысячи» (≈10000× under) — a possible magnitude error (hyperbole risk, §5-A4)") } } if c.shushiwanOKRE != nil && c.shushiwanFireRE != nil && c.shushiwanSrc != "" && strings.Contains(source, c.shushiwanSrc) && !c.shushiwanOKRE.MatchString(final) { if c.shushiwanFireRE.MatchString(final) { flags = append(flags, "DC2 数十万≈several×10^5 rendered as «десятки тысяч» (≈10× under)") } } return len(flags), flags } // --- DC-6: register-lexicon negative-list (ws5.register_checker) ---------------------------------- // lintRegisterLexicon flags whole-word occurrences of a register negative-list lexeme in the FINAL text. // The negative-list is pair langpack DATA (DCCheckerData.RegisterNeg): fairy-tale-Russian / chancery lexemes // that break the xianxia register. Lower-cased; whole-word matched. Empty (a no-pack book) → nothing to flag. func (c *Checkers) lintRegisterLexicon(final string) (int, []string) { if c == nil || len(c.registerNeg) == 0 { return 0, nil } low := []rune(strings.ToLower(final)) hitSet := map[string]bool{} for _, w := range c.registerNeg { wr := []rune(w) for i := 0; i+len(wr) <= len(low); i++ { if !text.RunesEqual(low[i:i+len(wr)], wr) { continue } if (i == 0 || !isCyrLetter(low[i-1])) && (i+len(wr) == len(low) || !isCyrLetter(low[i+len(wr)])) { hitSet[w] = true } } } if len(hitSet) == 0 { return 0, nil } hits := make([]string, 0, len(hitSet)) for w := range hitSet { hits = append(hits, w) } sort.Strings(hits) return len(hits), []string{"DC6 register: fairy-tale Russian lexis out of the xianxia genre: " + strings.Join(hits, ", ")} } // isCyrLetter reports whether r is a Cyrillic letter (the word boundary for the register match). func isCyrLetter(r rune) bool { return unicode.IsLetter(r) && unicode.Is(unicode.Cyrillic, r) } // --- percent-scale checker (成 = tenths) ----------------------------------------------------------- // // In Chinese, 成 is one tenth: 六成 = 60%, 六成六 = 66%. A common error renders this as a decimal FRACTION // instead of a percentage (a ~100× scale error). Precision over recall: it fires only when the source has a // «成[]» (cheng_re, pair data), the output has NO percent form (percent_word / «%»), AND the // output carries a decimal-fraction cue (decimal_fraction_re). Inert when the pair ships no percent patterns. func (c *Checkers) lintPercentScale(source, final string) (int, []string) { if c == nil || c.chengRE == nil || c.decimalFractionRE == nil { return 0, nil } m := c.chengRE.FindStringSubmatch(source) if m == nil { return 0, nil } low := strings.ToLower(final) if (c.percentWord != "" && strings.Contains(low, c.percentWord)) || strings.Contains(final, "%") { return 0, nil // the output uses a percent form — the scale is handled correctly } if !c.decimalFractionRE.MatchString(low) { return 0, nil // no fraction cue — the magnitude was paraphrased, not mis-scaled } tens, _ := dcParseCount(m[1], c.numeral) pct := tens * 10 if m[2] != "" { if ones, ok := dcParseCount(m[2], c.numeral); ok { pct += ones } } return 1, []string{fmt.Sprintf("成-percent: %s成%s = %d%% rendered as a decimal fraction instead of a percentage (~%d%%)", m[1], m[2], pct, pct)} } // --- Latin residue in the Russian output ----------------------------------------------------------- // // A whole Latin WORD left untranslated in the output (e.g. «открыл их again, …»). Language-general (Latin is // not pair data): it splits the output into maximal alphanumeric tokens and flags an all-lowercase Latin // token (leaked prose is lowercase; a capital signals a proper noun/brand) with no digit, ≥ minLatinResidueLen // letters, not a Roman numeral, not on the per-project allowlist. Precision over recall. const minLatinResidueLen = 3 func lintLatinResidue(final string, allow map[string]bool) (int, []string) { hits := map[string]bool{} rs := []rune(final) for i := 0; i < len(rs); { if !isLatinLetterOrDigit(rs[i]) { i++ continue } j := i reject := false // set on any digit or uppercase letter — not a lowercase leaked word for j < len(rs) && isLatinLetterOrDigit(rs[j]) { if (rs[j] >= '0' && rs[j] <= '9') || (rs[j] >= 'A' && rs[j] <= 'Z') { reject = true } j++ } tok := string(rs[i:j]) i = j if !reject && len([]rune(tok)) >= minLatinResidueLen && !isRomanNumeral(tok) && !allow[tok] { hits[tok] = true } } if len(hits) == 0 { return 0, nil } surfaces := make([]string, 0, len(hits)) for s := range hits { surfaces = append(surfaces, s) } sort.Strings(surfaces) return len(surfaces), []string{"Latin word left untranslated in the Russian output: " + strings.Join(surfaces, ", ")} } // isLatinLetterOrDigit reports whether r is an ASCII Latin letter or digit (the alphanumeric-token alphabet). func isLatinLetterOrDigit(r rune) bool { return (r >= 'A' && r <= 'Z') || (r >= 'a' && r <= 'z') || (r >= '0' && r <= '9') } // isRomanNumeral reports whether a lowercase token is a Roman numeral (all chars in ivxlcdm) — «iii» reads // as a numeral, not a leaked word; excluded to hold precision. (Uppercase «II» is already skipped as a cap.) func isRomanNumeral(tok string) bool { for _, r := range tok { switch r { case 'i', 'v', 'x', 'l', 'c', 'd', 'm': default: return false } } return true } // --- broken word forms (target-general) ------------------------------------------------------------ // // Flags a target word ending in a structurally-impossible suffix — for ru, «-йть», which no well-formed // Russian word does (the shape of a mangled infinitive, «войть» for «войти»). The suffix set is TARGET data // (lang.TargetChecks "broken_suffix"), so the ALGORITHM is language-general; a target with no suffix data // flags nothing. Zero false-positive by construction (only a structural signature, no dictionary). func (c *Checkers) lintBrokenWord(final string) (int, []string) { if c == nil || len(c.brokenSuffix) == 0 { return 0, nil } seen := map[string]bool{} var det []string for _, w := range text.TokenizeCyrillic(final) { wl := len([]rune(w)) for _, suf := range c.brokenSuffix { if wl >= 4 && strings.HasSuffix(w, suf) && !seen[w] { seen[w] = true det = append(det, "malformed word ending in «-"+suf+"» (no valid Russian word does): "+w) } } } sort.Strings(det) return len(det), det }