package pipeline import ( "fmt" "regexp" "sort" "strconv" "strings" "unicode" ) // checkers_zh_ru.go: the WS5 defect-class checkers DC1 (时辰 double-hour units), DC2 (千万/数十万 magnitude // scale) and DC6 (register negative-list) — deterministic, $0 flaggers on the source↔FINAL text, ported // from the frozen ws5_checkers_verify.py. Like the four cheap style gates they are OBSERVABILITY, NEVER a // disposition (research/20: judges too noisy → deterministic gates; a hit is recorded in the retrieval- // state / report, never dropping a chunk). They are tuned PRECISION over recall — DC1/DC2 fire only on an // EXPLICIT src↔target mismatch; the LANDING (whether the flag is trusted) is gated by a §5(д) false- // positive measure on a fresh rerun, not by this build. Their rule VERSION rides cheapGateVersion (folded // into the snapshot as StyleCheckVersion), so editing a rule / the pack is a loud --resnapshot. // // PER-PAIR PACK (слой-2 data, §5(в)): the unit table + negative-list below are the zh-ru convention pack. // They are code consts versioned by cheapGateVersion (the same discipline the existing cheap-gate data // uses), not a config-loaded per-pair file — a config-loaded versioned pack is a cleaner future form // (noted residual). The checkers self-gate on source content (时辰/千万 are zh-specific → no fire on a // non-zh source), and DC6 is target-side (gated behind ru like the other readability flaggers). // --- DC-1: 时辰 (double-hour) unit checker (ws5.shichen_checker) ----------------------------------- // dcCNNum maps the small CJK numerals the 时辰 pattern accepts to their value (1 时辰 = 2 modern hours). var dcCNNum = map[rune]int{ '一': 1, '二': 2, '两': 2, '三': 3, '四': 4, '五': 5, '六': 6, '七': 7, '八': 8, '九': 9, '十': 10, } // dcShichenRE captures the count before 时辰 (an Arabic digit or a small CJK numeral), tolerating 个. var dcShichenRE = regexp.MustCompile(`([0-9一二两三四五六七八九十])\s*个?\s*时辰`) // dcRuHoursRE captures a Russian " час…" rendering; the alternatives mirror the reference map. var dcRuHoursRE = regexp.MustCompile(`(\d+|один|два|двух|три|трёх|трех|четыре|пять|шесть)\s+час`) // dcRuHourWord maps the Russian count words dcRuHoursRE captures to their value. var dcRuHourWord = map[string]int{ "один": 1, "два": 2, "двух": 2, "три": 3, "трёх": 3, "трех": 3, "четыре": 4, "пять": 5, "шесть": 6, } // lintTimeUnits flags a 时辰 (=2h) unit error: N个时辰 rendered as N часов (the count copied as hours) // instead of ~2N hours (三个时辰 → «три часа» should be ~6h). It fires ONLY on an explicit mismatch — a // paraphrase with no hours count is a valid rendering, not a defect (the reference dropped the "no hours // found" branch that false-flagged 1.8%). Pure and deterministic. func lintTimeUnits(source, final string) (int, []string) { m := dcShichenRE.FindStringSubmatch(source) if m == nil { return 0, nil } n, ok := dcParseCount(m[1]) if !ok { return 0, nil } hm := dcRuHoursRE.FindStringSubmatch(final) if hm == nil { return 0, nil // no explicit hours rendering → a valid paraphrase, not a defect } ruNum, ok := dcParseRuHours(hm[1]) if !ok { return 0, nil } expectedHours := n * 2 if ruNum == n && ruNum != expectedHours { return 1, []string{fmt.Sprintf("DC1 时辰: %d个时辰 передано как «%d час…» (счёт как часы) вместо ~%d ч (1 时辰 = 2 ч)", n, ruNum, expectedHours)} } return 0, nil } // dcParseCount parses the DC1 count group: an Arabic digit string or a single small CJK numeral. func dcParseCount(s string) (int, bool) { if v, err := strconv.Atoi(s); err == nil { return v, true } r := []rune(s) if len(r) == 1 { if v, ok := dcCNNum[r[0]]; ok { return v, true } } return 0, false } // dcParseRuHours parses the DC1 hours group: an Arabic digit string or a Russian count word. func dcParseRuHours(s string) (int, bool) { if v, err := strconv.Atoi(s); err == nil { return v, true } if v, ok := dcRuHourWord[s]; ok { return v, true } return 0, false } // --- DC-2: number-scale magnitude checker (ws5.magnitude_checker, 千万 / 数十万) -------------------- // DC2 regexes — a BYTE-FAITHFUL port of ws5.magnitude_checker (_MAG). The ok_re SUPPRESSION guards are // case-INsensitive (the reference passes re.I); the FIRE predicates are case-SENSITIVE (the reference // does NOT pass re.I to the inner тысяч/десятки-тысяч searches). \w is ported as \p{L}* (RE2's \w is // ASCII; the Cyrillic word-continuation the reference intends is letters). var ( dcQianwanOkRE = regexp.MustCompile(`(?i)десят\p{L}* миллион|10\s*000\s*000|10000000`) // 千万 correct render → suppress dcShushiwanOkRE = regexp.MustCompile(`(?i)сотн\p{L}* тысяч|нескольк\p{L}* сот\p{L}* тысяч|[1-9]00\s*000`) // 数十万 correct → suppress dcDesyatkiTysRE = regexp.MustCompile(`десятк\p{L}* тысяч`) // 数十万 under-render (case-SENSITIVE, per reference) ) // lintMagnitudeScale flags a 千万 (10^7) / 数十万 (~several×10^5) magnitude rendered at a WRONG smaller // scale (a разряд error DC2 targets). 千万 → «тысячи» without «миллион» = 10000× under; 数十万 → «десятки // тысяч» = 10× under. A CORRECT rendering anywhere in the chunk (the ok_re guard) SUPPRESSES the flag, so // a chunk that renders 数十万 as «сотни тысяч» does not fire even if «десятки тысяч» appears elsewhere as // an unrelated quantity (the ws5 reference parity — required so the §5(д) FP-measure taken against the // frozen reference matches what ships). ⚠ 千万 is also a stock HYPERBOLE («несметно») whose «тысячи» // rendering is in-register literary, NOT a hard error (§5 A4) — hyperbole-exposed, landing FP-gated §5(д). // Observability only. (Word↔word fraction inversion — 四成四=44% via q4a rule_b1/b2 — is a residual // sub-checker, not built here.) func lintMagnitudeScale(source, final string) (int, []string) { var flags []string if strings.Contains(source, "千万") && !dcQianwanOkRE.MatchString(final) { if strings.Contains(final, "тысяч") && !strings.Contains(final, "миллион") { // case-sensitive, per reference flags = append(flags, "DC2 千万=10^7 передано как «тысячи» (≈10000× занижение) — возможна ошибка разряда (гипербола-риск, §5-A4)") } } if strings.Contains(source, "数十万") && !dcShushiwanOkRE.MatchString(final) { if dcDesyatkiTysRE.MatchString(final) { flags = append(flags, "DC2 数十万≈неск.×10^5 передано как «десятки тысяч» (≈10× занижение)") } } return len(flags), flags } // --- DC-6: register-lexicon negative-list (ws5.register_checker) ---------------------------------- // dcRegisterNegList is the zh-ru register negative-list (слой-2 pack): fairy-tale-Russian / chancery // lexemes that break the xianxia register («терем» in a cultivation novel is a hard register error). The // owner extends this per corpus finding. Lower-cased; whole-word matched. var dcRegisterNegList = []string{"терем", "терема", "тереме", "теремом", "терему", "теремах"} // lintRegisterLexicon flags whole-word occurrences of a register negative-list lexeme in the FINAL text. func lintRegisterLexicon(final string) (int, []string) { low := []rune(strings.ToLower(final)) hitSet := map[string]bool{} for _, w := range dcRegisterNegList { wr := []rune(w) for i := 0; i+len(wr) <= len(low); i++ { if !runesEqual(low[i:i+len(wr)], wr) { continue } if (i == 0 || !isCyrLetter(low[i-1])) && (i+len(wr) == len(low) || !isCyrLetter(low[i+len(wr)])) { hitSet[w] = true } } } if len(hitSet) == 0 { return 0, nil } hits := make([]string, 0, len(hitSet)) for w := range hitSet { hits = append(hits, w) } sort.Strings(hits) return len(hits), []string{"DC6 регистр: сказочно-русская лексика вне жанра сянься: " + strings.Join(hits, ", ")} } // isCyrLetter reports whether r is a Cyrillic letter (the word boundary for the register match). func isCyrLetter(r rune) bool { return unicode.IsLetter(r) && unicode.Is(unicode.Cyrillic, r) }