textmachine/backend/internal/checks/repair.go

406 lines
17 KiB
Go

package checks
import (
"fmt"
"sort"
"strings"
"unicode"
"textmachine/backend/internal/lang"
)
// repair.go: the POSITION-AWARE half of the defect checkers (pack-16, D39.24) — the same deterministic
// facts the lint functions report, plus WHERE in the text they sit, so a caller can address the minimal
// span instead of the whole unit.
//
// VERDICT-NEUTRAL BY CONSTRUCTION. Not one lint function is touched: this file re-derives the hit with the
// SAME compiled pair patterns and the SAME predicates, and the existing checkers keep returning byte-identical
// counts and details. That is why CheapGateVersion does NOT move (a bump would invalidate every book's
// snapshot and re-bill it, snapshot.go: StyleCheckVersion is folded unconditionally). The equivalence is
// asserted by execution, not by claim — see TestRepairCandidatesAgreeWithLints (a Title/ALL-CAPS token is in
// the corpus since #6 dropped the caps-reject, so the latin twin cannot silently re-drift from its lint).
//
// WHY ONLY SOME CLASSES APPEAR HERE. A candidate needs a TRUSTWORTHY anchor on the side that would be
// rewritten. Three of the shipped defect classes do not have one and are deliberately absent:
// - DC2 (千万/数十万) and the 成-percent class fire on `strings.Contains` over the WHOLE unit
// (checkers.go: qianwanFireWord is the bare substring «тысяч», which also matches «тысячелетие»); a unit
// of a few thousand target characters routinely contains an unrelated cue, so a span derived from the
// first occurrence would point at innocent prose. Cardinality is not alignment.
// - the register, ё, dialogue-dash and coverage classes are either chunk-global judgements or scalar
// ratios, so "the defect is HERE" is not defined for them.
// They remain observability through the ordinary lint path; only the anchored classes are addressable.
// RepairClass identifies an addressable defect class. These are ENGINE identifiers (not pair data): the
// pair supplies the patterns that make a class fire, the target supplies its word lists, but the class
// vocabulary is the engine's, exactly like the flag-reason vocabulary of the driver.
type RepairClass string
const (
// RepairDC1TimeUnits is the counted double-hour defect: N个时辰 rendered as N hours instead of 2N.
RepairDC1TimeUnits RepairClass = "dc1_time_units"
// RepairDC1Fractional is the fractional double-hour defect: 半个时辰 (≈1 h) rendered as half an hour.
RepairDC1Fractional RepairClass = "dc1_fractional"
// RepairLatinResidue is a whole Latin word left untranslated in the target output.
RepairLatinResidue RepairClass = "latin_residue"
// RepairBrokenWord is a target word ending in a structurally impossible suffix («-йть»).
RepairBrokenWord RepairClass = "broken_word"
)
// RepairCandidate is one addressable defect: the class, the byte range in the FINAL text that would be
// rewritten, the byte range in the SOURCE that evidences it (zero for a target-only class), and the
// operator-facing detail. Spans are the NARROW match; widening them to a sentence is ExpandToSentence,
// kept separate so the caller can decide how much context a rewrite needs.
type RepairCandidate struct {
Class RepairClass
DstSpan [2]int
SrcSpan [2]int
Detail string
}
// hasSrc reports whether the candidate carries a source-side anchor.
func (c RepairCandidate) hasSrc() bool { return c.SrcSpan[1] > c.SrcSpan[0] }
// RepairCandidates returns every addressable defect of `final` (translated from `source`), in deterministic
// order (class, then span start). It is PURE — no store, no clock, no LLM — so a resumed run re-derives the
// identical list, and it is inert for a book whose pack/target ships no data, exactly like the lints.
//
// UNIQUENESS GUARD (load-bearing): a class contributes a candidate ONLY when its evidence occurs exactly
// once on each side it consults. The shipped checkers pair the FIRST source match with the FIRST target
// match, which is a heuristic, not an alignment; with two or more matches that heuristic can pair unrelated
// occurrences, so the honest answer is to report no addressable candidate and leave the class as a flag.
// This is a deliberate recall sacrifice: a multi-defect unit stays observability-only.
func RepairCandidates(source, final string, cfg CheapGateConfig) []RepairCandidate {
var out []RepairCandidate
c := cfg.Checkers
if c != nil {
out = append(out, c.dc1Candidates(source, final)...)
}
out = append(out, latinResidueCandidates(final, cfg.Allowlist)...)
if c != nil {
out = append(out, c.brokenWordCandidates(final)...)
}
sort.SliceStable(out, func(i, j int) bool {
if out[i].Class != out[j].Class {
return out[i].Class < out[j].Class
}
return out[i].DstSpan[0] < out[j].DstSpan[0]
})
return out
}
// dc1Candidates yields the counted and the fractional double-hour candidates. Both mirror lintTimeUnits'
// predicates exactly; the counted branch additionally requires that BOTH sides matched exactly once.
func (c *Checkers) dc1Candidates(source, final string) []RepairCandidate {
var out []RepairCandidate
// Fractional probe (checked first, mirroring the lint's order).
if c.halfShichenRE != nil && c.halfShichenFireWord != "" {
src := c.halfShichenRE.FindAllStringIndex(source, -1)
dst := allIndex(final, c.halfShichenFireWord)
if len(src) == 1 && len(dst) == 1 {
out = append(out, RepairCandidate{
Class: RepairDC1Fractional,
SrcSpan: [2]int{src[0][0], src[0][1]},
DstSpan: dst[0],
Detail: fmt.Sprintf("DC1 时辰 (fractional): 半个时辰 ≈ 1 h rendered as %q", c.halfShichenFireWord),
})
return out // the lint returns on the fractional hit too — same single-verdict semantics
}
}
if c.shichenRE == nil || c.ruHoursRE == nil {
return out
}
srcAll := c.shichenRE.FindAllStringSubmatchIndex(source, -1)
dstAll := c.ruHoursRE.FindAllStringSubmatchIndex(final, -1)
if len(srcAll) != 1 || len(dstAll) != 1 {
return out // ambiguous anchor — report nothing addressable rather than guess a pairing
}
n, ok := dcParseCount(group(source, srcAll[0], 1), c.numeral)
if !ok {
return out
}
ruNum, ok := dcParseRuHours(group(final, dstAll[0], 1), c.ruHours)
if !ok {
return out
}
if ruNum != n || ruNum == n*2 {
return out // same fire condition as lintTimeUnits
}
// The target span is the MATCH minus its right-boundary character: the hours pattern consumes one
// non-target-word rune (or end of text) to bound the hour word, and that rune belongs to the surrounding
// prose, not to the defect. Rewriting it away would silently eat a comma or a space. The boundary is the
// TARGET's declared word script, not a hardcoded Cyrillic predicate (строка 79, D39.39).
dstEnd := dstAll[0][1]
if r := lastRune(final[dstAll[0][0]:dstEnd]); r != 0 && !c.isTargetWordLetter(r) && !unicode.IsDigit(r) {
dstEnd -= len(string(r))
}
out = append(out, RepairCandidate{
Class: RepairDC1TimeUnits,
SrcSpan: [2]int{srcAll[0][0], srcAll[0][1]},
DstSpan: [2]int{dstAll[0][0], dstEnd},
Detail: fmt.Sprintf("DC1 时辰: %d个时辰 rendered as %d hours (counted as hours) instead of ~%d h", n, ruNum, n*2),
})
return out
}
// latinResidueCandidates yields one candidate per leaked Latin token, using the SAME token rule as
// lintLatinResidue AFTER defect #6 (D39.39): a maximal alphanumeric run with no inner digit, >=
// minLatinResidueLen runes, not a Roman numeral (folded to lower), not allow-listed (exact OR lower-folded) —
// INCLUDING Title/ALL-CAPS words. That last clause is the sync: #6 dropped the lint's caps-reject, so a
// «Cultivation»/«BANK» leak fires the lint (count 1); this twin used to keep the old caps-reject and produce
// 0 candidates, blinding a future actuator to exactly the class #6 made visible. It now matches, retaining the
// byte offsets the lint discards. Uniqueness is per-TOKEN: each leaked word is its own addressable defect.
func latinResidueCandidates(final string, allow map[string]bool) []RepairCandidate {
var out []RepairCandidate
rs := []rune(final)
byteOf := runeByteOffsets(rs)
for i := 0; i < len(rs); {
if !isLatinLetterOrDigit(rs[i]) {
i++
continue
}
j := i
hasDigit := false // an inner digit (v1, x2) is not a leaked word — mirror the lint, not a caps-reject
for j < len(rs) && isLatinLetterOrDigit(rs[j]) {
if rs[j] >= '0' && rs[j] <= '9' {
hasDigit = true
}
j++
}
tok := string(rs[i:j])
low := strings.ToLower(tok)
if !hasDigit && len([]rune(tok)) >= minLatinResidueLen && !isRomanNumeral(low) && !allow[tok] && !allow[low] {
out = append(out, RepairCandidate{
Class: RepairLatinResidue,
DstSpan: [2]int{byteOf[i], byteOf[j]},
Detail: "Latin word left untranslated in the target output: " + tok,
})
}
i = j
}
return out
}
// brokenWordCandidates yields one candidate per malformed target word, using the SAME suffix rule as
// lintBrokenWord. It cannot reuse the token slice: it needs byte OFFSETS into the ORIGINAL text for the repair
// span, which the bare-word tokenizer discards. It therefore walks the original runes but tokenizes on the
// SHARED target seam — the same word script (c.wordScript) and the same combining-mark folding
// (Checkers.tokenizeWords / text.TokenizeScript) the lint twin uses — so the two can never disagree and the
// verdict-neutrality this file rests on (a candidate only where the lint fires) holds by construction.
//
// TARGET-SCRIPT DEBT REPAID (D39.24 §15.3 amended by D39.62/П1): the word boundary was unicode.Cyrillic in
// BOTH the lint tokenizer and here; it is now the target's DECLARED word script (data), so a non-Cyrillic
// target drives the class from its own data, no Go edit. The debt was repayable only at the shared tokenizer,
// as one deliberate change to the target seam — this is that change, made on both consumers at once.
func (c *Checkers) brokenWordCandidates(final string) []RepairCandidate {
if c.wordScript == nil || len(c.brokenSuffix) == 0 {
return nil
}
var out []RepairCandidate
rs := []rune(final)
byteOf := runeByteOffsets(rs)
for i := 0; i < len(rs); {
if !unicode.Is(c.wordScript, rs[i]) {
i++
continue
}
// One word run: word-script runes ARE the word; a combining mark (Mn) stays inside the run (so the
// span covers the whole visual word) but is dropped from the comparison form — byte-for-byte the token
// tokenizeWords/TokenizeScript build, so «сло́во» reads as «слово» in BOTH (#11).
j := i
var wb strings.Builder
for j < len(rs) && (unicode.Is(c.wordScript, rs[j]) || unicode.Is(unicode.Mn, rs[j])) {
if !unicode.Is(unicode.Mn, rs[j]) {
wb.WriteRune(unicode.ToLower(rs[j]))
}
j++
}
word := wb.String()
if len([]rune(word)) >= 4 {
for _, suf := range c.brokenSuffix {
if strings.HasSuffix(word, suf) {
out = append(out, RepairCandidate{
Class: RepairBrokenWord,
DstSpan: [2]int{byteOf[i], byteOf[j]},
Detail: "malformed word ending in «-" + suf + "»: " + word,
})
break
}
}
}
i = j
}
return out
}
// ExpandToSentence widens a byte span to the sentence(s) containing it, using the SHARED terminator data
// (the same table the chunker and the coverage gate segment on), so the widening is pair-agnostic: a target
// whose terminators are not in the data simply widens to the whole text, which the caller's size guard then
// rejects. Returns the widened span, clamped to the text and aligned to rune boundaries.
func ExpandToSentence(text string, span [2]int) [2]int {
if span[0] < 0 || span[1] > len(text) || span[0] >= span[1] {
return span
}
term := lang.DefaultTerminators()
start := 0
for i, r := range text {
if i >= span[0] {
break
}
if term.IsTerminator(r) {
start = i + len(string(r))
}
}
for start < span[0] && isSpaceByte(text[start]) {
start++
}
end := len(text)
for i, r := range text {
if i < span[1] {
continue
}
if term.IsTerminator(r) {
end = i + len(string(r))
break
}
if r == '\n' { // a paragraph break bounds the span even without a terminator
end = i
break
}
}
return [2]int{start, end}
}
// DisjointCandidates expands every candidate to its sentence and drops any whose expanded span INTERSECTS an
// already-accepted one, keeping the first in deterministic order. Overlap is the common case rather than a
// corner: two classes firing inside one sentence expand to the SAME span, and splicing two replacements into
// one range would cut an already-mutated string — at best duplicating a fragment, at worst slicing out of
// bounds or mid-rune in a paid run. Dropping happens BEFORE any call is made, so an unusable candidate is
// never billed for.
func DisjointCandidates(final string, cands []RepairCandidate) []RepairCandidate {
var out []RepairCandidate
for _, c := range cands {
c.DstSpan = ExpandToSentence(final, c.DstSpan)
overlaps := false
for _, kept := range out {
if c.DstSpan[0] < kept.DstSpan[1] && kept.DstSpan[0] < c.DstSpan[1] {
overlaps = true
break
}
}
if !overlaps {
out = append(out, c)
}
}
return out
}
// --- small helpers -------------------------------------------------------------
// group returns submatch n of a FindAllStringSubmatchIndex row, or "" when the group did not participate.
func group(s string, idx []int, n int) string {
if len(idx) < 2*n+2 || idx[2*n] < 0 {
return ""
}
return s[idx[2*n]:idx[2*n+1]]
}
// allIndex returns the byte spans of every occurrence of sub in s (non-overlapping, left to right).
func allIndex(s, sub string) [][2]int {
var out [][2]int
for off := 0; ; {
i := strings.Index(s[off:], sub)
if i < 0 {
return out
}
out = append(out, [2]int{off + i, off + i + len(sub)})
off += i + len(sub)
}
}
// runeByteOffsets maps rune index → byte offset, with a final entry for the end of text.
func runeByteOffsets(rs []rune) []int {
out := make([]int, len(rs)+1)
b := 0
for i, r := range rs {
out[i] = b
b += len(string(r))
}
out[len(rs)] = b
return out
}
// lastRune returns the final rune of s, or 0 for the empty string.
func lastRune(s string) rune {
var last rune
for _, r := range s {
last = r
}
return last
}
func isSpaceByte(b byte) bool { return b == ' ' || b == '\t' || b == '\n' || b == '\r' }
// --- positive post-condition helpers (pack-16 §15.2 A) ---------------------------------------------
//
// A repair is accepted only when the class invariant is RESTORED, not merely silenced. These helpers expose
// exactly the predicates the driver's guard needs, over the SAME compiled pair/target data the detectors
// use, so a new pair inherits them without touching Go.
// LatinResidueCount counts leaked Latin tokens of ANY case — after defect #6 the caps-reject is gone, so BANK
// and Cultivation count alongside lowercase ones (allowlist-free — the guard compares a span with its
// replacement, and an allow-listed surface is equally allowed on both sides).
func LatinResidueCount(s string) int {
n, _ := lintLatinResidue(s, nil)
return n
}
// BrokenWordCount counts malformed target words in s (0 when the target ships no suffix data).
func BrokenWordCount(c *Checkers, s string) int {
n, _ := c.lintBrokenWord(s)
return n
}
// FractionalUnitPresent reports whether the fractional-unit FIRE word is still present — the defect the
// fractional class repairs. Inert (false) when the pair ships no fractional probe.
func FractionalUnitPresent(c *Checkers, s string) bool {
if c == nil || c.halfShichenFireWord == "" {
return false
}
return strings.Contains(s, c.halfShichenFireWord)
}
// MentionsHourWord reports whether s still STATES a duration in hours — the positive half of a
// fractional-unit repair: the replacement must CONVERT the duration, not delete it. The probe is PAIR DATA
// (`hour_word_re`), not a Go literal, and it carries BOTH word boundaries: without a left boundary the stem
// matches inside «полчаса»/«тотчас»/«сейчас», so a reply that deletes the duration and leaves any «-час»
// filler would satisfy the condition it is supposed to enforce.
//
// A pair that ships the fractional probe but no hour word CANNOT have this invariant asserted, so this
// returns false (reject) rather than true (accept). The runner refuses to enable the class in that state at
// LOAD time, so the rejecting branch is a backstop, not the operating mode.
func MentionsHourWord(c *Checkers, s string) bool {
if c == nil || c.hourWordRE == nil {
return false
}
return c.hourWordRE.MatchString(s)
}
// HasHourWordProbe reports whether the pair ships the positive post-condition the fractional class needs.
func HasHourWordProbe(c *Checkers) bool { return c != nil && c.hourWordRE != nil }
// HoursCountDoubled reports whether the replacement states a duration equal to TWICE the source count the
// counted-hours detector found in the ORIGINAL span. This is the positive assertion the class needs: a reply
// that changes the wrong number to a DIFFERENT wrong number silences the detector but fails here.
func HoursCountDoubled(c *Checkers, original, reply string) bool {
if c == nil || c.ruHoursRE == nil {
return false
}
om := c.ruHoursRE.FindStringSubmatch(original)
rm := c.ruHoursRE.FindStringSubmatch(reply)
if om == nil || rm == nil {
return false // no parsable hours count on one side — cannot assert the invariant, so do not accept
}
oldNum, ok1 := dcParseRuHours(om[1], c.ruHours)
newNum, ok2 := dcParseRuHours(rm[1], c.ruHours)
return ok1 && ok2 && newNum == oldNum*2
}