textmachine/backend/internal/pipeline/chunker.go

395 lines
15 KiB
Go
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

package pipeline
import (
"strings"
"unicode"
)
// chunker.go: the шаг-3a SOURCE segmenter. It turns the per-chapter normalized
// text produced by ingest.go into an ordered list of (chapter, chunk) units for
// the runner loop. It replaced the Веха-2 placeholder (chapter split on \f +
// blank-line paragraph packing to a CHAR budget) with real segmentation: pack
// whole paragraphs to a ~12k-TOKEN budget, and when a single paragraph exceeds
// the budget descend to SENTENCE boundaries — never splitting a sentence.
//
// Contract kept intact for the loop (bookrun.go/chunkrun.go depend on it): SplitChunks emits
// Chunk{Chapter,ChunkIdx,Text}; ChunkIdx resets to 0 per chapter; the function is
// a PURE, deterministic function of its input (no time/rand, no map iteration).
// Its behavior is versioned by chunkerVersion (render.go), folded into the job
// snapshot — so changing the segmentation is a loud --resnapshot re-translation,
// never a silent cache miss (§3.4/R6/D5.2). The chunk target is a CONST covered by
// that version (Р2 = code); were it ever made config-tunable it MUST be folded into
// the top-level snapshot like coverageSnap (§7d), not left to the version string.
//
// IMPORTANT: this sentence splitter is a SEPARATE, independent piece of logic from
// coverage.go's splitSentences. That one is a bit-for-bit port of the Python oracle
// (refusal_bench.py) over the RUSSIAN OUTPUT and is Python-locked (D12 Q1). THIS one
// segments the SOURCE (zh/ja/en), is versioned by chunkerVersion (Р2 = code), and is
// free to be more faithful (closing-bracket absorption, an abbreviation guard). Do
// not couple them.
// Chunk is one ordered unit of translation work.
type Chunk struct {
Chapter int // 1-based
ChunkIdx int // 0-based within its chapter
Text string
}
// targetChunkTokens is the packing budget: paragraphs (or, inside an oversize
// paragraph, sentences) are greedily packed until the WHOLE chunk's estimated
// tokens (EstimateTokens' formula, applied once — see appendChapterChunks) would
// exceed this. ~1500 sits in the plan's "12k токенов по границам абзацев" corridor
// (02-mvp Фаза 1); a chunk of ordinary prose lands well above the coverage gate's
// 500-char floor, so the excision gate is not silently disabled (§7b). A chunk can
// still fall under the floor only when a whole chapter (or its tail) is genuinely
// that short — a documented boundary. A const, not config: the size is a code
// segmentation rule (Р2), covered by chunkerVersion.
const targetChunkTokens = 1500
// SplitChunks segments the ordered, per-chapter NORMALIZED text of a Document
// (ingest.go already applied NormalizeSource and split chapters) into an ordered
// chunk list. Chapters that yield no text (blank/whitespace-only, e.g. an epub
// cover/nav page) do NOT consume a chapter number, so numbering stays dense over
// real content (Фаза-0 backward compat). Fully deterministic.
func SplitChunks(chapters []string) []Chunk {
var out []Chunk
chapterNo := 0
for _, chapText := range chapters {
paras := splitParagraphs(chapText)
if len(paras) == 0 {
continue // an empty chapter does not consume a chapter number
}
chapterNo++
out = appendChapterChunks(out, chapterNo, paras)
}
return out
}
// appendChapterChunks packs one chapter's paragraphs into chunks and appends them.
// Rule: pack whole paragraphs while the running token estimate stays within
// targetChunkTokens (prefer paragraph boundaries); a paragraph that alone exceeds
// the budget is flushed on its own and then SENTENCE-packed (never splitting a
// sentence; a single sentence over budget is its own chunk).
//
// The running budget is tracked as ADDITIVE char-class counts (CJK vs other),
// applying EstimateTokens' formula (and its 16-token floor) ONCE to the whole
// chunk — NOT by summing per-unit EstimateTokens. Summing per-unit floors every
// tiny paragraph up to 16 tokens, so a dialogue chapter of many short lines would
// flush a chunk at ~⅓ its true token content, needlessly dropping real content
// under the coverage gate's 500-char floor (self-review finding). Counting classes
// and flooring once reproduces EstimateTokens(joined text) exactly (the "\n\n"
// separators are whitespace, which EstimateTokens ignores).
func appendChapterChunks(out []Chunk, chapterNo int, paras []string) []Chunk {
chunkIdx := 0
var buf []string // paragraphs accumulated for the current chunk
var bufCJK, bufOther int
flush := func() {
if len(buf) == 0 {
return
}
if text := strings.TrimSpace(strings.Join(buf, "\n\n")); text != "" {
out = append(out, Chunk{Chapter: chapterNo, ChunkIdx: chunkIdx, Text: text})
chunkIdx++
}
buf = buf[:0]
bufCJK, bufOther = 0, 0
}
for _, p := range paras {
pCJK, pOther := tokenClassCounts(p)
if estTokensFrom(pCJK, pOther) > targetChunkTokens {
// Oversize paragraph: close any open chunk at this paragraph boundary,
// then split the paragraph at sentence boundaries into its own chunks.
flush()
for _, sub := range packSentences(p) {
if text := strings.TrimSpace(sub); text != "" {
out = append(out, Chunk{Chapter: chapterNo, ChunkIdx: chunkIdx, Text: text})
chunkIdx++
}
}
continue
}
// Normal paragraph: start a new chunk if adding it would exceed the budget
// and the current chunk is non-empty (never split a paragraph here).
if len(buf) > 0 && estTokensFrom(bufCJK+pCJK, bufOther+pOther) > targetChunkTokens {
flush()
}
buf = append(buf, p)
bufCJK, bufOther = bufCJK+pCJK, bufOther+pOther
}
flush()
return out
}
// packSentences splits one oversize paragraph into sentence segments (lossless
// tiling — concatenation reproduces the paragraph) and packs consecutive segments
// into ≤ targetChunkTokens sub-chunks, never splitting a sentence. A single
// sentence larger than the budget becomes its own (oversize) sub-chunk — the only
// way to honor "never split a sentence". Returns the sub-chunk texts in order. Uses
// the same additive char-class budget as appendChapterChunks (floor applied once).
func packSentences(paragraph string) []string {
segs := splitSourceSentences(paragraph)
var chunks []string
var buf strings.Builder
var bufCJK, bufOther int
flush := func() {
if buf.Len() == 0 {
return
}
chunks = append(chunks, buf.String())
buf.Reset()
bufCJK, bufOther = 0, 0
}
for _, s := range segs {
sCJK, sOther := tokenClassCounts(s)
if buf.Len() > 0 && estTokensFrom(bufCJK+sCJK, bufOther+sOther) > targetChunkTokens {
flush()
}
buf.WriteString(s)
bufCJK, bufOther = bufCJK+sCJK, bufOther+sOther
}
flush()
return chunks
}
// tokenClassCounts counts the two token-estimate character classes of EstimateTokens
// (render.go): CJK ideographs/kana/hangul, and "other" non-space runes (whitespace
// folds into neighbours). Kept separate from EstimateTokens so the packer can sum
// counts ADDITIVELY across units and floor once at the chunk level. Mirrors
// EstimateTokens' classes; a recalibration there that changes packing granularity
// is itself a chunkerVersion bump (both are behaviour changes covered by the snapshot).
func tokenClassCounts(s string) (cjk, other int) {
for _, r := range s {
switch {
case unicode.In(r, unicode.Han, unicode.Hiragana, unicode.Katakana, unicode.Hangul):
cjk++
case unicode.IsSpace(r):
default:
other++
}
}
return cjk, other
}
// estTokensFrom applies EstimateTokens' formula (cjk + other/3, floored at 16) to
// pre-counted class totals — identical to EstimateTokens over the joined text.
func estTokensFrom(cjk, other int) int {
est := cjk + other/3
if est < 16 {
est = 16
}
return est
}
// splitParagraphs breaks a chapter into paragraphs on blank lines and trims each,
// dropping empties. Deterministic; whitespace-only paragraphs vanish. (Ingest emits
// a blank line between block-level epub elements, so paragraphs survive extraction.)
func splitParagraphs(chapter string) []string {
var paras []string
for _, raw := range strings.Split(chapter, "\n\n") {
if p := strings.TrimSpace(raw); p != "" {
paras = append(paras, p)
}
}
return paras
}
// --- source sentence segmentation (zh/ja/en) — chunkerVersion-versioned ---------
// splitSourceSentences segments one paragraph at sentence boundaries, returning
// substrings that TILE the input (their concatenation is exactly the paragraph —
// no character is lost or moved), so a packed sub-chunk is an exact source slice.
// A boundary is a run of sentence terminators, plus any closing quotes/brackets, at:
//
// (1) a CJK terminator (。!?) — zero-width, CJK prose has no inter-sentence space —
// but ONLY at quote-depth 0. A terminator inside a quote is dialogue-internal:
// 「止まれ!」と言った。 is ONE sentence ending at the outer 。, not two (the same
// for zh “…!”)— this is the dialogue case §3.7/D12-Q1 defers for the coverage
// oracle, but our source splitter is free to get it right;
// (2) an ASCII terminator (.!?) or ellipsis (…) — only when followed by whitespace
// or end-of-paragraph (so a decimal/URL dot mid-token does not split, and en
// dialogue "Stop!" She ran. still splits on the space after the quote).
//
// Trailing whitespace after a boundary is attached to the LEFT segment (the next
// segment starts at the next content char), keeping the tiling exact. An
// abbreviation guard suppresses a split after a lone "." following a known
// abbreviation or a single-letter initial (never split "Mr. Smith" / "J. R. R.").
// When unsure it prefers NOT to split (over-merge is safe — units just pack
// together; over-split cuts a sentence, the forbidden case). Unbalanced quotes are
// tolerated (depth floors at 0), erring toward merge.
func splitSourceSentences(paragraph string) []string {
runes := []rune(paragraph)
var segs []string
start, i, depth := 0, 0, 0
for i < len(runes) {
r := runes[i]
if isOpenQuote(r) {
depth++
i++
continue
}
if isCloseQuote(r) {
if depth > 0 {
depth--
}
i++
continue
}
if !isSourceTerminator(r) {
i++
continue
}
// Consume a run of terminators (collapse "。。。" / "!?" into one boundary).
j := i
cjk := false
for j < len(runes) && isSourceTerminator(runes[j]) {
if isCJKTerminator(runes[j]) {
cjk = true
}
j++
}
// The terminator's quote-depth is the depth HERE — before absorbing the
// closing quote that may follow it (「…!」: the is at depth 1, the 」 that
// closes the quote comes after and must not retro-promote it to depth 0).
depthHere := depth
// Absorb a run of closing quotes/brackets that belong to the sentence,
// decrementing quote-depth for the depth-tracked closers among them.
e := j
for e < len(runes) && isClosingBracket(runes[e]) {
if isCloseQuote(runes[e]) && depth > 0 {
depth--
}
e++
}
boundary := false
if cjk {
boundary = depthHere == 0 // dialogue-internal terminators do not end a sentence
} else {
// ASCII terminator / ellipsis: a boundary only outside a tracked quote AND
// before whitespace or EOS. The quote-depth guard must apply here too, not
// just to the CJK branch — otherwise 「Stop. Now.」 over-splits mid-dialogue
// (external-review #4). Straight-quote en ("Stop!" She ran.) is untracked
// (depth 0), so it still splits on the space as before.
boundary = depthHere == 0 && (e >= len(runes) || isASCIISpace(runes[e]))
// Abbreviation guard: a lone "." after an abbreviation/initial is not a end.
if boundary && j-i == 1 && runes[i] == '.' && isAbbrevBefore(runes[start:i]) {
boundary = false
}
}
if !boundary {
i = e
continue
}
// Attach the following whitespace run to the left segment (keeps tiling exact).
w := e
for w < len(runes) && isASCIISpace(runes[w]) {
w++
}
segs = append(segs, string(runes[start:w]))
start, i = w, w
}
if start < len(runes) {
segs = append(segs, string(runes[start:]))
}
return segs
}
// isOpenQuote / isCloseQuote are the DEPTH-TRACKED paired quote/bracket delimiters
// (CJK brackets + curly quotes with distinct open/close glyphs). Straight quotes
// (" ') are NOT tracked — same glyph opens and closes, so depth is undecidable; en
// relies on the whitespace rule instead, which is correct for spaced prose.
func isOpenQuote(r rune) bool {
switch r {
case '「', '『', '', '【', '《', '〈', '', '', '“', '':
return true
}
return false
}
func isCloseQuote(r rune) bool {
switch r {
case '」', '』', '', '】', '》', '〉', '', '', '”', '':
return true
}
return false
}
// isSourceTerminator is the sentence-terminator class for the SOURCE: CJK 。!?,
// ASCII .!?, and the ellipsis … (U+2026).
func isSourceTerminator(r rune) bool {
switch r {
case '。', '', '', '.', '!', '?', '…':
return true
}
return false
}
// isCJKTerminator is the zero-width-splitting subset (fullwidth CJK). The ellipsis
// is deliberately NOT here: "……" mid-CJK-sentence must not split, but "……。" still
// splits on the 。 in the same run.
func isCJKTerminator(r rune) bool {
switch r {
case '。', '', '':
return true
}
return false
}
// isClosingBracket is the closing-quote/bracket class absorbed into the left
// sentence after its terminator (ja/zh brackets + straight/curly quotes).
func isClosingBracket(r rune) bool {
switch r {
case '」', '』', '】', '', ')', '》', '〉', '', ']', '', '}',
'"', '\'', '”', '':
return true
}
return false
}
// isASCIISpace is the whitespace used for ASCII-terminator boundary detection and
// tiling — the ordinary prose separators (space, tab, newline, CR, form feed,
// vertical tab, and the ideographic space U+3000).
func isASCIISpace(r rune) bool {
switch r {
case ' ', '\t', '\n', '\r', '\f', '\v', ' ':
return true
}
return false
}
// sourceAbbrevs are trailing tokens after which a lone "." is treated as an
// abbreviation, not a sentence end (case-insensitive). A pragmatic set for prose;
// the acceptance path is ja→ru where CJK terminators dominate, so this only guards
// the en source. Single-letter initials are handled separately (isAbbrevBefore).
var sourceAbbrevs = map[string]bool{
"mr": true, "mrs": true, "ms": true, "dr": true, "prof": true, "st": true,
"jr": true, "sr": true, "vs": true, "no": true, "vol": true, "ch": true,
"fig": true, "col": true, "gen": true, "sgt": true, "capt": true, "lt": true,
"rev": true, "gov": true, "sen": true, "rep": true, "etc": true, "inc": true,
"ltd": true, "co": true, "mt": true, "ave": true, "rd": true,
}
// isAbbrevBefore reports whether the text immediately before a lone "." ends in a
// known abbreviation or a single-letter initial (so the "." is not a boundary).
func isAbbrevBefore(left []rune) bool {
// Trailing run of ASCII letters (the "word" before the period).
end := len(left)
k := end
for k > 0 && isASCIILetter(left[k-1]) {
k--
}
word := left[k:end]
if len(word) == 0 {
return false
}
// Single-letter initial ("J.") — only when it stands alone (start or preceded by
// a space/opening punctuation), never a one-letter word ending a real sentence
// mid-flow would be rare; over-merge is the safe side.
if len(word) == 1 {
return k == 0 || !isASCIILetter(left[k-1])
}
return sourceAbbrevs[strings.ToLower(string(word))]
}
func isASCIILetter(r rune) bool {
return (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z')
}