textmachine/backend/internal/pipeline/chunker.go

88 lines
3.3 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

package pipeline
import "strings"
// chunker.go: the Веха-2 segmenter SEAM. Its only job right now is to turn a
// normalized source into an ordered list of (chapter, chunk) units so the runner
// loop, resume and disposition machinery can be exercised over MANY chunks — the
// hardcoded "chapter=1/chunk=0" of Фаза 0 is gone.
//
// This is a deliberate PLACEHOLDER. The real linguistic chunker of шаг 3
// (sentence segmentation, read-only overlap, ruby name-reading extraction for
// ja) replaces SplitChunks WITHOUT touching the loop — it only has to keep the
// Chunk contract and bump chunkerVersion. chunkerVersion is folded into the
// snapshot (render.go), so changing the segmentation is an explicit
// re-translation (a loud --resnapshot), never a silent cache miss (R6/D5.2).
// Chunk is one ordered unit of translation work.
type Chunk struct {
Chapter int // 1-based
ChunkIdx int // 0-based within its chapter
Text string
}
// chapterSep splits the source into chapters. Form feed (U+000C) is the ASCII
// "page/section break" — semantically a chapter boundary, invisible in prose,
// and untouched by NormalizeSource (BOM/CRLF/NFC/outer-trim leave an internal
// \f intact). A source with no \f is a single chapter (Фаза-0 backward compat:
// the one-file example stays one chapter).
const chapterSep = "\f"
// targetChunkChars is the placeholder chunk size (≈ the plan's "12k tokens";
// for a CJK source ~1 char ≈ 1 token). Paragraphs are packed up to this size and
// never split, so the 264-char example remains a single chunk.
const targetChunkChars = 1500
// SplitChunks segments a NORMALIZED source (post-NormalizeSource) into an ordered
// chunk list. Rule: split on \f into 1-based chapters; within a chapter pack
// blank-line paragraphs greedily into chunks of ≤ targetChunkChars, never
// splitting a paragraph (a paragraph larger than the target is its own chunk);
// drop empty pieces. Fully deterministic.
func SplitChunks(source string) []Chunk {
var out []Chunk
chapterNo := 0
for _, chapRaw := range strings.Split(source, chapterSep) {
paras := splitParagraphs(chapRaw)
if len(paras) == 0 {
continue // an empty chapter block does not consume a chapter number
}
chapterNo++
chunkIdx := 0
var buf strings.Builder
flush := func() {
if buf.Len() == 0 {
return
}
out = append(out, Chunk{Chapter: chapterNo, ChunkIdx: chunkIdx, Text: buf.String()})
chunkIdx++
buf.Reset()
}
for _, p := range paras {
// Start a new chunk if adding this paragraph would exceed the target
// and the current chunk is non-empty (never split a paragraph).
if buf.Len() > 0 && buf.Len()+len("\n\n")+len(p) > targetChunkChars {
flush()
}
if buf.Len() > 0 {
buf.WriteString("\n\n")
}
buf.WriteString(p)
}
flush()
}
return out
}
// splitParagraphs breaks a chapter block on blank lines and trims each paragraph,
// dropping empties. Deterministic; whitespace-only paragraphs vanish.
func splitParagraphs(block string) []string {
var paras []string
for _, raw := range strings.Split(block, "\n\n") {
// Collapse a run of blank lines: a "paragraph" that is itself only blank
// lines (from 3+ consecutive newlines) trims to "".
if p := strings.TrimSpace(raw); p != "" {
paras = append(paras, p)
}
}
return paras
}