package pipeline import ( "archive/zip" "bytes" "encoding/xml" "fmt" "io" "net/url" "os" "path" "regexp" "strings" "unicode" "unicode/utf8" "golang.org/x/text/encoding/simplifiedchinese" xunicode "golang.org/x/text/encoding/unicode" "golang.org/x/text/transform" ) // ingest.go: the шаг-3a import layer. Ingest turns a source file (txt or epub) // into an ordered list of per-chapter NORMALIZED text plus the ruby/furigana // readings captured on the way (04-unhappy §4 / D9). It is the ONLY place that // reads the source; the runner then feeds doc.Chapters to SplitChunks and persists // doc.Ruby (seeding.go). It is offline and deterministic — no LLM, no time/rand — so // the whole path is $0 and reproducible. // // epub v1 is a text extraction (02-mvp:25): read the chapters in spine order, // strip tags, drop inline markup/styling. The ONE thing we do NOT drop is ruby: // basereading carries an author reading of a name/term that a // plain text export would silently lose (the documented hole, 04-unhappy §4). We // keep the BASE in the body and hand the READING to the caller for a glossary lock // (memory v2, шаг 4) — never injecting it into a prompt here (§7d). // chapterSep splits a TXT source into chapters. Form feed (U+000C) is the ASCII // "page/section break": semantically a chapter boundary, invisible in prose, and // untouched by NormalizeSource. A txt with no \f is a single chapter (Фаза-0 backward // compat: the one-file example stays one chapter). epub carries real chapter // structure in its spine, so it does not use this. const chapterSep = "\f" // RubyReading is one captured (base, reading) pair and the chapter it appeared in. // The pipeline aggregates these (min chapter, count) before persisting (seeding.go). type RubyReading struct { Base string // the ruby body: the kanji/base surface form Reading string // the reading (furigana) Chapter int // 1-based chapter where this occurrence was found } // Document is the ingested source: per-chapter normalized text (in reading order) // plus every ruby occurrence found. Chapters is what SplitChunks consumes. type Document struct { Chapters []string Ruby []RubyReading } // Ingest reads a source file with automatic encoding detection (the txt path), source language // unspecified. Dispatch is by extension: .epub → the epub reader; anything else → plain text. Kept // as the terse form used by tests and the epub path. func Ingest(p string) (*Document, error) { return IngestEncoded(p, "", "") } // IngestEncoded reads a source file and returns its Document, decoding a txt source per the book's // declared encoding (auto|utf8|gb18030; "" == auto) and source language (zh|ja|en; "" == unknown). // sourceLang scopes the GB18030 auto-probe to Chinese sources — a Shift-JIS/EUC-JP ja file would // otherwise decode to Han-shaped mojibake that passes the plausibility guard (self-review critical). // Dispatch is by extension: .epub → the epub reader (its own per-document charset handling); // anything else → plain text. func IngestEncoded(p, encoding, sourceLang string) (*Document, error) { if strings.EqualFold(extOf(p), ".epub") { return ingestEPUB(p) } return ingestTXT(p, encoding, sourceLang) } // extOf returns the lower-cased file extension incl. the dot ("" if none). func extOf(p string) string { i := strings.LastIndexByte(p, '.') if i < 0 || strings.ContainsAny(p[i:], "/\\") { return "" } return strings.ToLower(p[i:]) } // ingestTXT reads a plain-text source: decode the raw bytes to UTF-8 per `encoding`, split on \f // into chapters, NormalizeSource each (BOM/CRLF/NFC/outer-trim). No ruby in txt. Encoding decode // runs BEFORE NormalizeSource so the whole downstream path (chunker fuzz-invariant, memory keys) // sees valid UTF-8; a decode failure is a LOUD error, never silent U+FFFD corruption. func ingestTXT(p, encoding, sourceLang string) (*Document, error) { raw, err := os.ReadFile(p) if err != nil { return nil, fmt.Errorf("pipeline: ingest txt: %w", err) } text, err := decodeSourceBytes(raw, encoding, sourceLang) if err != nil { return nil, fmt.Errorf("pipeline: ingest txt %s: %w", p, err) } var chapters []string // Honor the ASCII form feed as a hard chapter boundary (Фаза-0 example), THEN split each part // on CJK chapter-header lines (第N章/节/回) — real zh/ja web novels mark chapters that way, not // with \f (the 蛊真人 acceptance book uses «第N节», ~2283 of them). A part with neither stays one // chapter (backward-compat). for _, part := range strings.Split(text, chapterSep) { for _, chap := range splitTextChapters(part) { chapters = append(chapters, NormalizeSource(chap)) } } return &Document{Chapters: chapters}, nil } // --- CJK chapter-header splitting (D18: real zh/ja txt mark chapters as «第N章/节/回») --------- // chapterUnitRunes are the section-level chapter markers auto-detected in a txt. 卷 (volume) is // deliberately EXCLUDED — it is coarser than a chapter and would carve a tiny title-only "chapter". var chapterUnitRunes = []rune{'章', '节', '節', '回'} // chapterNumeralRE matches the numeral run of a chapter header: Arabic (half/fullwidth) or CJK. var chapterNumeralRE = regexp.MustCompile(`^\s*第[0-90-9〇零一二三四五六七八九十百千两兩]+`) // chapterHeaderMaxRunes bounds a header line so a prose sentence that merely opens with «第三节…» // (a longer line) is not mistaken for a header. The 蛊真人 headers are ≤23 runes; 60 leaves room for // a long subtitle while still excluding a full prose sentence (self-review: 40 could miss a long // legitimate subtitle → merged chapters). const chapterHeaderMaxRunes = 60 // isCJKChapterHeader reports whether a single line is a chapter header for the given unit rune. It // requires: a short trimmed line (≤60 runes); a leading 第; the unit rune immediately // after the numerals; and — critically — the char AFTER the unit is a SEPARATOR (whitespace / ::、,. // / dash) or end-of-line, NOT a content glyph. The separator guard is the precision fix (self- // review): 回 is a common measure word, so prose like «第一回见面…» glues a content char (见) right // after the unit and must NOT split; a real header writes «第一节:…» or «第1章 …» with a separator. // (A glued-title header «第一章天空…» without a separator is not detected — rare; a recall trade for // no false splits. Deterministic and pure.) func isCJKChapterHeader(line string, unit rune) bool { t := strings.TrimSpace(line) if utf8.RuneCountInString(t) == 0 || utf8.RuneCountInString(t) > chapterHeaderMaxRunes { return false } loc := chapterNumeralRE.FindStringIndex(t) if loc == nil { return false } rest := t[loc[1]:] // the runes right after the numerals r, sz := utf8.DecodeRuneInString(rest) if r != unit { return false } after := rest[sz:] if after == "" { return true // the header is exactly 第N章 } nr, _ := utf8.DecodeRuneInString(after) return isHeaderSeparator(nr) } // isHeaderSeparator reports whether a rune separates a chapter number from its title (so the line is // a header, not a prose sentence that continues with a content glyph after 第N章). Anything that is // NOT a letter / ideograph / kana / digit counts as a separator (space, ::、,,.。-—— etc.). func isHeaderSeparator(r rune) bool { if unicode.IsLetter(r) || unicode.IsDigit(r) || unicode.Is(unicode.Han, r) || unicode.Is(unicode.Hiragana, r) || unicode.Is(unicode.Katakana, r) { return false } return true } // detectChapterUnit picks the section marker that appears most as a header line (≥2 to avoid a // single stray match). Returns 0 when no unit qualifies (→ the part stays a single chapter). func detectChapterUnit(lines []string) rune { best, bestN := rune(0), 0 for _, unit := range chapterUnitRunes { n := 0 for _, ln := range lines { if isCJKChapterHeader(ln, unit) { n++ } } if n >= 2 && n > bestN { best, bestN = unit, n } } return best } // splitTextChapters splits one text block on its dominant CJK chapter-header lines. The header line // STARTS its chapter (the title is kept). Any preamble before the FIRST header is merged into // chapter 1 so «第一节» == chapter 1 (dense numbering aligns with the section numbers). A block with // no detectable headers is returned unchanged as a single chapter. func splitTextChapters(text string) []string { norm := strings.ReplaceAll(text, "\r\n", "\n") norm = strings.ReplaceAll(norm, "\r", "\n") // bare-CR (old Mac) lines too, so a CR-only header is seen lines := strings.Split(norm, "\n") unit := detectChapterUnit(lines) if unit == 0 { return []string{text} } var chapters []string var cur []string seenHeader := false for _, ln := range lines { if isCJKChapterHeader(ln, unit) { if seenHeader && hasNonBlank(cur) { chapters = append(chapters, strings.Join(cur, "\n")) cur = nil } seenHeader = true } cur = append(cur, ln) } if hasNonBlank(cur) { chapters = append(chapters, strings.Join(cur, "\n")) } if len(chapters) == 0 { return []string{text} } return chapters } // hasNonBlank reports whether any line in the slice has non-whitespace content. func hasNonBlank(lines []string) bool { for _, ln := range lines { if strings.TrimSpace(ln) != "" { return true } } return false } // --- source encoding detection (Task 2 / D18) ---------------------------------- // // Real zh .txt are frequently GB18030 (a superset of GBK/GB2312), not UTF-8 — the 蛊真人 acceptance // book is exactly this case. decodeSourceBytes converts raw source bytes to a UTF-8 string per the // book's declared encoding, defaulting to a conservative auto-detect. The invariant is fail-LOUD: // any doubt is an error, never silent mojibake — a corrupt/foreign byte must never reach a U+FFFD // replacement in ch.Text (which would defeat the lossless-chunker fuzz invariant and mis-key the // glossary). Dependency: golang.org/x/text/encoding (already in go.mod for NFC) — first use of its // simplifiedchinese + unicode subpackages (noted in the session journal). func decodeSourceBytes(raw []byte, encoding, sourceLang string) (string, error) { switch enc := strings.ToLower(strings.TrimSpace(encoding)); enc { case "", "auto": return autoDecodeSource(raw, sourceLang) case "utf8", "utf-8": b := bytes.TrimPrefix(raw, utf8BOM) if !utf8.Valid(b) { return "", fmt.Errorf("declared encoding utf8 but the bytes are not valid UTF-8 (declare gb18030, or convert the file)") } return checkNoNUL(string(b)) case "gb18030", "gbk", "gb2312": // Operator asserted the encoding: decode strictly (fail on undecodable bytes / U+FFFD) but // skip the Chinese-plausibility heuristic — they told us it is GB18030. return decodeGB18030(raw, false) default: return "", fmt.Errorf("unknown encoding %q in book.yaml (use auto|utf8|gb18030)", encoding) } } // checkNoNUL rejects a decoded string that contains a NUL (U+0000). Real prose never contains NUL; // its presence means a mis-detected / mis-declared encoding: a UTF-16 stream WITHOUT a BOM whose // ASCII half is valid UTF-8 with interleaved NULs, OR a UTF-16/binary file wrongly fed to the // GB18030/UTF-16 decoder (0x00 decodes to a VALID U+0000, not U+FFFD, so the replacement-char check // cannot catch it — self-review major, D20.4). Applied on EVERY decode path (utf8/gb18030/utf16), // not just UTF-8, so the guard is symmetric across encodings. func checkNoNUL(s string) (string, error) { if strings.IndexByte(s, 0) >= 0 { return "", fmt.Errorf("decoded text contains NUL (U+0000) — the source is likely UTF-16 without a BOM, or a binary/mis-declared file; convert it to UTF-8 or declare its encoding correctly") } return s, nil } var ( utf8BOM = []byte{0xEF, 0xBB, 0xBF} utf16LEBOM = []byte{0xFF, 0xFE} utf16BEBOM = []byte{0xFE, 0xFF} ) // autoDecodeSource is the conservative detection ladder: BOM → valid UTF-8 (+NUL guard) → GB18030 // probe (ZH sources only) → fail-loud. Order matters: a BOM is definitive; valid UTF-8 is // unambiguous (but rejected if it carries interleaved NUL — UTF-16 without a BOM); GB18030 is tried // only when the bytes are NOT valid UTF-8 AND the source is Chinese (or unspecified). GB18030 is a // Chinese encoding that maps almost every byte sequence to Han-shaped codepoints, so a ja Shift-JIS // / EUC-JP or a European Latin-1 file would decode to CJK-shaped MOJIBAKE that the plausibility // guard cannot distinguish from real Chinese (self-review CRITICAL). So we auto-probe GB18030 only // for zh; for ja/en a non-UTF-8 file fails loud and the operator must convert or declare it. func autoDecodeSource(raw []byte, sourceLang string) (string, error) { switch { case bytes.HasPrefix(raw, utf8BOM): b := raw[len(utf8BOM):] if !utf8.Valid(b) { return "", fmt.Errorf("file has a UTF-8 BOM but the body is not valid UTF-8 (corrupt file)") } return checkNoNUL(string(b)) case bytes.HasPrefix(raw, utf16LEBOM): return decodeUTF16(raw, xunicode.LittleEndian) case bytes.HasPrefix(raw, utf16BEBOM): return decodeUTF16(raw, xunicode.BigEndian) } if utf8.Valid(raw) { return checkNoNUL(string(raw)) // no BOM, already UTF-8 (covers en/ja/UTF-8 zh); NUL ⇒ UTF-16-no-BOM } // Not valid UTF-8. GB18030 auto-probe is scoped to Chinese sources (see doc above). zhScoped := sourceLang == "" || strings.EqualFold(sourceLang, "zh") if zhScoped { if s, err := decodeGB18030(raw, true); err == nil { return s, nil } } probe := "was tried and failed (truncated/corrupt or not GB18030)" if !zhScoped { probe = fmt.Sprintf("is restricted to zh sources (this source is %q) — a Shift-JIS/EUC-JP/Latin-1 file would decode to Han-shaped mojibake and is NOT auto-detected", sourceLang) } return "", fmt.Errorf("could not auto-detect the source encoding: the bytes are not valid UTF-8, and the GB18030 probe %s; convert the file to UTF-8 or declare `encoding:` explicitly in book.yaml", probe) } // decodeGB18030 decodes raw as GB18030 (superset of GBK/GB2312). It is strict: a decode error, or a // U+FFFD replacement char in the output (undecodable/truncated bytes), is a failure — broken bytes // must never reach ch.Text. When probe is true (auto-detect) it additionally requires the decoded // text to look like Chinese (plausibleCJK), because GB18030 maps almost every byte sequence to // SOMETHING, so a Latin-1/other file decodes without error into mojibake unless we sanity-check it. func decodeGB18030(raw []byte, probe bool) (string, error) { out, _, err := transform.Bytes(simplifiedchinese.GB18030.NewDecoder(), raw) if err != nil { return "", fmt.Errorf("GB18030 decode error (not valid GB18030): %w", err) } s := string(out) if strings.ContainsRune(s, utf8.RuneError) { return "", fmt.Errorf("GB18030 decode produced replacement chars (U+FFFD) — bytes are truncated/corrupt or not GB18030") } if probe { if err := plausibleCJK(s); err != nil { return "", err } } return checkNoNUL(s) // 0x00 bytes decode to a valid U+0000, not U+FFFD — a mis-declared UTF-16/binary file slips the replacement-char check (D20.4) } // decodeUTF16 decodes a BOM-prefixed UTF-16 stream (ExpectBOM consumes the BOM). Strict: a decode // error or a U+FFFD in the output is a failure. func decodeUTF16(raw []byte, endian xunicode.Endianness) (string, error) { dec := xunicode.UTF16(endian, xunicode.ExpectBOM).NewDecoder() out, _, err := transform.Bytes(dec, raw) if err != nil { return "", fmt.Errorf("UTF-16 decode error: %w", err) } s := string(out) if strings.ContainsRune(s, utf8.RuneError) { return "", fmt.Errorf("UTF-16 decode produced replacement chars (U+FFFD) — corrupt") } return checkNoNUL(s) // a genuine UTF-16 stream carrying embedded NULs (or a mis-detected binary) is caught here (D20.4) } // plausibleCJK is the GB18030 auto-accept guard: a real GB18030 source is Chinese, so the decoded // text must be (a) almost entirely CJK-plausible runes (Han/kana/CJK-punct/ASCII/whitespace) and // (b) genuinely Han-dense. Latin-1/other mojibake fails (b) — it decodes to sparse Han among ASCII // (café → "caf" + 轳), so every rune is "plausible" in isolation yet Han density is ~0. The Han // floor is intentionally low (5%) so a heavily Latin-mixed but real Chinese source still passes, // while mojibake (Han≈0) is rejected loudly. Thresholds bias toward fail-loud (D18 invariant); a // false reject is recoverable (declare `encoding: gb18030`), a false accept is silent corruption. func plausibleCJK(s string) error { var total, han, plausible int for _, r := range s { total++ switch { case unicode.Is(unicode.Han, r): han++ plausible++ case isPlausibleSourceRune(r): plausible++ } } if total == 0 { return fmt.Errorf("GB18030 probe: empty decode") } if pf := float64(plausible) / float64(total); pf < 0.90 { return fmt.Errorf("GB18030 probe: only %.1f%% of decoded runes are CJK-plausible — likely mojibake, not Chinese", 100*pf) } if hf := float64(han) / float64(total); hf < 0.05 { return fmt.Errorf("GB18030 probe: Han density %.2f%% is too low for a Chinese source — likely a mis-detected encoding; declare `encoding` explicitly if this really is GB18030", 100*hf) } return nil } // isPlausibleSourceRune reports whether a rune belongs to the expected repertoire of a zh/ja source // (excluding Han, which the caller counts separately): ASCII text, kana, CJK/general punctuation, // and fullwidth/halfwidth forms. Deliberately tight — it must NOT whitelist the Latin-1 supplement // or arbitrary symbols, or Latin-1 mojibake would pass the plausibility fraction. func isPlausibleSourceRune(r rune) bool { switch { case r == '\n', r == '\r', r == '\t', r == ' ': return true case r >= 0x20 && r <= 0x7E: // printable ASCII return true case unicode.Is(unicode.Hiragana, r), unicode.Is(unicode.Katakana, r): return true case r >= 0x3000 && r <= 0x303F: // CJK symbols & punctuation (。、「」〜…) return true case r >= 0x2010 && r <= 0x206F: // general punctuation (— – ‘’ “” … ‰ etc.) return true case r >= 0xFF00 && r <= 0xFFEF: // halfwidth & fullwidth forms (!?()ア fullwidth digits/letters) return true case r == 0x30FC || r == 0x30FB: // prolonged-sound mark ー, middle dot ・ (also in Katakana block) return true } return false } // --- epub ---------------------------------------------------------------------- // containerXML is META-INF/container.xml: it points at the OPF package file. Tags // are matched by LOCAL name (no namespace in the struct tags), so a namespaced or // prefixed container still binds. type containerXML struct { Rootfiles []struct { FullPath string `xml:"full-path,attr"` MediaType string `xml:"media-type,attr"` } `xml:"rootfiles>rootfile"` } // opfPackage is the OPF: the manifest (id→href→media-type) and the spine (the // reading order of itemref idrefs). type opfPackage struct { Manifest []struct { ID string `xml:"id,attr"` Href string `xml:"href,attr"` MediaType string `xml:"media-type,attr"` } `xml:"manifest>item"` Spine []struct { IDRef string `xml:"idref,attr"` } `xml:"spine>itemref"` } // ingestEPUB reads chapters in spine order and captures ruby. Every structural // problem (bad zip, missing container/opf, empty spine, no readable chapter) is a // loud error, never a panic (a truncated/foreign epub must not crash a run). func ingestEPUB(p string) (*Document, error) { zr, err := zip.OpenReader(p) if err != nil { return nil, fmt.Errorf("pipeline: ingest epub: open zip %s: %w", p, err) } defer zr.Close() files := map[string]*zip.File{} for _, f := range zr.File { files[path.Clean(f.Name)] = f } // 1) container.xml → OPF path. cdata, err := readZipEntry(files, "META-INF/container.xml") if err != nil { return nil, fmt.Errorf("pipeline: ingest epub: %w", err) } var container containerXML if err := xml.Unmarshal(cdata, &container); err != nil { return nil, fmt.Errorf("pipeline: ingest epub: parse container.xml: %w", err) } opfPath := "" for _, rf := range container.Rootfiles { if strings.TrimSpace(rf.FullPath) != "" { opfPath = path.Clean(rf.FullPath) break } } if opfPath == "" { return nil, fmt.Errorf("pipeline: ingest epub: container.xml has no rootfile full-path") } // 2) OPF → manifest + spine. odata, err := readZipEntry(files, opfPath) if err != nil { return nil, fmt.Errorf("pipeline: ingest epub: %w", err) } var opf opfPackage if err := xml.Unmarshal(odata, &opf); err != nil { return nil, fmt.Errorf("pipeline: ingest epub: parse opf %s: %w", opfPath, err) } if len(opf.Spine) == 0 { return nil, fmt.Errorf("pipeline: ingest epub: opf %s has an empty spine", opfPath) } byID := map[string]struct{ href, mediaType string }{} for _, it := range opf.Manifest { byID[it.ID] = struct{ href, mediaType string }{it.Href, it.MediaType} } opfDir := path.Dir(opfPath) // 3) Read each spine document in order (= one chapter). A dangling idref or a // non-(x)html item is skipped, not fatal; a spine that yields zero readable // content documents is a loud error. doc := &Document{} denseNo := 0 // matches SplitChunks: only a NON-empty chapter takes a number read := 0 var dangling []string for _, ref := range opf.Spine { item, ok := byID[ref.IDRef] if !ok { // A spine itemref is a DECLARED reading-order document; a dangling idref // (no manifest item) is a LOST chapter, not a skip. Silently continuing // would drop its content AND shift every later chapter's dense number // (since_ch for memory v2). Collect all, fail loud after the loop — the // same fail-loud a missing zip entry already gets (external-review #1). dangling = append(dangling, ref.IDRef) continue } if !isXHTML(item.mediaType, item.href) { continue // a genuine non-content asset in the spine (image/css) — no text } entry := resolveHref(opfDir, item.href) data, err := readZipEntry(files, entry) if err != nil { return nil, fmt.Errorf("pipeline: ingest epub: spine item %q → %s: %w", ref.IDRef, entry, err) } read++ text, ruby, err := extractXHTML(data) if err != nil { return nil, fmt.Errorf("pipeline: ingest epub: extract %s: %w", entry, err) } norm := NormalizeSource(text) doc.Chapters = append(doc.Chapters, norm) // The chapter number ruby carries MUST equal the number SplitChunks assigns // (dense — empty chapters are skipped), otherwise memory v2's since_ch is off // by the count of empty spine items (cover/nav) before the reading. A // ruby-bearing chapter is never empty (its base sits in the body), so denseNo // is well-defined here; empty chapters carry no ruby. if len(splitParagraphs(norm)) == 0 { continue } denseNo++ for i := range ruby { ruby[i].Chapter = denseNo } doc.Ruby = append(doc.Ruby, ruby...) } if len(dangling) > 0 { return nil, fmt.Errorf("pipeline: ingest epub: spine references %d manifest id(s) with no manifest item %v — a chapter would be silently lost", len(dangling), dangling) } if read == 0 { return nil, fmt.Errorf("pipeline: ingest epub: spine resolved to no readable (x)html chapters") } return doc, nil } // isXHTML reports whether a manifest item is an (x)html content document. It // checks the media-type (parameters like "; charset=utf-8" stripped), accepting // the (x)html types AND the generic XML types real epubs mislabel xhtml chapters // with (application/xml, text/xml), then falls back to a content-document href // extension for ANY unrecognized/absent type. A genuine non-content asset keeps its // own extension (.jpg/.css/.ncx/.svg), so the fallback does not misclassify it. // (External-review #2: a chapter typed application/xml with a .xml href used to slip // past BOTH the type switch and a .xhtml-only extension fallback and vanish silently.) func isXHTML(mediaType, href string) bool { mt := strings.ToLower(strings.TrimSpace(mediaType)) if i := strings.IndexByte(mt, ';'); i >= 0 { mt = strings.TrimSpace(mt[:i]) } switch mt { case "application/xhtml+xml", "text/html", "application/html", "application/xml", "text/xml": return true } switch extOf(href) { case ".xhtml", ".html", ".htm", ".xml": return true } return false } // resolveHref joins an OPF-relative href to the OPF directory (percent-decoded, // slash-cleaned) into a zip entry path. func resolveHref(opfDir, href string) string { if h, err := url.PathUnescape(href); err == nil { href = h } // Drop any fragment (chapter.xhtml#frag). if i := strings.IndexByte(href, '#'); i >= 0 { href = href[:i] } return path.Clean(path.Join(opfDir, href)) } // readZipEntry reads a zip entry by cleaned path. func readZipEntry(files map[string]*zip.File, name string) ([]byte, error) { f, ok := files[path.Clean(name)] if !ok { return nil, fmt.Errorf("missing zip entry %q", name) } rc, err := f.Open() if err != nil { return nil, fmt.Errorf("open zip entry %q: %w", name, err) } defer rc.Close() return io.ReadAll(rc) } // --- xhtml text + ruby extraction ---------------------------------------------- // blockTags emit a paragraph break (blank line) so the chunker's paragraph packing // survives extraction; without them a whole chapter collapses into one paragraph. var blockTags = map[string]bool{ "p": true, "div": true, "h1": true, "h2": true, "h3": true, "h4": true, "h5": true, "h6": true, "li": true, "blockquote": true, "section": true, "article": true, "tr": true, "table": true, "ul": true, "ol": true, "dl": true, "dd": true, "dt": true, "pre": true, "figure": true, "figcaption": true, "header": true, "footer": true, "aside": true, "nav": true, "hr": true, "main": true, "td": true, "th": true, "caption": true, } // skipRoots are subtrees whose text is not prose (CSS/JS/metadata). var skipRoots = map[string]bool{"script": true, "style": true, "head": true} // extractXHTML strips tags to text and captures ruby. Rules: // - block tag → blank line;
→ newline (paragraph structure for the chunker); // - basereading: BASE goes to the body, READING is captured // as a (base,reading) pair, parenthesis fallbacks are dropped; // -