package chunk import ( "archive/zip" "io" "os" "path/filepath" "strings" "testing" "textmachine/backend/internal/chunk/chunktest" "textmachine/backend/internal/text" ) // --- txt ------------------------------------------------------------------------ func TestIngestTXTSingleChapter(t *testing.T) { path := filepath.Join(t.TempDir(), "src.txt") if err := os.WriteFile(path, []byte("静かな図書館の朝。"), 0o644); err != nil { t.Fatal(err) } doc, err := ingest(path) if err != nil { t.Fatal(err) } if len(doc.Chapters) != 1 || doc.Chapters[0] != "静かな図書館の朝。" { t.Fatalf("chapters = %#v", doc.Chapters) } if len(doc.Ruby) != 0 { t.Fatalf("txt has no ruby, got %#v", doc.Ruby) } } func TestIngestTXTFormFeedChapters(t *testing.T) { path := filepath.Join(t.TempDir(), "src.txt") if err := os.WriteFile(path, []byte("Глава A"+chapterSep+"Глава B"+chapterSep+"Глава C"), 0o644); err != nil { t.Fatal(err) } doc, err := ingest(path) if err != nil { t.Fatal(err) } want := []string{"Глава A", "Глава B", "Глава C"} if strings.Join(doc.Chapters, "|") != strings.Join(want, "|") { t.Fatalf("chapters = %#v, want %#v", doc.Chapters, want) } } func TestIngestTXTNormalizes(t *testing.T) { path := filepath.Join(t.TempDir(), "src.txt") // UTF-8 BOM + CRLF + surrounding whitespace must be normalized away. if err := os.WriteFile(path, []byte("\uFEFF строка один\r\nстрока два "), 0o644); err != nil { t.Fatal(err) } doc, err := ingest(path) if err != nil { t.Fatal(err) } if doc.Chapters[0] != "строка один\nстрока два" { t.Fatalf("normalized = %q", doc.Chapters[0]) } } func TestIngestEPUBSpineOrder(t *testing.T) { // Manifest order (c3,c1,c2) differs from spine order (c1,c2,c3): the spine wins. chapters := []chunktest.Chapter{ {ID: "c3", Href: "ch3.xhtml", Body: `
第三章。
`}, {ID: "c1", Href: "ch1.xhtml", Body: `第一章。
`}, {ID: "c2", Href: "ch2.xhtml", Body: `第二章。
`}, } doc, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"c1", "c2", "c3"})) if err != nil { t.Fatal(err) } if len(doc.Chapters) != 3 { t.Fatalf("want 3 chapters, got %d: %#v", len(doc.Chapters), doc.Chapters) } for i, want := range []string{"第一章。", "第二章。", "第三章。"} { if strings.TrimSpace(doc.Chapters[i]) != want { t.Fatalf("chapter %d = %q, want %q (spine order broken)", i+1, doc.Chapters[i], want) } } } func TestIngestEPUBStripsTagsAndEntities(t *testing.T) { body := `Hello & bold world.
Second paragraph.
` doc, err := ingest(chunktest.BuildEPUB(t, []chunktest.Chapter{{ID: "c1", Href: "ch1.xhtml", Body: body}}, []string{"c1"})) if err != nil { t.Fatal(err) } txt := doc.Chapters[0] if !strings.Contains(txt, "Hello & bold world.") { t.Fatalf("entities/tags not handled: %q", txt) } if strings.Contains(txt, "") || strings.Contains(txt, "color:red") || strings.Contains(txt, "blocks must be separated by a blank line so the chunker sees paragraphs. if paras := splitParagraphs(text.NormalizeSource(txt)); len(paras) != 2 { t.Fatalf("want 2 paragraphs from 2
, got %d: %#v", len(paras), paras) } } func TestIngestEPUBRubyCaptureAndBaseInBody(t *testing.T) { body := `
漢字を読む。
` + `東京タワー。
` + `字だけ。
` // empty reading → not captured, base kept doc, err := ingest(chunktest.BuildEPUB(t, []chunktest.Chapter{{ID: "c1", Href: "ch1.xhtml", Body: body}}, []string{"c1"})) if err != nil { t.Fatal(err) } txt := doc.Chapters[0] // Base stays in the body; readings and rp parens do NOT. for _, want := range []string{"漢字を読む", "東京タワー", "字だけ"} { if !strings.Contains(txt, want) { t.Fatalf("base missing from body %q (want %q)", txt, want) } } for _, bad := range []string{"かんじ", "とう", "きょう", "(", ")"} { if strings.Contains(txt, bad) { t.Fatalf("reading/paren %q leaked into body %q", bad, txt) } } // Two readings captured (mono-ruby merged to one base+reading); empty rt skipped. got := map[string]string{} for _, r := range doc.Ruby { got[r.Base] = r.Reading if r.Chapter != 1 { t.Fatalf("ruby chapter = %d, want 1", r.Chapter) } } if len(doc.Ruby) != 2 || got["漢字"] != "かんじ" || got["東京"] != "とうきょう" { t.Fatalf("ruby capture = %#v", doc.Ruby) } } func TestIngestEPUBRubyChapterMatchesDenseNumbering(t *testing.T) { // An empty cover page in the spine before a ruby-bearing chapter must NOT shift // ruby's first_chapter off the number SplitChunks assigns (dense — cover skipped). chapters := []chunktest.Chapter{ {ID: "cover", Href: "cover.xhtml", Body: `ふつうの文。
`}, {ID: "c2", Href: "ch2.xhtml", Body: `朱雀が舞う。
`}, } doc, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"cover", "c1", "c2"})) if err != nil { t.Fatal(err) } // The ruby is in the 3rd spine item but the 2nd NON-empty chapter → chapter 2, // exactly what SplitChunks(doc.Chapters) labels it. if len(doc.Ruby) != 1 || doc.Ruby[0].Chapter != 2 { t.Fatalf("ruby dense chapter = %#v, want chapter 2", doc.Ruby) } chunks := SplitChunks(doc.Chapters, testSeg(), nil, testAbbrevs()) var rubyChunkChapter int for _, c := range chunks { if strings.Contains(c.Text, "朱雀") { rubyChunkChapter = c.Chapter } } if rubyChunkChapter != doc.Ruby[0].Chapter { t.Fatalf("ruby first_chapter %d != chunk chapter %d — memory-v2 since_ch would be off", doc.Ruby[0].Chapter, rubyChunkChapter) } } func TestIngestEPUBAcceptsGenericAndParameterizedMediaTypes(t *testing.T) { // Real epubs mislabel xhtml chapters as application/xml or text/xml (incl. a .xml // href — external-review #2), or add a "; charset=utf-8" parameter. All must be // read, not silently dropped (a dropped chapter also shifts every since_ch after). chapters := []chunktest.Chapter{ {ID: "c1", Href: "ch1.xml", MType: "application/xml", Body: `第一章。
`}, {ID: "c2", Href: "ch2.xhtml", MType: "text/xml", Body: `第二章。
`}, {ID: "c3", Href: "ch3.xhtml", MType: "application/xhtml+xml; charset=utf-8", Body: `第三章。
`}, } doc, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"c1", "c2", "c3"})) if err != nil { t.Fatal(err) } if len(doc.Chapters) != 3 { t.Fatalf("mislabeled/parameterized media-types must all be read, got %d: %#v", len(doc.Chapters), doc.Chapters) } for i, want := range []string{"第一章。", "第二章。", "第三章。"} { if strings.TrimSpace(doc.Chapters[i]) != want { t.Fatalf("chapter %d = %q, want %q", i+1, doc.Chapters[i], want) } } } // The four ways real dirty epubs used to kill an import outright: encoding/xml aborted the whole // chapter on each, so ONE malformed document lost the book. The tokenizer reads prose out of all // four. Each case asserts the prose survives AND that the markup noise does not enter it. func TestIngestEPUBDirtyXHTMLImports(t *testing.T) { cases := []struct { name, body string wantIn []string wantNotIn []string }{{ name: "bare < in prose", body: `Если a < b, то дальше.
Второй абзац.
`, wantIn: []string{"Если a < b, то дальше.", "Второй абзац."}, wantNotIn: []string{""}, }, { name: "overlapping tags", body: `
жирный оба курсив хвост.
`, wantIn: []string{"жирный", "оба", "курсив", "хвост."}, wantNotIn: []string{"", ""}, }, { name: "script content looks like markup", body: `Настоящий текст.
`, wantIn: []string{"Настоящий текст."}, wantNotIn: []string{"document.write", "aНастоящий текст.
`, wantIn: []string{"Настоящий текст."}, wantNotIn: []string{"черновая заметка", "v2 (26.07)"}, }} for _, c := range cases { t.Run(c.name, func(t *testing.T) { doc, err := ingest(chunktest.BuildEPUB(t, []chunktest.Chapter{{ID: "c1", Href: "ch1.xhtml", Body: c.body}}, []string{"c1"})) if err != nil { t.Fatalf("a dirty xhtml chapter must still import: %v", err) } if len(doc.Chapters) != 1 { t.Fatalf("want 1 chapter, got %#v", doc.Chapters) } got := doc.Chapters[0] for _, w := range c.wantIn { if !strings.Contains(got, w) { t.Errorf("prose %q lost from extraction: %q", w, got) } } for _, w := range c.wantNotIn { if strings.Contains(got, w) { t.Errorf("noise %q leaked into extraction: %q", w, got) } } }) } } // The subtree is skipped by NAME, not by raw nesting depth: its void children (, // ) emit no end tag, so a depth counter would never unwind and the whole chapter would // vanish. Both spellings — self-closed and bare — must leave the body intact. func TestIngestEPUBVoidTagsInHeadDoNotSwallowChapter(t *testing.T) { for _, head := range []string{ `Тело главы.
` body, _, err := extractXHTML([]byte(raw)) if err != nil { t.Fatalf("head %q: %v", head, err) } if !strings.Contains(body, "Тело главы.") { t.Fatalf("head %q swallowed the body: %q", head, body) } if strings.Contains(body, "T") && strings.Contains(body, "Тело главы.
` body, _, err := extractXHTML([]byte(raw)) if err != nil { t.Fatal(err) } if !strings.Contains(body, "Тело главы.") { t.Fatalf("an unclosed swallowed the chapter: %q", body) } } // Byte-parity with the previous encoding/xml reader on the constructs where the tokenizer differs // most: it emits ONE token for a void or self-closed element where the xml decoder synthesised a // start AND an end (HTMLAutoClose). The expected strings below were measured against that reader, // so a regression here is a silent re-chunk of every book carryingA
B
`, "\n\nA\n\n\n\n\n\n\n\nB\n\n"}, {"hr bare", `A
B
`, "\n\nA\n\n\n\n\n\n\n\nB\n\n"}, //A
B
A
B
`, "\n\nA\n\n\n\nB\n\n"}, {"void inside script", `A
B
`, "\n\nA\n\n\n\nB\n\n"}, {"void inside head", `A
`, "\n\nA\n\n"}, // No wrapper: nothing rescues a skip that failed to unwind, so this is what pins // that the subtree is tracked by NAME. Its void children emit no end tag, and a // blind depth counter would still be inside here and drop the prose entirely. {"head without body, bare void", `Проза.
`, "\n\nПроза.\n\n"}, {"head without body, self-closed void", `Проза.
`, "\n\nПроза.\n\n"}, } { t.Run(c.name, func(t *testing.T) { got, _, err := extractXHTML([]byte(c.doc)) if err != nil { t.Fatal(err) } if got != c.want { t.Fatalf("extraction drifted from the previous reader\n got %q\n want %q", got, c.want) } }) } } // UNBALANCED ruby. HTML5 makes , and optional, and the tokenizer — unlike the xml // decoder — does not synthesise the implied end tags. Tracking the sub-element as a nesting depth // leaks on the first omission and routes every LATER base into the reading buffer, deleting it from // the prose; an unclosed likewise suppresses every later paragraph break. Both are content // loss on the wire, and both are invisible to a clean-corpus comparison. Expected values measured // against the previous reader. func TestExtractXHTMLUnbalancedRuby(t *testing.T) { t.Run("omitted does not eat the next ruby", func(t *testing.T) { body, ruby, err := extractXHTML([]byte( `漢текст
字ещё
`)) if err != nil { t.Fatal(err) } if body != "\n\n漢текст\n\n\n\n字ещё\n\n" { t.Fatalf("prose lost after an omitted : %q", body) } if len(ruby) != 2 || ruby[0].Base != "漢" || ruby[1].Base != "字" || ruby[1].Reading != "じ" { t.Fatalf("ruby capture broken after an omitted : %#v", ruby) } }) t.Run("unclosed does not swallow the chapter", func(t *testing.T) { body, ruby, err := extractXHTML([]byte( `漢
Второй абзац.
Третий.
`)) if err != nil { t.Fatal(err) } if body != "\n\n漢\n\n\n\nВторой абзац.\n\n\n\nТретий.\n\n" { t.Fatalf("an unclosed suppressed the paragraph breaks: %q", body) } if paras := splitParagraphs(text.NormalizeSource(body)); len(paras) != 3 { t.Fatalf("want 3 paragraphs, got %d: %#v", len(paras), paras) } if len(ruby) != 1 || ruby[0].Base != "漢" || ruby[0].Reading != "かん" { t.Fatalf("ruby lost when was closed by its block: %#v", ruby) } }) t.Run("nested ruby with an unclosed inner ", func(t *testing.T) { // The inner must clear the sub-mode even though it does not close the OUTER ruby, // or the outer base 字 is routed into the reading and disappears from the prose. body, ruby, err := extractXHTML([]byte( `漢字текст
`)) if err != nil { t.Fatal(err) } if body != "\n\n漢字текст\n\n" { t.Fatalf("nested ruby lost the outer base: %q", body) } if len(ruby) != 1 || ruby[0].Base != "漢字" || ruby[0].Reading != "かんじ" { t.Fatalf("nested ruby capture = %#v", ruby) } }) t.Run("omitted keeps the parenthesis fallback out of the prose", func(t *testing.T) { body, ruby, err := extractXHTML([]byte( `東текст
`)) if err != nil { t.Fatal(err) } if body != "\n\n東текст\n\n" { t.Fatalf("rp fallback leaked or base lost: %q", body) } if len(ruby) != 1 || ruby[0].Base != "東" || ruby[0].Reading != "とう" { t.Fatalf("ruby capture = %#v", ruby) } }) } // Raw-text elements are the tokenizer's sharpest edge. Its raw mode is what makes markup-shaped //Проза.
`)) if err != nil { t.Fatal(err) } if strings.TrimSpace(got) != "Проза." { t.Fatalf("script raw-text mode broken: %q", got) } } // The tokenizer resolves HTML5 legacy entities the xml decoder left literal — a bare «&» followed // by a known name (&, ©,   …) with no semicolon. That is the correct HTML reading, but it // means a bare ampersand in prose is now interpreted, so the ordinary prose case is pinned here: // «AT&T», «R&D» and «Р&Б» must survive untouched, because the letter after & starts no entity name. func TestExtractXHTMLBareAmpersandInProseSurvives(t *testing.T) { for _, s := range []string{"AT&T", "R&D", "Тим & Ко", "1 & 2"} { body, _, err := extractXHTML([]byte(`` + s + `
`)) if err != nil { t.Fatal(err) } if !strings.Contains(body, s) { t.Errorf("bare ampersand prose %q was altered: %q", s, body) } } } // A namespace-prefixed block tag must still break paragraphs: the xml decoder matched on the LOCAL // name (До.
& амперсанд ]]>После.
`, "сырой <текст> & амперсанд"}, {"markup kept literal", `не разметка]]>Проза.
`, "не разметка
"}, } { t.Run(c.name, func(t *testing.T) { body, _, err := extractXHTML([]byte(c.doc)) if err != nil { t.Fatal(err) } if !strings.Contains(body, c.want) { t.Fatalf("CDATA content lost or mangled: got %q, want it to contain %q", body, c.want) } if strings.Contains(body, "]]>") || strings.Contains(body, "CDATA") { t.Fatalf("CDATA delimiters leaked into prose: %q", body) } }) } // The real-world spelling: CDATA guards inside aПроза.
")) if err != nil { t.Fatal(err) } if strings.TrimSpace(body) != "Проза." { t.Fatalf("style CDATA leaked: %q", body) } } // A declared non-UTF-8 charset stays a LOUD error (epub v1 = UTF-8) — the guard the xml decoder's // CharsetReader used to provide. Silent mojibake in ch.Text is the failure this prevents. func TestExtractXHTMLRejectsNonUTF8Charset(t *testing.T) { bad := []byte(`текст
`) if _, _, err := extractXHTML(bad); err == nil { t.Fatal("a declared non-UTF-8 xhtml charset must fail loud") } else if !strings.Contains(err.Error(), "gb18030") { t.Fatalf("the error must name the charset, got: %v", err) } for _, ok := range []string{``, ``, ``} { if _, _, err := extractXHTML([]byte(ok + `текст
`)); err != nil { t.Fatalf("declaration %q must be accepted: %v", ok, err) } } } func TestIngestEPUBPercentEncodedHref(t *testing.T) { // Real epubs percent-encode non-ASCII (and spaced) filenames in the manifest while the zip // entry carries the decoded name. Without resolveHref's url.PathUnescape the entry is missed // and ingest fails loud on a chapter that is actually present. chapters := []chunktest.Chapter{ {ID: "c1", Href: "%E7%AC%AC%E4%B8%80%E7%AB%A0.xhtml", EntryName: "第一章.xhtml", Body: `第一章。
`}, {ID: "c2", Href: "ch%202.xhtml", EntryName: "ch 2.xhtml", Body: `第二章。
`}, } doc, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"c1", "c2"})) if err != nil { t.Fatalf("percent-encoded hrefs must resolve to their decoded zip entries: %v", err) } if len(doc.Chapters) != 2 { t.Fatalf("want 2 chapters, got %d: %#v", len(doc.Chapters), doc.Chapters) } for i, want := range []string{"第一章。", "第二章。"} { if strings.TrimSpace(doc.Chapters[i]) != want { t.Fatalf("chapter %d = %q, want %q", i+1, doc.Chapters[i], want) } } } func TestIngestEPUBHrefFragmentIgnored(t *testing.T) { // A spine href may point at an anchor inside a document (chapter.xhtml#part2). The fragment // addresses a position, not a file: it must be dropped before the zip lookup, or the chapter // is reported missing. chapters := []chunktest.Chapter{ {ID: "c1", Href: "ch1.xhtml#part2", EntryName: "ch1.xhtml", Body: `第一章。
`}, } doc, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"c1"})) if err != nil { t.Fatalf("an href #fragment must be dropped before the zip lookup: %v", err) } if len(doc.Chapters) != 1 || strings.TrimSpace(doc.Chapters[0]) != "第一章。" { t.Fatalf("chapters = %#v", doc.Chapters) } } func TestEPUBFixtureMimetypeIsOCFConformant(t *testing.T) { // OCF: "mimetype" must be the FIRST entry and STORED (uncompressed). All three stand epubs // ship it that way; the fixture must too, or it is not the file shape the reader will meet. p := chunktest.BuildEPUB(t, []chunktest.Chapter{{ID: "c1", Href: "ch1.xhtml", Body: `x
`}}, []string{"c1"}) zr, err := zip.OpenReader(p) if err != nil { t.Fatal(err) } defer zr.Close() if len(zr.File) == 0 || zr.File[0].Name != "mimetype" { t.Fatalf("mimetype must be the first zip entry, got %q", zr.File[0].Name) } if zr.File[0].Method != zip.Store { t.Fatalf("mimetype must be STORED (method %d), got method %d", zip.Store, zr.File[0].Method) } rc, err := zr.File[0].Open() if err != nil { t.Fatal(err) } defer rc.Close() b, _ := io.ReadAll(rc) if string(b) != "application/epub+zip" { t.Fatalf("mimetype content = %q", b) } } func TestIngestEPUBDanglingIdrefFailsLoud(t *testing.T) { // A spine idref with no manifest item, AMONG valid chapters, must fail loud — not // silently drop that chapter (which would also shift every later since_ch). chapters := []chunktest.Chapter{ {ID: "c1", Href: "ch1.xhtml", Body: `第一章。
`}, {ID: "c2", Href: "ch2.xhtml", Body: `第二章。
`}, {ID: "c3", Href: "ch3.xhtml", Body: `第三章。
`}, } _, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"c1", "ghost", "c3"})) if err == nil { t.Fatal("a dangling spine idref among valid chapters must fail loud, not silently lose a chapter") } if !strings.Contains(err.Error(), "ghost") { t.Fatalf("error should name the dangling idref, got: %v", err) } } func TestIngestEPUBBrInsideRubyDoesNotLeak(t *testing.T) { // A漢
字を読む。
\n
本文。
`}, } doc, err := ingest(chunktest.BuildEPUB(t, chapters, []string{"img", "c1"})) if err != nil { t.Fatal(err) } if len(doc.Chapters) != 1 || strings.TrimSpace(doc.Chapters[0]) != "本文。" { t.Fatalf("non-xhtml spine item must be skipped: %#v", doc.Chapters) } } func TestIngestEPUBBrokenFailsLoud(t *testing.T) { t.Run("not a zip", func(t *testing.T) { p := filepath.Join(t.TempDir(), "bad.epub") os.WriteFile(p, []byte("this is not a zip"), 0o644) if _, err := ingest(p); err == nil { t.Fatal("a non-zip .epub must error, not panic") } }) t.Run("missing container", func(t *testing.T) { p := filepath.Join(t.TempDir(), "noc.epub") f, _ := os.Create(p) zw := zip.NewWriter(f) w, _ := zw.Create("OEBPS/content.opf") w.Write([]byte("x
`}}, nil) if _, err := ingest(p); err == nil { t.Fatal("an empty spine must error") } }) t.Run("dangling spine idref", func(t *testing.T) { // spine references a missing manifest id → resolves to zero chapters → error. p := chunktest.BuildEPUB(t, []chunktest.Chapter{{ID: "c1", Href: "ch1.xhtml", Body: `x
`}}, []string{"ghost"}) if _, err := ingest(p); err == nil { t.Fatal("a spine resolving to no readable chapters must error") } }) }