707 lines
27 KiB
Go
707 lines
27 KiB
Go
package chunk
|
|
|
|
import (
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
|
|
"textmachine/backend/internal/chunk/chunktest"
|
|
"textmachine/backend/internal/lang"
|
|
"textmachine/backend/internal/text"
|
|
)
|
|
|
|
// ch is a content document with enough prose to be a chapter.
|
|
func ch(id, href, body string) chunktest.Chapter {
|
|
return chunktest.Chapter{ID: id, Href: href, Body: "<p>" + body + "</p>"}
|
|
}
|
|
|
|
func refs(ids ...string) []chunktest.SpineRef {
|
|
out := make([]chunktest.SpineRef, 0, len(ids))
|
|
for _, id := range ids {
|
|
out = append(out, chunktest.SpineRef{ID: id})
|
|
}
|
|
return out
|
|
}
|
|
|
|
// TestEPUBNavDrawsChaptersNotTheSpine is the pack's central claim: the spine is reading ORDER and the nav is
|
|
// the book's own statement about chapters. Six documents, three nav targets ⇒ three chapters, and the
|
|
// documents between targets are FOLDED IN rather than becoming chapters of their own.
|
|
func TestEPUBNavDrawsChaptersNotTheSpine(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", "one a"), ch("c1b", "c1b.xhtml", "one b"),
|
|
ch("c2", "c2.xhtml", "two a"), ch("c2b", "c2b.xhtml", "two b"),
|
|
ch("c3", "c3.xhtml", "three a"), ch("c3b", "c3b.xhtml", "three b"),
|
|
},
|
|
Spine: refs("c1", "c1b", "c2", "c2b", "c3", "c3b"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "c2.xhtml", "c3.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 3 {
|
|
t.Fatalf("chapters = %d, want 3 (the nav named three)", len(doc.Chapters))
|
|
}
|
|
if doc.Structure != StructureDeclared {
|
|
t.Fatalf("structure = %q, want %q — a nav IS the format declaring its chapters", doc.Structure, StructureDeclared)
|
|
}
|
|
// ⛔ No text was dropped: every document's prose survives, in its own chapter.
|
|
for i, want := range []string{"one a", "one b"} {
|
|
if !strings.Contains(doc.Chapters[0], want) {
|
|
t.Fatalf("chapter 1 lost %q (part %d): %q", want, i, doc.Chapters[0])
|
|
}
|
|
}
|
|
if doc.DocumentsAttached != 3 {
|
|
t.Fatalf("documents attached = %d, want 3 (one folded into each chapter)", doc.DocumentsAttached)
|
|
}
|
|
}
|
|
|
|
// TestEPUB2NCXDrawsChaptersWhenThereIsNoNav is the EPUB 2 half. An EPUB 2 has no navigation document at all,
|
|
// so a reader that knows only about nav falls straight through to the spine and reports the wrong chapters.
|
|
func TestEPUB2NCXDrawsChaptersWhenThereIsNoNav(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", "one a"), ch("c1b", "c1b.xhtml", "one b"),
|
|
ch("c2", "c2.xhtml", "two a"),
|
|
},
|
|
Spine: refs("c1", "c1b", "c2"),
|
|
NCX: []string{"c1.xhtml", "c2.xhtml"},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 || doc.Structure != StructureDeclared {
|
|
t.Fatalf("chapters = %d structure = %q, want 2 / %q", len(doc.Chapters), doc.Structure, StructureDeclared)
|
|
}
|
|
if !strings.Contains(doc.Chapters[0], "one b") {
|
|
t.Fatalf("the document between NCX targets was lost: %q", doc.Chapters[0])
|
|
}
|
|
}
|
|
|
|
// TestEPUBSpineIsDelimitedNotDeclared pins the vocabulary fix: with no nav and no NCX the engine still cuts
|
|
// on the spine, but it no longer calls that the format declaring its chapters.
|
|
func TestEPUBSpineIsDelimitedNotDeclared(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")},
|
|
Spine: refs("c1", "c2"),
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if doc.Structure != StructureDelimited {
|
|
t.Fatalf("structure = %q, want %q — a spine is reading order, not a chapter list", doc.Structure, StructureDelimited)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2", len(doc.Chapters))
|
|
}
|
|
}
|
|
|
|
// TestEPUB3LandmarksExcludeTocAndCover: EPUB 3 declares service pages inside the navigation document.
|
|
func TestEPUB3LandmarksExcludeTocAndCover(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("cov", "cover.xhtml", "COVERPAGE"), ch("toc", "toc.xhtml", "TOCPAGE"),
|
|
ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: refs("cov", "toc", "c1", "c2"),
|
|
Nav: &chunktest.Nav{
|
|
TOC: []string{"c1.xhtml", "c2.xhtml"},
|
|
Landmarks: map[string]string{"cover.xhtml": "cover", "toc.xhtml": "toc"},
|
|
InSpine: true,
|
|
},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2", len(doc.Chapters))
|
|
}
|
|
for _, bad := range []string{"COVERPAGE", "TOCPAGE"} {
|
|
for i, c := range doc.Chapters {
|
|
if strings.Contains(c, bad) {
|
|
t.Fatalf("service page %q leaked into chapter %d", bad, i+1)
|
|
}
|
|
}
|
|
}
|
|
// The nav document itself is in the spine here and is excluded on its `properties`, so three pages went.
|
|
if len(doc.Excluded) != 3 {
|
|
t.Fatalf("documents excluded = %d, want 3 (cover + toc + the nav document)", len(doc.Excluded))
|
|
}
|
|
}
|
|
|
|
// TestEPUB2GuideExcludesTocAndCover is the case a landmarks-only reader gets wrong. An EPUB 2 has no
|
|
// navigation document, so the ONLY declaration of "this page is the toc" is the OPF <guide>. Miss it and a
|
|
// text table of contents is glued to chapter one — the mass case, not a corner one.
|
|
func TestEPUB2GuideExcludesTocAndCover(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("cov", "cover.xhtml", "COVERPAGE"), ch("toc", "toc.xhtml", "TOCPAGE"),
|
|
ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: refs("cov", "toc", "c1", "c2"),
|
|
NCX: []string{"c1.xhtml", "c2.xhtml"},
|
|
Guide: []chunktest.GuideRef{{Type: "cover", Href: "cover.xhtml"}, {Type: "toc", Href: "toc.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2", len(doc.Chapters))
|
|
}
|
|
if strings.Contains(doc.Chapters[0], "TOCPAGE") {
|
|
t.Fatal("the EPUB 2 table-of-contents page was glued to chapter one — <guide> was not read")
|
|
}
|
|
if len(doc.Excluded) != 2 {
|
|
t.Fatalf("documents excluded = %d, want 2", len(doc.Excluded))
|
|
}
|
|
}
|
|
|
|
// TestEPUBCoverImagePropertyExcludesNothing: `properties="cover-image"` sits on an IMAGE, not on a document.
|
|
// A reader that greps the attribute for "cover" would drop a real chapter.
|
|
func TestEPUBCoverImagePropertyExcludesNothing(t *testing.T) {
|
|
chapters := []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")}
|
|
chapters[0].Properties = "cover-image"
|
|
p := chunktest.EPUB{Chapters: chapters, Spine: refs("c1", "c2")}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 || len(doc.Excluded) != 0 {
|
|
t.Fatalf("chapters = %d excluded = %d, want 2 / 0 — cover-image is not the cover DOCUMENT",
|
|
len(doc.Chapters), len(doc.Excluded))
|
|
}
|
|
}
|
|
|
|
// TestEPUBNonLinearNeverStartsAChapter: linear="no" marks appendices and footnote pages. They are text and
|
|
// are kept, but they do not open a chapter even when the nav points at them.
|
|
func TestEPUBNonLinearNeverStartsAChapter(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", "one"), ch("notes", "notes.xhtml", "FOOTNOTES"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: []chunktest.SpineRef{{ID: "c1"}, {ID: "notes", Linear: "no"}, {ID: "c2"}},
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "notes.xhtml", "c2.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2 — the non-linear page must not open one", len(doc.Chapters))
|
|
}
|
|
if !strings.Contains(doc.Chapters[0], "FOOTNOTES") {
|
|
t.Fatalf("the non-linear page was DROPPED instead of attached: %q", doc.Chapters[0])
|
|
}
|
|
}
|
|
|
|
// TestEPUBUncoveredBeforeFirstTargetJoinsChapterOne mirrors the txt preamble rule: a title page the nav does
|
|
// not mention joins chapter one rather than becoming a chapter or vanishing.
|
|
func TestEPUBUncoveredBeforeFirstTargetJoinsChapterOne(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("title", "title.xhtml", "TITLEPAGE"), ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: refs("title", "c1", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "c2.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2", len(doc.Chapters))
|
|
}
|
|
if !strings.Contains(doc.Chapters[0], "TITLEPAGE") {
|
|
t.Fatalf("the uncovered title page was lost: %q", doc.Chapters[0])
|
|
}
|
|
}
|
|
|
|
// TestEPUBNavWithNoResolvableTargetsIsAbsent: a nav pointing at documents that are not in the spine is a
|
|
// BROKEN nav. It must fall through to the next witness, not produce zero chapters.
|
|
func TestEPUBNavWithNoResolvableTargetsIsAbsent(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")},
|
|
Spine: refs("c1", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"gone-a.xhtml", "gone-b.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 || doc.Structure != StructureDelimited {
|
|
t.Fatalf("chapters = %d structure = %q, want 2 / %q", len(doc.Chapters), doc.Structure, StructureDelimited)
|
|
}
|
|
if doc.TOCUnresolved != 2 {
|
|
t.Fatalf("unresolved targets = %d, want 2 — the count is the whole alarm", doc.TOCUnresolved)
|
|
}
|
|
}
|
|
|
|
// TestEPUBDuplicateNavTargetsCollapse: several nav entries pointing into ONE document (chapter + its
|
|
// sections) are one boundary, not several empty ones.
|
|
func TestEPUBDuplicateNavTargetsCollapse(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")},
|
|
Spine: refs("c1", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "c1.xhtml#s2", "c2.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2 — a fragment into the same document is the same boundary", len(doc.Chapters))
|
|
}
|
|
if doc.TargetsCollapsed != 1 {
|
|
t.Fatalf("targets collapsed = %d, want 1", doc.TargetsCollapsed)
|
|
}
|
|
// ⛔ AND THE PROVENANCE IS DOWNGRADED. The nav named three chapters; this package hands back two, because
|
|
// it groups whole documents and does not slice one by its anchors. Reporting that as `declared` would put
|
|
// the engine's own coarser answer out under the format's name — the very lie the spine used to tell.
|
|
if doc.Structure != StructureDelimited {
|
|
t.Fatalf("structure = %q, want %q — the format named more chapters than the cut returned",
|
|
doc.Structure, StructureDelimited)
|
|
}
|
|
}
|
|
|
|
// TestEPUBNavOracleReproducesTheTxtCut is the code oracle the format work is measured by: ground truth with
|
|
// no hand annotation and no model call. The nav DECLARES where chapters begin; unroll the same book into a
|
|
// plain txt whose chapters open with source headers, and the txt path must recover the same boundaries.
|
|
// Two independent readers, one answer.
|
|
func TestEPUBNavOracleReproducesTheTxtCut(t *testing.T) {
|
|
bodies := []string{"第一章\n" + strings.Repeat("蛊", 200), "第二章\n" + strings.Repeat("蛊", 200), "第三章\n" + strings.Repeat("蛊", 200)}
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", bodies[0]), ch("c2", "c2.xhtml", bodies[1]), ch("c3", "c3.xhtml", bodies[2]),
|
|
},
|
|
Spine: refs("c1", "c2", "c3"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "c2.xhtml", "c3.xhtml"}},
|
|
}.Build(t)
|
|
|
|
fromEPUB, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
txt := writeTempTXT(t, strings.Join(bodies, "\n"))
|
|
fromTXT, err := ingest(txt)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(fromTXT.Chapters) != len(fromEPUB.Chapters) {
|
|
t.Fatalf("txt cut %d chapters, the nav declared %d", len(fromTXT.Chapters), len(fromEPUB.Chapters))
|
|
}
|
|
for i := range fromEPUB.Chapters {
|
|
if fromTXT.Chapters[i] != fromEPUB.Chapters[i] {
|
|
t.Fatalf("chapter %d differs:\n txt: %.60q\nepub: %.60q", i+1, fromTXT.Chapters[i], fromEPUB.Chapters[i])
|
|
}
|
|
}
|
|
if fromEPUB.Structure != StructureDeclared || fromTXT.Structure != StructureDetected {
|
|
t.Fatalf("provenance should differ even where the cut agrees: epub=%q txt=%q", fromEPUB.Structure, fromTXT.Structure)
|
|
}
|
|
}
|
|
|
|
// writeTempTXT writes body to a temp .txt and returns its path.
|
|
func writeTempTXT(t *testing.T, body string) string {
|
|
t.Helper()
|
|
p := filepath.Join(t.TempDir(), "src.txt")
|
|
if err := os.WriteFile(p, []byte(body), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
return p
|
|
}
|
|
|
|
// TestEPUBNavTargetsFollowSpineOrderNotTocOrder: a table of contents may list its entries in any order (an
|
|
// appendix first, a foreword last). Boundaries are spine POSITIONS, so a mis-ordered toc must not interleave
|
|
// the chapters or hand a later document to an earlier chapter.
|
|
func TestEPUBNavTargetsFollowSpineOrderNotTocOrder(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", "ONE"), ch("c1b", "c1b.xhtml", "ONEB"),
|
|
ch("c2", "c2.xhtml", "TWO"), ch("c2b", "c2b.xhtml", "TWOB"),
|
|
},
|
|
Spine: refs("c1", "c1b", "c2", "c2b"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c2.xhtml", "c1.xhtml"}}, // declared out of reading order
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2", len(doc.Chapters))
|
|
}
|
|
if !strings.Contains(doc.Chapters[0], "ONE") || !strings.Contains(doc.Chapters[0], "ONEB") {
|
|
t.Fatalf("chapter 1 is not the FIRST span of the book: %q", doc.Chapters[0])
|
|
}
|
|
if !strings.Contains(doc.Chapters[1], "TWO") || !strings.Contains(doc.Chapters[1], "TWOB") {
|
|
t.Fatalf("chapter 2 is not the SECOND span of the book: %q", doc.Chapters[1])
|
|
}
|
|
}
|
|
|
|
// TestEPUBPartiallyUnresolvedNavStillCuts: a scrape whose toc points half at real documents and half at
|
|
// nothing must use what resolved and COUNT the rest, not fall back and not fail.
|
|
func TestEPUBPartiallyUnresolvedNavStillCuts(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")},
|
|
Spine: refs("c1", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "gone.xhtml", "c2.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(doc.Chapters) != 2 || doc.Structure != StructureDeclared {
|
|
t.Fatalf("chapters = %d structure = %q, want 2 / %q", len(doc.Chapters), doc.Structure, StructureDeclared)
|
|
}
|
|
if doc.TOCUnresolved != 1 {
|
|
t.Fatalf("unresolved = %d, want 1", doc.TOCUnresolved)
|
|
}
|
|
}
|
|
|
|
// TestEPUBAllServiceSpineFailsLoud: a book whose every spine document is a declared service page has no text
|
|
// to translate. That must be a loud refusal, never an empty book that costs money to discover.
|
|
func TestEPUBAllServiceSpineFailsLoud(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("cov", "cover.xhtml", "COVER"), ch("toc", "toc.xhtml", "TOC")},
|
|
Spine: refs("cov", "toc"),
|
|
NCX: []string{"cover.xhtml"},
|
|
Guide: []chunktest.GuideRef{{Type: "cover", Href: "cover.xhtml"}, {Type: "toc", Href: "toc.xhtml"}},
|
|
}.Build(t)
|
|
|
|
if _, err := ingest(p); err == nil {
|
|
t.Fatal("a spine of nothing but service pages must fail loud, not return an empty book")
|
|
}
|
|
}
|
|
|
|
// TestEPUBSingleDocumentChapterIsByteIdenticalToTheOldPath guards the refactor itself: grouping now JOINS a
|
|
// chapter's documents, and the one-document case must go through that join unchanged — otherwise every
|
|
// existing EPUB book silently re-cuts and re-pays.
|
|
func TestEPUBSingleDocumentChapterIsByteIdenticalToTheOldPath(t *testing.T) {
|
|
body := "<p>Первый абзац.</p><p>Второй абзац.</p>"
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{{ID: "c1", Href: "c1.xhtml", Body: body}, {ID: "c2", Href: "c2.xhtml", Body: body}},
|
|
Spine: refs("c1", "c2"),
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
// The expectation is COMPUTED from the primitives, not guessed: whatever extractXHTML produces for this
|
|
// document, a one-document chapter must be exactly that, normalized — the join must add nothing.
|
|
raw, _, err := extractXHTML([]byte(`<?xml version="1.0" encoding="utf-8"?>
|
|
<html xmlns="http://www.w3.org/1999/xhtml"><head><title>c</title><style>.x{color:red}</style></head>
|
|
<body>` + body + `</body></html>`))
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
want := text.NormalizeSource(raw)
|
|
for i, c := range doc.Chapters {
|
|
if c != want {
|
|
t.Fatalf("chapter %d = %q, want %q — the join altered a single-document chapter", i+1, c, want)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestANewLanguageCutsFromItsDataFileAlone answers the project's default review question end to end: a pair
|
|
// that is NOT in this repository ships one data file and gets a real chapter cut, with no Go change.
|
|
//
|
|
// Korean proves both halves. With no file it falls to the embedded CJK default, whose marker 第 never appears
|
|
// in a Korean book, so the text stays ONE chapter. With ko/structure.txt naming 제 and 장 — the same binary,
|
|
// the same text — it cuts into three and reports `detected`.
|
|
func TestANewLanguageCutsFromItsDataFileAlone(t *testing.T) {
|
|
prose := strings.Repeat("가", 200)
|
|
body := "제1장 시작\n" + prose + "\n제2장 전개\n" + prose + "\n제3장 결말\n" + prose
|
|
src := writeTempTXT(t, body)
|
|
|
|
bare, err := lang.LoadSourceStructure("", "ko")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
before, err := IngestEncoded(src, "", "ko", bare)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(before.Chapters) != 1 {
|
|
t.Fatalf("without its own grammar a Korean book must stay one chapter, got %d", len(before.Chapters))
|
|
}
|
|
|
|
root := t.TempDir()
|
|
if err := os.MkdirAll(filepath.Join(root, "ko"), 0o755); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if err := os.WriteFile(filepath.Join(root, "ko", "structure.txt"), []byte("marker\t제\nunits\t장\n"), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
st, err := lang.LoadSourceStructure(root, "ko")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
after, err := IngestEncoded(src, "", "ko", st)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(after.Chapters) != 3 {
|
|
t.Fatalf("chapters = %d, want 3 — the language's own grammar did not cut", len(after.Chapters))
|
|
}
|
|
if after.Structure != StructureDetected {
|
|
t.Fatalf("structure = %q, want %q", after.Structure, StructureDetected)
|
|
}
|
|
if after.Titles[1] != "제2장 전개" {
|
|
t.Fatalf("title_raw of chapter 2 = %q, want %q", after.Titles[1], "제2장 전개")
|
|
}
|
|
}
|
|
|
|
// TestEPUBServiceRoleWithAFragmentDoesNotDeleteTheDocument is the text-loss guard.
|
|
//
|
|
// ⛔ `<reference type="toc" href="c2.xhtml#pos"/>` says the table of contents BEGINS at that point — inside a
|
|
// document that may hold prose before and after it. Reading it as "this whole file is the toc" deletes book
|
|
// text while the chapter count stays plausible and the provenance stays a confident `declared`: nothing
|
|
// looks wrong. The form is not hypothetical — the owner's Kristoff EPUB carries exactly it
|
|
// (`type="toc" href="index_split_124.html#filepos2096584"`), and there the split file happened to be pure
|
|
// front matter. One calibre split the other way and it is a chapter.
|
|
func TestEPUBServiceRoleWithAFragmentDoesNotDeleteTheDocument(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", "PROSEONE"), ch("c2", "c2.xhtml", "PROSETWO"), ch("c3", "c3.xhtml", "PROSETHREE"),
|
|
},
|
|
Spine: refs("c1", "c2", "c3"),
|
|
NCX: []string{"c1.xhtml", "c2.xhtml", "c3.xhtml"},
|
|
Guide: []chunktest.GuideRef{{Type: "toc", Href: "c2.xhtml#tocpos"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
joined := strings.Join(doc.Chapters, "\x00")
|
|
for _, want := range []string{"PROSEONE", "PROSETWO", "PROSETHREE"} {
|
|
if !strings.Contains(joined, want) {
|
|
t.Fatalf("%s was deleted: a role declared at a POINT inside a document excluded the whole file", want)
|
|
}
|
|
}
|
|
if len(doc.Excluded) != 0 {
|
|
t.Fatalf("documents excluded = %d, want 0 — the declaration named a place, not a document", len(doc.Excluded))
|
|
}
|
|
}
|
|
|
|
// TestEPUBWholeDocumentServiceRoleStillExcludes is the other half: without a fragment the declaration IS
|
|
// about the document, and the page must still go.
|
|
func TestEPUBWholeDocumentServiceRoleStillExcludes(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("toc", "toc.xhtml", "TOCPAGE"), ch("c1", "c1.xhtml", "PROSE")},
|
|
Spine: refs("toc", "c1"),
|
|
NCX: []string{"c1.xhtml"},
|
|
Guide: []chunktest.GuideRef{{Type: "toc", Href: "toc.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if strings.Contains(strings.Join(doc.Chapters, ""), "TOCPAGE") {
|
|
t.Fatal("a whole-document toc declaration must still exclude the page")
|
|
}
|
|
if len(doc.Excluded) != 1 {
|
|
t.Fatalf("documents excluded = %d, want 1", len(doc.Excluded))
|
|
}
|
|
}
|
|
|
|
// TestEPUBOrdinaryBookDoesNotReportABrokenTOC: a table of contents that lists the book's own cover resolves
|
|
// perfectly — the cover is right there in the spine — it is simply not a chapter boundary. Counting that as
|
|
// "the toc named nothing" makes every ordinary EPUB 2 report a broken table, and an alarm that fires on the
|
|
// normal case is one nobody reads.
|
|
func TestEPUBOrdinaryBookDoesNotReportABrokenTOC(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("cov", "cover.xhtml", "COVER"), ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: refs("cov", "c1", "c2"),
|
|
NCX: []string{"cover.xhtml", "c1.xhtml", "c2.xhtml"},
|
|
Guide: []chunktest.GuideRef{{Type: "cover", Href: "cover.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if doc.TOCUnresolved != 0 {
|
|
t.Fatalf("unresolved = %d, want 0 — every target named a document this book has", doc.TOCUnresolved)
|
|
}
|
|
if len(doc.Excluded) != 1 {
|
|
t.Fatalf("documents excluded = %d, want 1 — the cover is counted ONCE, as an exclusion", len(doc.Excluded))
|
|
}
|
|
if len(doc.Chapters) != 2 {
|
|
t.Fatalf("chapters = %d, want 2", len(doc.Chapters))
|
|
}
|
|
}
|
|
|
|
// TestEPUBDanglingNavCountSurvivesTheFallToNCX: a nav whose targets all dangle is replaced by the NCX, and
|
|
// how much of it dangled is still a fact about the book. Overwriting the resolved table wholesale threw that
|
|
// count away with the table.
|
|
func TestEPUBDanglingNavCountSurvivesTheFallToNCX(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")},
|
|
Spine: refs("c1", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"gone-a.xhtml", "gone-b.xhtml"}},
|
|
NCX: []string{"c1.xhtml", "c2.xhtml"},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if doc.Structure != StructureDeclared || len(doc.Chapters) != 2 {
|
|
t.Fatalf("structure = %q chapters = %d, want %q / 2 — the NCX must take over", doc.Structure, len(doc.Chapters), StructureDeclared)
|
|
}
|
|
if doc.TOCUnresolved != 2 {
|
|
t.Fatalf("unresolved = %d, want 2 — the broken nav's dangling targets are still a fact", doc.TOCUnresolved)
|
|
}
|
|
}
|
|
|
|
// TestEPUBTitleRawComesFromTheNavLabel: the EPUB half of title_raw had no test at all — removing the label
|
|
// collection entirely left every chapter with an empty title and the battery green.
|
|
func TestEPUBTitleRawComesFromTheNavLabel(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("c1", "c1.xhtml", "one"), ch("c1b", "c1b.xhtml", "one more"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: refs("c1", "c1b", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "c2.xhtml"}},
|
|
NCX: []string{"c1.xhtml", "c2.xhtml"}, // present but LOSING: nav takes precedence
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
want := []string{"NAV c1.xhtml", "NAV c2.xhtml"}
|
|
if len(doc.Titles) != len(want) {
|
|
t.Fatalf("titles = %d, want %d", len(doc.Titles), len(want))
|
|
}
|
|
for i, w := range want {
|
|
if doc.Titles[i] != w {
|
|
// A wrong SOURCE is the interesting failure: an "NCX …" here means the losing table supplied it.
|
|
t.Fatalf("title %d = %q, want %q", i+1, doc.Titles[i], w)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestEPUB2TitleRawComesFromTheNCXLabel is the EPUB 2 half, where navLabel lives in the NCX.
|
|
func TestEPUB2TitleRawComesFromTheNCXLabel(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two")},
|
|
Spine: refs("c1", "c2"),
|
|
NCX: []string{"c1.xhtml", "c2.xhtml"},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
want := []string{"NCX c1.xhtml", "NCX c2.xhtml"}
|
|
for i, w := range want {
|
|
if doc.Titles[i] != w {
|
|
t.Fatalf("title %d = %q, want %q", i+1, doc.Titles[i], w)
|
|
}
|
|
}
|
|
}
|
|
|
|
// TestEPUBTitleRawFollowsTheChapterOpener: when documents are folded in AHEAD of a boundary, the chapter's
|
|
// title is the boundary's, not the first document's. The opener is stored rather than derived precisely
|
|
// because "first element of the group" is the wrong answer exactly here.
|
|
func TestEPUBTitleRawFollowsTheChapterOpener(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{
|
|
ch("title", "title.xhtml", "TITLEPAGE"), ch("c1", "c1.xhtml", "one"), ch("c2", "c2.xhtml", "two"),
|
|
},
|
|
Spine: refs("title", "c1", "c2"),
|
|
Nav: &chunktest.Nav{TOC: []string{"c1.xhtml", "c2.xhtml"}},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if doc.Titles[0] != "NAV c1.xhtml" {
|
|
t.Fatalf("chapter 1 title = %q — it must name the BOUNDARY, not the title page folded in ahead of it",
|
|
doc.Titles[0])
|
|
}
|
|
}
|
|
|
|
// TestEPUB3LandmarkWithAFragmentDoesNotDeleteTheDocument is the EPUB 3 half of the same rule the guide half
|
|
// keeps: a landmark naming a POINT inside a document marks a place, not the document.
|
|
func TestEPUB3LandmarkWithAFragmentDoesNotDeleteTheDocument(t *testing.T) {
|
|
p := chunktest.EPUB{
|
|
Chapters: []chunktest.Chapter{ch("c1", "c1.xhtml", "PROSEONE"), ch("c2", "c2.xhtml", "PROSETWO")},
|
|
Spine: refs("c1", "c2"),
|
|
Nav: &chunktest.Nav{
|
|
TOC: []string{"c1.xhtml", "c2.xhtml"},
|
|
Landmarks: map[string]string{"c1.xhtml#tocstart": "toc"},
|
|
},
|
|
}.Build(t)
|
|
|
|
doc, err := ingest(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if !strings.Contains(strings.Join(doc.Chapters, ""), "PROSEONE") {
|
|
t.Fatal("a landmark at a POINT inside a document excluded the whole file and deleted its prose")
|
|
}
|
|
if len(doc.Excluded) != 0 {
|
|
t.Fatalf("documents excluded = %d, want 0", len(doc.Excluded))
|
|
}
|
|
if doc.ServiceRolesInsideDocuments != 1 {
|
|
t.Fatalf("service roles inside documents = %d, want 1 — a refusal must be counted, not silent",
|
|
doc.ServiceRolesInsideDocuments)
|
|
}
|
|
}
|
|
|
|
// TestNavLabelIsTrimmedLikeItsSiblings pins the answer to be the same whichever table the book carries.
|
|
//
|
|
// ⛔ Real books indent their nav. Untrimmed, one title arrives as "\n Chapter One\n " through a
|
|
// nav and as "Chapter One" through an NCX — the same chapter, two answers, decided by which table the
|
|
// publisher happened to ship. It goes on the wire as title_raw, so a reader renders the whitespace.
|
|
// The pack's own fixtures could not catch it: they write labels inline, with no indentation to trim.
|
|
func TestNavLabelIsTrimmedLikeItsSiblings(t *testing.T) {
|
|
indented := []byte(`<html xmlns:epub="http://www.idpf.org/2007/ops"><body>
|
|
<nav epub:type="toc">
|
|
<ol>
|
|
<li>
|
|
<a href="c1.xhtml">
|
|
Chapter One
|
|
</a>
|
|
</li>
|
|
</ol>
|
|
</nav>
|
|
</body></html>`)
|
|
_, navLabels, _, _ := parseNavDoc(indented, "OEBPS")
|
|
ncx := []byte(`<ncx><navMap><navPoint><navLabel><text>
|
|
Chapter One
|
|
</text></navLabel><content src="c1.xhtml"/></navPoint></navMap></ncx>`)
|
|
_, ncxLabels := parseNCX(ncx, "OEBPS")
|
|
|
|
const key, want = "OEBPS/c1.xhtml", "Chapter One"
|
|
if navLabels[key] != want {
|
|
t.Fatalf("nav label = %q, want %q", navLabels[key], want)
|
|
}
|
|
if navLabels[key] != ncxLabels[key] {
|
|
t.Fatalf("one title, two answers: nav %q vs ncx %q", navLabels[key], ncxLabels[key])
|
|
}
|
|
}
|