// Package chunktest builds the on-disk source fixtures the ingest path reads. It lives in its own // package because two packages need the SAME epub shape: the ingest unit tests and the runner's // end-to-end tests. Duplicating a zip/OPF builder in both would let the two drift apart, and the // whole point of the e2e fixture is that it is the file the unit tests already pin. // // The CONTAINER layer (mimetype first and stored, container.xml, the entry writes) is not built here // any more: it is internal/bookfile's, the same code the engine's book writer ships an EPUB through, so // the fixture the reader is pinned against and the file the writer produces cannot drift apart at the // zip level. What stays here is deliberately the TEST half — the bare OPF stub, the decoy /<style> // in every chapter head that ingest_test proves are dropped, the MType/EntryName knobs for malformed // manifests — none of which may ever reach a reader. package chunktest import ( "os" "path/filepath" "strconv" "strings" "testing" "textmachine/backend/internal/bookfile" ) type Chapter struct { ID string Href string // relative to the OPF dir (OEBPS/), as the MANIFEST spells it MType string // "" → application/xhtml+xml Body string // inner xhtml of <body> // EntryName is the zip entry the file is actually written to (relative to the OPF dir), // for fixtures whose manifest href does NOT spell the entry name: a percent-encoded href // or one carrying a #fragment. "" → the href itself, so every existing fixture is unchanged. EntryName string // Properties is the OPF manifest `properties` attribute — "nav" marks the navigation document. Properties string // Raw writes Body verbatim instead of wrapping it in the xhtml skeleton, for the nav document and // anything else that must control its own markup. Raw bool // Declared writes the manifest item but NOT the file, for a book that promises a document it does not // carry — the shape a scraped EPUB arrives in. Declared bool } // SpineRef is one spine entry: the manifest id and its `linear` attribute ("" → linear, i.e. absent). type SpineRef struct { ID string Linear string } // Nav describes an EPUB 3 navigation document: the toc targets in order and the landmark roles. type Nav struct { Href string // "" → "nav.xhtml" TOC []string // hrefs, relative to the OPF dir Landmarks map[string]string // href → epub:type ("toc", "cover", …) // InSpine puts the navigation document in the spine, which real books do and which is exactly the case // the service-page exclusion has to handle. InSpine bool // Missing declares the navigation document in the manifest and omits its bytes. Missing bool } // GuideRef is one EPUB 2 `<guide><reference>`: how a book with no navigation document declares that a // spine document is its toc or its cover. type GuideRef struct{ Type, Href string } // EPUB is a fixture book. The zero value plus Chapters/Spine reproduces what BuildEPUB has always built; // Nav, NCX and Guide add the structure layers the format work reads. type EPUB struct { Chapters []Chapter Spine []SpineRef Nav *Nav NCX []string // EPUB 2 navPoint hrefs, in order; nesting is not needed to test flattening // NCXBroken writes bytes that are not an NCX, for a book whose spine names one it cannot parse. NCXBroken bool // NCXMissing names an NCX in the spine and omits its bytes. NCXMissing bool Guide []GuideRef } // entryName is where this chapter's bytes are written inside OEBPS/. func (c Chapter) entryName() string { if c.EntryName != "" { return c.EntryName } return c.Href } // BuildEPUB writes a minimal but real epub (container.xml → OPF → spine → xhtml) // to a temp .epub file and returns its path. spineIDs gives the reading order // (may differ from manifest order to prove the spine drives it). func BuildEPUB(t *testing.T, chapters []Chapter, spineIDs []string) string { t.Helper() path := filepath.Join(t.TempDir(), "book.epub") BuildEPUBAt(t, path, chapters, spineIDs) return path } // Build writes the described book to a temp .epub and returns its path. func (e EPUB) Build(t *testing.T) string { t.Helper() path := filepath.Join(t.TempDir(), "book.epub") e.BuildAt(t, path) return path } // BuildEPUBAt writes the epub to a caller-chosen path (used by the runner e2e). func BuildEPUBAt(t *testing.T, path string, chapters []Chapter, spineIDs []string) { t.Helper() refs := make([]SpineRef, 0, len(spineIDs)) for _, id := range spineIDs { refs = append(refs, SpineRef{ID: id}) } EPUB{Chapters: chapters, Spine: refs}.BuildAt(t, path) } // BuildAt writes the described book to a caller-chosen path. func (e EPUB) BuildAt(t *testing.T, path string) { t.Helper() chapters := append([]Chapter(nil), e.Chapters...) refs := e.Spine f, err := os.Create(path) if err != nil { t.Fatal(err) } defer f.Close() // OCF requires "mimetype" first in the archive and STORED, not deflated, and container.xml pointing // at the OPF. The production container writes both (bookfile.NewContainer); the fixture rides it so // it stays the real epub shape the reader will meet, by construction rather than by copy. c, err := bookfile.NewContainer(f, "OEBPS/content.opf") if err != nil { t.Fatal(err) } add := func(name, content string) { if err := c.Add(name, []byte(content)); err != nil { t.Fatal(err) } } // The navigation document is a manifest item like any other, and real books put it in the spine too. if e.Nav != nil { href := e.Nav.Href if href == "" { href = "nav.xhtml" } chapters = append(chapters, Chapter{ID: "navdoc", Href: href, Properties: "nav", Raw: true, Body: navXHTML(*e.Nav), Declared: e.Nav.Missing}) if e.Nav.InSpine { refs = append([]SpineRef{{ID: "navdoc"}}, refs...) } } if len(e.NCX) > 0 || e.NCXBroken || e.NCXMissing { body := ncxXML(e.NCX) if e.NCXBroken { body = "<ncx><navMap><navPoint><unclosed></navMap>" // not an NCX: xml.Unmarshal refuses it } chapters = append(chapters, Chapter{ID: "ncx", Href: "toc.ncx", MType: "application/x-dtbncx+xml", Raw: true, Body: body, Declared: e.NCXMissing}) } var manifest, spine strings.Builder for _, c := range chapters { mt := c.MType if mt == "" { mt = "application/xhtml+xml" } props := "" if c.Properties != "" { props = ` properties="` + c.Properties + `"` } manifest.WriteString(`<item id="` + c.ID + `" href="` + c.Href + `" media-type="` + mt + `"` + props + `/>` + "\n") } for _, r := range refs { lin := "" if r.Linear != "" { lin = ` linear="` + r.Linear + `"` } spine.WriteString(`<itemref idref="` + r.ID + `"` + lin + `/>` + "\n") } spineAttr := "" if len(e.NCX) > 0 || e.NCXBroken || e.NCXMissing { spineAttr = ` toc="ncx"` } var guide strings.Builder if len(e.Guide) > 0 { guide.WriteString("<guide>") for _, g := range e.Guide { guide.WriteString(`<reference type="` + g.Type + `" href="` + g.Href + `"/>`) } guide.WriteString("</guide>") } add("OEBPS/content.opf", `<?xml version="1.0" encoding="utf-8"?> <package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="uid"> <metadata xmlns:dc="http://purl.org/dc/elements/1.1/"><dc:title>Test</dc:title></metadata> <manifest>`+manifest.String()+`</manifest> <spine`+spineAttr+`>`+spine.String()+`</spine> `+guide.String()+` </package>`) for _, c := range chapters { // The FILE is (x)html when its ENTRY name says so, regardless of how the manifest // media-type labels it — real epubs mislabel xhtml as application/xml, etc. The entry // name (not the href) decides, because an href may be percent-encoded or fragment-bearing. entry := c.entryName() if c.Declared { continue // promised in the manifest, absent from the archive } if c.Raw { add("OEBPS/"+entry, c.Body) continue } switch strings.ToLower(filepath.Ext(entry)) { case ".xhtml", ".html", ".htm", ".xml": add("OEBPS/"+entry, `<?xml version="1.0" encoding="utf-8"?> <html xmlns="http://www.w3.org/1999/xhtml"><head><title>c `+c.Body+``) default: add("OEBPS/"+entry, c.Body) // non-xhtml asset (e.g. image bytes stand-in) } } if err := c.Close(); err != nil { t.Fatal(err) } } // navXHTML renders an EPUB 3 navigation document: a toc nav and, when landmarks are given, a landmarks nav. // The toc is deliberately NESTED so the flattening rule is exercised by every fixture that uses it. func navXHTML(n Nav) string { var toc strings.Builder for i, h := range n.TOC { // ⚠ The label is DISTINCT per target and NAMES ITS SOURCE. Writing the same string for every entry — // as this builder first did — makes a fixture that cannot tell a nav label from an NCX one, nor the // first entry for a document from the second, so an assertion on titles would pass while reading // them from the wrong place entirely. label := "NAV " + h // The LAST entry is nested one level deeper, so every fixture using a nav exercises flattening. if i == len(n.TOC)-1 && len(n.TOC) > 1 { toc.WriteString(`
  1. ` + label + `
`) continue } toc.WriteString(`
  • ` + label + `
  • `) } var lm strings.Builder if len(n.Landmarks) > 0 { lm.WriteString(``) } return ` nav ` + lm.String() + `` } // ncxXML renders an EPUB 2 NCX with one navPoint per href. func ncxXML(hrefs []string) string { var b strings.Builder for i, h := range hrefs { b.WriteString(`NCX ` + h + ``) } return ` ` + b.String() + `` }