tests/engine/epub.test.mjs (15036 bytes)
1 /* The read-along epub: made from a small EPUB 2 book with the markup that makes 2 * this hard (inline elements, furigana, pages in folders, a space in a file 3 * name), then read back and checked the way a reader would use it. 4 * 5 * node --test tests/engine 6 * 7 * With EPUBCHECK=<path to epubcheck.jar> the W3C validator judges the result too. 8 */ 9 import test from 'node:test'; 10 import assert from 'node:assert/strict'; 11 import { existsSync, mkdtempSync, readFileSync, writeFileSync, rmSync } from 'node:fs'; 12 import { spawnSync } from 'node:child_process'; 13 import { tmpdir } from 'node:os'; 14 import path from 'node:path'; 15 import { JSDOM } from 'jsdom'; 16 17 const { window } = new JSDOM(''); 18 Object.assign(globalThis, { DOMParser: window.DOMParser, XMLSerializer: window.XMLSerializer }); 19 20 const { syncedEpub, writeZip, relative, clock } = await import('../../frontend/engine/epub.js'); 21 const { zipEntries, readBook, decode } = await import('../../frontend/engine/book.js'); 22 const E = await import('../../frontend/engine/align.js'); 23 24 const enc = new TextEncoder(); 25 const bare = (s) => s.replace(/\s+/g, ''); 26 27 const page = (body) => `<?xml version="1.0" encoding="utf-8"?> 28 <html xmlns="http://www.w3.org/1999/xhtml"><head><title>A Test</title></head><body>${body}</body></html>`; 29 30 async function book() { 31 const files = { 32 mimetype: 'application/epub+zip', 33 'META-INF/container.xml': `<?xml version="1.0"?><container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container"> 34 <rootfiles><rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/></rootfiles></container>`, 35 'OEBPS/content.opf': `<?xml version="1.0"?><package xmlns="http://www.idpf.org/2007/opf" version="2.0" unique-identifier="id"> 36 <metadata xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:opf="http://www.idpf.org/2007/opf"> 37 <dc:title>A Test</dc:title><dc:language>en</dc:language><dc:identifier id="id">urn:uuid:0e6a7c8e-1111-4222-8333-444455556666</dc:identifier> 38 <dc:creator opf:role="aut">Nobody</dc:creator><dc:date opf:event="publication">2001-01-01</dc:date><dc:date opf:event="modification">2002-02-02</dc:date> 39 </metadata> 40 <manifest> 41 <item id="ncx" href="toc.ncx" media-type="application/x-dtbncx+xml"/> 42 <item id="front" href="front.xhtml" media-type="application/xhtml+xml"/> 43 <item id="one" href="text/chapter%20one.xhtml" media-type="application/xhtml+xml"/> 44 <item id="two" href="text/two.xhtml" media-type="application/xhtml+xml"/> 45 </manifest> 46 <spine toc="ncx"><itemref idref="front"/><itemref idref="one"/><itemref idref="two"/></spine></package>`, 47 'OEBPS/toc.ncx': `<?xml version="1.0"?><ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" version="2005-1"><head> 48 <meta name="dtb:uid" content="urn:uuid:0e6a7c8e-1111-4222-8333-444455556666"/></head><docTitle><text>A Test</text></docTitle><navMap> 49 <navPoint id="n1" playOrder="1"><navLabel><text>One & only</text></navLabel><content src="text/chapter%20one.xhtml"/></navPoint> 50 <navPoint id="n2" playOrder="2"><navLabel><text>Two</text></navLabel><content src="text/two.xhtml#top"/></navPoint></navMap></ncx>`, 51 'OEBPS/front.xhtml': page('<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 10 10"><title>Cover</title><rect width="10" height="10"/></svg><h1>A Test</h1><p>Printed somewhere, by someone. Nobody reads this aloud.</p>'), 52 'OEBPS/text/chapter one.xhtml': page('<h2>Chapter one</h2><p>It was a <em>dark and stormy</em> night; the rain fell in torrents.</p>' + 53 '<p>Yes.</p><p>She said <no> & left.</p>'), 54 'OEBPS/text/two.xhtml': page('<h2 id="top">Two</h2><p><ruby>吾輩<rt>わがはい</rt></ruby>は<ruby>猫<rt>ねこ</rt></ruby>である。名前はまだ無い。</p><p>Yes.</p>'), 55 }; 56 const zip = await writeZip(Object.entries(files).map(([name, s]) => ({ name, data: enc.encode(s) }))); 57 return new File([zip], 'test.epub'); 58 } 59 60 const cues = [ 61 { text: 'Chapter one It was a dark', start: 1, end: 3 }, // a heading, a paragraph, and into <em> 62 { text: 'and stormy night; the rain fell in torrents.', start: 3, end: 6 }, 63 { text: E.UNMATCHED + 'mumble', start: 6, end: 7 }, // not from the book: no place in it 64 { text: 'Yes.', start: 7, end: 8 }, 65 { text: 'She said <no> & left.', start: 8, end: 10 }, 66 { text: 'Two 吾輩は猫である。', start: 61, end: 64 }, // in the second audio file 67 { text: '名前はまだ無い。', start: 64, end: 66 }, 68 { text: 'Yes.', start: 66, end: 67 }, 69 ]; 70 const audio = (n) => new File([new Uint8Array(2000).fill(n)], `part${n}.mp3`); 71 const parts = [{ file: audio(1), duration: 60, codec: 'mp3' }, { file: audio(2), duration: 30, codec: 'mp3' }]; 72 73 test('each cue is tied to its own words in the pages, and to its stretch of audio', async () => { 74 const { file, located, of } = await syncedEpub({ book: await book(), cues, parts, stem: 'test' }); 75 assert.equal(of, 7); 76 assert.equal(located, 7); 77 assert.equal(file.name, 'test.read-along.epub'); 78 79 const bytes = new Uint8Array(await file.arrayBuffer()); 80 // A reader knows an epub by these bytes at these places. 81 assert.equal(decode(bytes.subarray(30, 38)), 'mimetype'); 82 assert.equal(decode(bytes.subarray(38, 58)), 'application/epub+zip'); 83 84 const entries = zipEntries(bytes); 85 const read = async (name) => decode(await entries.get(name)()); 86 const xml = (s) => new window.DOMParser().parseFromString(s, 'application/xml'); 87 88 const opf = xml(await read('OEBPS/content.opf')); 89 assert.equal(opf.documentElement.getAttribute('version'), '3.0'); 90 const items = [...opf.querySelectorAll('manifest > item')]; 91 const item = (id) => items.find((i) => i.getAttribute('id') === id); 92 assert.equal(item('front').getAttribute('media-overlay'), null, 'nothing of the front page is read'); 93 assert.equal(item('front').getAttribute('properties'), 'svg'); 94 assert.equal(item('one').getAttribute('properties'), null); 95 96 // Follow each overlay as a reader does: SMIL -> the element in the page, and the clip. 97 const said = new Map(); // cue index -> the words its spans hold 98 const clips = []; 99 for (const id of ['one', 'two']) { 100 const smilItem = item(item(id).getAttribute('media-overlay')); 101 assert.equal(smilItem.getAttribute('media-type'), 'application/smil+xml'); 102 const smilPath = `OEBPS/${smilItem.getAttribute('href')}`; 103 const smil = xml(await read(smilPath)); 104 assert.equal(smil.querySelector('parsererror'), null); 105 const from = path.posix.dirname(smilPath); 106 for (const par of smil.querySelectorAll('par')) { 107 const [src, fragment] = par.querySelector('text').getAttribute('src').split('#'); 108 const pagePath = path.posix.normalize(`${from}/${decodeURI(src)}`); 109 assert.equal(pagePath, `OEBPS/${decodeURIComponent(item(id).getAttribute('href'))}`); 110 const doc = xml(await read(pagePath)); 111 assert.equal(doc.querySelector('parsererror'), null, `${pagePath} is not well-formed`); 112 const el = doc.getElementById(fragment); 113 assert.ok(el, `${fragment} is not in ${pagePath}`); 114 const index = Number(fragment.split('-')[1]); 115 said.set(index, (said.get(index) ?? '') + el.textContent); 116 117 const a = par.querySelector('audio'); 118 const audioPath = path.posix.normalize(`${from}/${decodeURI(a.getAttribute('src'))}`); 119 assert.ok(entries.has(audioPath), audioPath); 120 clips.push({ index, audioPath, begin: parseFloat(a.getAttribute('clipBegin')), end: parseFloat(a.getAttribute('clipEnd')) }); 121 } 122 } 123 cues.forEach((cue, i) => { 124 if (cue.text.startsWith(E.UNMATCHED)) assert.ok(!said.has(i)); 125 else assert.equal(bare(said.get(i)), bare(cue.text), `cue ${i}`); 126 }); 127 128 // The pieces of a cue share its time, in order, inside the right file. 129 for (const [i, cue] of cues.entries()) { 130 const mine = clips.filter((c) => c.index === i); 131 if (!mine.length) continue; 132 const base = cue.start >= 60 ? 60 : 0; 133 assert.ok(mine.every((c) => c.audioPath === `OEBPS/subread/audio-${base ? 2 : 1}.mp3`)); 134 assert.ok(Math.abs(mine[0].begin - (cue.start - base)) < 0.002); 135 assert.ok(Math.abs(mine.at(-1).end - (cue.end - base)) < 0.002); 136 mine.forEach((c, k) => { assert.ok(c.end > c.begin); if (k) assert.ok(Math.abs(c.begin - mine[k - 1].end) < 0.002); }); 137 } 138 assert.equal(clips.filter((c) => c.index === 0).length, 3, 'heading, paragraph text, and the text inside <em>'); 139 140 // The second "Yes." is the one in chapter two, not the first again. 141 const two = xml(await read('OEBPS/text/two.xhtml')); 142 assert.equal(two.getElementById('subread-7').textContent, 'Yes.'); 143 // Furigana is still in the book, and outside the spans. 144 assert.equal(two.querySelectorAll('rt').length, 2); 145 assert.equal([...two.querySelectorAll('span[id^=subread]')].some((s) => s.querySelector('rt') || s.closest('rt')), false); 146 // The words of the book are as they were. 147 const before = await readBook(await book()); 148 const after = await readBook(file); 149 assert.deepEqual(after.paragraphs, before.paragraphs); 150 151 // The audio is in the book, byte for byte. 152 assert.deepEqual(await entries.get('OEBPS/subread/audio-2.mp3')(), new Uint8Array(2000).fill(2)); 153 154 // What EPUB 3 asks for that EPUB 2 did not have. 155 const metas = [...opf.querySelectorAll('metadata > meta')]; 156 const meta = (property) => metas.filter((m) => m.getAttribute('property') === property); 157 assert.equal(meta('dcterms:modified').length, 1); 158 assert.equal(meta('media:active-class')[0].textContent, '-epub-media-overlay-active'); 159 assert.equal(meta('media:duration').length, 3); // two overlays and the total 160 assert.equal(meta('media:duration').find((m) => !m.getAttribute('refines')).textContent, clock(2 + 3 + 1 + 2 + 3 + 2 + 1)); 161 const nav = xml(await read(`OEBPS/${items.find((i) => i.getAttribute('properties') === 'nav').getAttribute('href')}`)); 162 assert.deepEqual([...nav.querySelectorAll('a')].map((a) => [a.textContent, a.getAttribute('href')]), 163 [['One & only', '../text/chapter%20one.xhtml'], ['Two', '../text/two.xhtml#top']]); 164 165 const jar = process.env.EPUBCHECK; 166 if (jar && existsSync(jar)) { 167 const dir = mkdtempSync(path.join(tmpdir(), 'subread-')); 168 try { 169 const out = path.join(dir, 'test.epub'); 170 writeFileSync(out, bytes); 171 const r = spawnSync('java', ['-jar', jar, out], { encoding: 'utf8' }); 172 assert.equal(r.status, 0, r.stdout + r.stderr); 173 } finally { rmSync(dir, { recursive: true, force: true }); } 174 } 175 }); 176 177 test('audio that an EPUB 3 reader need not play is refused', async () => { 178 await assert.rejects( 179 syncedEpub({ book: await book(), cues, parts: [{ file: audio(1), duration: 90, codec: 'flac' }], stem: 'x' }), 180 /MP3 or AAC/); 181 }); 182 183 test('subtitles of another book are refused', async () => { 184 await assert.rejects( 185 syncedEpub({ book: await book(), cues: [{ text: 'Call me Ishmael. Some years ago', start: 0, end: 2 }], parts, stem: 'x' }), 186 /None of the subtitles/); 187 }); 188 189 test('a short line is not looked for far ahead', async () => { 190 // "Yes." is next only in chapter two... but here chapter one's is near, so it is found there. 191 const r = await syncedEpub({ book: await book(), cues: [{ text: 'Yes.', start: 0, end: 1 }], parts, stem: 'x' }); 192 assert.equal(r.located, 1); 193 }); 194 195 test('a zip that is written reads back the same, whatever is in it', async () => { 196 let seed = 7; 197 const random = () => (seed = (seed * 1103515245 + 12345) >>> 0) / 2 ** 32; 198 for (let round = 0; round < 25; round++) { 199 const files = Array.from({ length: 1 + Math.floor(random() * 6) }, (_, i) => { 200 const data = Uint8Array.from({ length: Math.floor(random() * 3000) }, () => Math.floor(random() * 256)); 201 return { name: `dir ${i}/ファイル-${round}-${i}.bin`, data: i % 2 ? new Blob([data]) : data, bytes: data }; 202 }); 203 const back = zipEntries(new Uint8Array(await (await writeZip(files)).arrayBuffer())); 204 assert.deepEqual([...back.keys()], files.map((f) => f.name)); 205 for (const f of files) assert.deepEqual(await back.get(f.name)(), f.bytes); 206 } 207 // And a standard tool agrees about the checksums. 208 const dir = mkdtempSync(path.join(tmpdir(), 'subread-')); 209 try { 210 const out = path.join(dir, 'a.zip'); 211 writeFileSync(out, new Uint8Array(await (await writeZip([{ name: 'a.txt', data: enc.encode('hello') }])).arrayBuffer())); 212 const r = spawnSync('python', ['-c', 'import sys,zipfile; z=zipfile.ZipFile(sys.argv[1]); assert z.testzip() is None; print(z.read("a.txt").decode())', out], { encoding: 'utf8' }); 213 if (!r.error) assert.equal(r.stdout.trim(), 'hello', r.stderr); 214 } finally { rmSync(dir, { recursive: true, force: true }); } 215 }); 216 217 test('relative paths', () => { 218 assert.equal(relative('OEBPS/subread', 'OEBPS/text/a b.xhtml'), '../text/a%20b.xhtml'); 219 assert.equal(relative('OEBPS/text', 'OEBPS/text/two.xhtml'), 'two.xhtml'); 220 assert.equal(relative('', 'subread/overlay.css'), 'subread/overlay.css'); 221 assert.equal(relative('a/b', 'c.css'), '../../c.css'); 222 }); 223 224 /* A real book, when it is on this machine (it is in copyright, so it is not in the repository). */ 225 const local = path.join(path.dirname(new URL(import.meta.url).pathname.replace(/^\/(\w:)/, '$1')), 'excerpt-local'); 226 const realBook = path.join(local, 'moskva.epub'), realAudio = path.join(local, 'ru.mp3'); 227 test('a real book: nearly each line is found, and the validator accepts the result', 228 { skip: !(existsSync(realBook) && existsSync(realAudio)) && 'no local book' }, async () => { 229 const fixture = JSON.parse(readFileSync(path.join(local, 'moskva_device.json'), 'utf8')); 230 const bookFile = new File([readFileSync(realBook)], 'moskva.epub'); 231 const { paragraphs } = await readBook(bookFile); 232 const { cues } = E.alignBook(fixture.transcript, paragraphs, E.language(fixture.language)); 233 const duration = fixture.transcript.at(-1).end + 1; 234 const r = await syncedEpub({ 235 book: bookFile, cues, stem: 'moskva', 236 parts: [{ file: new File([readFileSync(realAudio)], 'ru.mp3'), duration, codec: 'mp3' }], 237 }); 238 console.log(`located ${r.located} of ${r.of} cues; ${r.file.size} bytes`); 239 assert.ok(r.located >= 0.95 * r.of, `${r.located} of ${r.of}`); 240 assert.deepEqual((await readBook(r.file)).paragraphs, paragraphs); 241 242 const out = process.env.SUBREAD_KEEP_EPUB; 243 if (out) writeFileSync(out, new Uint8Array(await r.file.arrayBuffer())); 244 const jar = process.env.EPUBCHECK; 245 if (jar && existsSync(jar) && out) { 246 const before = spawnSync('java', ['-jar', jar, realBook], { encoding: 'utf8' }); 247 const after = spawnSync('java', ['-jar', jar, out], { encoding: 'utf8' }); 248 const errors = (s) => Number(/(\d+) errors?/.exec(s.stdout + s.stderr)?.[1] ?? -1); 249 console.log(`epubcheck errors: ${errors(before)} in the source, ${errors(after)} in the result`); 250 console.log((after.stdout + after.stderr).split('\n').filter((l) => /ERROR|FATAL/.test(l)).slice(0, 12).join('\n')); 251 } 252 });