Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


tests/engine/epub.test.mjs (15036 bytes)

1 /* The read-along epub: made from a small EPUB 2 book with the markup that makes
2  * this hard (inline elements, furigana, pages in folders, a space in a file
3  * name), then read back and checked the way a reader would use it.
4  *
5  *   node --test tests/engine
6  *
7  * With EPUBCHECK=<path to epubcheck.jar> the W3C validator judges the result too.
8  */
9 import test from 'node:test';
10 import assert from 'node:assert/strict';
11 import { existsSync, mkdtempSync, readFileSync, writeFileSync, rmSync } from 'node:fs';
12 import { spawnSync } from 'node:child_process';
13 import { tmpdir } from 'node:os';
14 import path from 'node:path';
15 import { JSDOM } from 'jsdom';
16 
17 const { window } = new JSDOM('');
18 Object.assign(globalThis, { DOMParser: window.DOMParser, XMLSerializer: window.XMLSerializer });
19 
20 const { syncedEpub, writeZip, relative, clock } = await import('../../frontend/engine/epub.js');
21 const { zipEntries, readBook, decode } = await import('../../frontend/engine/book.js');
22 const E = await import('../../frontend/engine/align.js');
23 
24 const enc = new TextEncoder();
25 const bare = (s) => s.replace(/\s+/g, '');
26 
27 const page = (body) => `<?xml version="1.0" encoding="utf-8"?>
28 <html xmlns="http://www.w3.org/1999/xhtml"><head><title>A Test</title></head><body>${body}</body></html>`;
29 
30 async function book() {
31   const files = {
32     mimetype: 'application/epub+zip',
33     'META-INF/container.xml': `<?xml version="1.0"?><container version="1.0" xmlns="urn:oasis:names:tc:opendocument:xmlns:container">
34       <rootfiles><rootfile full-path="OEBPS/content.opf" media-type="application/oebps-package+xml"/></rootfiles></container>`,
35     'OEBPS/content.opf': `<?xml version="1.0"?><package xmlns="http://www.idpf.org/2007/opf" version="2.0" unique-identifier="id">
36       <metadata xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:opf="http://www.idpf.org/2007/opf">
37         <dc:title>A Test</dc:title><dc:language>en</dc:language><dc:identifier id="id">urn:uuid:0e6a7c8e-1111-4222-8333-444455556666</dc:identifier>
38         <dc:creator opf:role="aut">Nobody</dc:creator><dc:date opf:event="publication">2001-01-01</dc:date><dc:date opf:event="modification">2002-02-02</dc:date>
39       </metadata>
40       <manifest>
41         <item id="ncx" href="toc.ncx" media-type="application/x-dtbncx+xml"/>
42         <item id="front" href="front.xhtml" media-type="application/xhtml+xml"/>
43         <item id="one" href="text/chapter%20one.xhtml" media-type="application/xhtml+xml"/>
44         <item id="two" href="text/two.xhtml" media-type="application/xhtml+xml"/>
45       </manifest>
46       <spine toc="ncx"><itemref idref="front"/><itemref idref="one"/><itemref idref="two"/></spine></package>`,
47     'OEBPS/toc.ncx': `<?xml version="1.0"?><ncx xmlns="http://www.daisy.org/z3986/2005/ncx/" version="2005-1"><head>
48       <meta name="dtb:uid" content="urn:uuid:0e6a7c8e-1111-4222-8333-444455556666"/></head><docTitle><text>A Test</text></docTitle><navMap>
49       <navPoint id="n1" playOrder="1"><navLabel><text>One &amp; only</text></navLabel><content src="text/chapter%20one.xhtml"/></navPoint>
50       <navPoint id="n2" playOrder="2"><navLabel><text>Two</text></navLabel><content src="text/two.xhtml#top"/></navPoint></navMap></ncx>`,
51     'OEBPS/front.xhtml': page('<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 10 10"><title>Cover</title><rect width="10" height="10"/></svg><h1>A Test</h1><p>Printed somewhere, by someone. Nobody reads this aloud.</p>'),
52     'OEBPS/text/chapter one.xhtml': page('<h2>Chapter one</h2><p>It was a <em>dark and stormy</em> night; the rain fell in torrents.</p>' +
53       '<p>Yes.</p><p>She said &lt;no&gt; &amp; left.</p>'),
54     'OEBPS/text/two.xhtml': page('<h2 id="top">Two</h2><p><ruby>吾輩<rt>わがはい</rt></ruby>は<ruby>猫<rt>ねこ</rt></ruby>である。名前はまだ無い。</p><p>Yes.</p>'),
55   };
56   const zip = await writeZip(Object.entries(files).map(([name, s]) => ({ name, data: enc.encode(s) })));
57   return new File([zip], 'test.epub');
58 }
59 
60 const cues = [
61   { text: 'Chapter one It was a dark', start: 1, end: 3 },            // a heading, a paragraph, and into <em>
62   { text: 'and stormy night; the rain fell in torrents.', start: 3, end: 6 },
63   { text: E.UNMATCHED + 'mumble', start: 6, end: 7 },                 // not from the book: no place in it
64   { text: 'Yes.', start: 7, end: 8 },
65   { text: 'She said <no> & left.', start: 8, end: 10 },
66   { text: 'Two 吾輩は猫である。', start: 61, end: 64 },                   // in the second audio file
67   { text: '名前はまだ無い。', start: 64, end: 66 },
68   { text: 'Yes.', start: 66, end: 67 },
69 ];
70 const audio = (n) => new File([new Uint8Array(2000).fill(n)], `part${n}.mp3`);
71 const parts = [{ file: audio(1), duration: 60, codec: 'mp3' }, { file: audio(2), duration: 30, codec: 'mp3' }];
72 
73 test('each cue is tied to its own words in the pages, and to its stretch of audio', async () => {
74   const { file, located, of } = await syncedEpub({ book: await book(), cues, parts, stem: 'test' });
75   assert.equal(of, 7);
76   assert.equal(located, 7);
77   assert.equal(file.name, 'test.read-along.epub');
78 
79   const bytes = new Uint8Array(await file.arrayBuffer());
80   // A reader knows an epub by these bytes at these places.
81   assert.equal(decode(bytes.subarray(30, 38)), 'mimetype');
82   assert.equal(decode(bytes.subarray(38, 58)), 'application/epub+zip');
83 
84   const entries = zipEntries(bytes);
85   const read = async (name) => decode(await entries.get(name)());
86   const xml = (s) => new window.DOMParser().parseFromString(s, 'application/xml');
87 
88   const opf = xml(await read('OEBPS/content.opf'));
89   assert.equal(opf.documentElement.getAttribute('version'), '3.0');
90   const items = [...opf.querySelectorAll('manifest > item')];
91   const item = (id) => items.find((i) => i.getAttribute('id') === id);
92   assert.equal(item('front').getAttribute('media-overlay'), null, 'nothing of the front page is read');
93   assert.equal(item('front').getAttribute('properties'), 'svg');
94   assert.equal(item('one').getAttribute('properties'), null);
95 
96   // Follow each overlay as a reader does: SMIL -> the element in the page, and the clip.
97   const said = new Map();                    // cue index -> the words its spans hold
98   const clips = [];
99   for (const id of ['one', 'two']) {
100     const smilItem = item(item(id).getAttribute('media-overlay'));
101     assert.equal(smilItem.getAttribute('media-type'), 'application/smil+xml');
102     const smilPath = `OEBPS/${smilItem.getAttribute('href')}`;
103     const smil = xml(await read(smilPath));
104     assert.equal(smil.querySelector('parsererror'), null);
105     const from = path.posix.dirname(smilPath);
106     for (const par of smil.querySelectorAll('par')) {
107       const [src, fragment] = par.querySelector('text').getAttribute('src').split('#');
108       const pagePath = path.posix.normalize(`${from}/${decodeURI(src)}`);
109       assert.equal(pagePath, `OEBPS/${decodeURIComponent(item(id).getAttribute('href'))}`);
110       const doc = xml(await read(pagePath));
111       assert.equal(doc.querySelector('parsererror'), null, `${pagePath} is not well-formed`);
112       const el = doc.getElementById(fragment);
113       assert.ok(el, `${fragment} is not in ${pagePath}`);
114       const index = Number(fragment.split('-')[1]);
115       said.set(index, (said.get(index) ?? '') + el.textContent);
116 
117       const a = par.querySelector('audio');
118       const audioPath = path.posix.normalize(`${from}/${decodeURI(a.getAttribute('src'))}`);
119       assert.ok(entries.has(audioPath), audioPath);
120       clips.push({ index, audioPath, begin: parseFloat(a.getAttribute('clipBegin')), end: parseFloat(a.getAttribute('clipEnd')) });
121     }
122   }
123   cues.forEach((cue, i) => {
124     if (cue.text.startsWith(E.UNMATCHED)) assert.ok(!said.has(i));
125     else assert.equal(bare(said.get(i)), bare(cue.text), `cue ${i}`);
126   });
127 
128   // The pieces of a cue share its time, in order, inside the right file.
129   for (const [i, cue] of cues.entries()) {
130     const mine = clips.filter((c) => c.index === i);
131     if (!mine.length) continue;
132     const base = cue.start >= 60 ? 60 : 0;
133     assert.ok(mine.every((c) => c.audioPath === `OEBPS/subread/audio-${base ? 2 : 1}.mp3`));
134     assert.ok(Math.abs(mine[0].begin - (cue.start - base)) < 0.002);
135     assert.ok(Math.abs(mine.at(-1).end - (cue.end - base)) < 0.002);
136     mine.forEach((c, k) => { assert.ok(c.end > c.begin); if (k) assert.ok(Math.abs(c.begin - mine[k - 1].end) < 0.002); });
137   }
138   assert.equal(clips.filter((c) => c.index === 0).length, 3, 'heading, paragraph text, and the text inside <em>');
139 
140   // The second "Yes." is the one in chapter two, not the first again.
141   const two = xml(await read('OEBPS/text/two.xhtml'));
142   assert.equal(two.getElementById('subread-7').textContent, 'Yes.');
143   // Furigana is still in the book, and outside the spans.
144   assert.equal(two.querySelectorAll('rt').length, 2);
145   assert.equal([...two.querySelectorAll('span[id^=subread]')].some((s) => s.querySelector('rt') || s.closest('rt')), false);
146   // The words of the book are as they were.
147   const before = await readBook(await book());
148   const after = await readBook(file);
149   assert.deepEqual(after.paragraphs, before.paragraphs);
150 
151   // The audio is in the book, byte for byte.
152   assert.deepEqual(await entries.get('OEBPS/subread/audio-2.mp3')(), new Uint8Array(2000).fill(2));
153 
154   // What EPUB 3 asks for that EPUB 2 did not have.
155   const metas = [...opf.querySelectorAll('metadata > meta')];
156   const meta = (property) => metas.filter((m) => m.getAttribute('property') === property);
157   assert.equal(meta('dcterms:modified').length, 1);
158   assert.equal(meta('media:active-class')[0].textContent, '-epub-media-overlay-active');
159   assert.equal(meta('media:duration').length, 3);              // two overlays and the total
160   assert.equal(meta('media:duration').find((m) => !m.getAttribute('refines')).textContent, clock(2 + 3 + 1 + 2 + 3 + 2 + 1));
161   const nav = xml(await read(`OEBPS/${items.find((i) => i.getAttribute('properties') === 'nav').getAttribute('href')}`));
162   assert.deepEqual([...nav.querySelectorAll('a')].map((a) => [a.textContent, a.getAttribute('href')]),
163     [['One & only', '../text/chapter%20one.xhtml'], ['Two', '../text/two.xhtml#top']]);
164 
165   const jar = process.env.EPUBCHECK;
166   if (jar && existsSync(jar)) {
167     const dir = mkdtempSync(path.join(tmpdir(), 'subread-'));
168     try {
169       const out = path.join(dir, 'test.epub');
170       writeFileSync(out, bytes);
171       const r = spawnSync('java', ['-jar', jar, out], { encoding: 'utf8' });
172       assert.equal(r.status, 0, r.stdout + r.stderr);
173     } finally { rmSync(dir, { recursive: true, force: true }); }
174   }
175 });
176 
177 test('audio that an EPUB 3 reader need not play is refused', async () => {
178   await assert.rejects(
179     syncedEpub({ book: await book(), cues, parts: [{ file: audio(1), duration: 90, codec: 'flac' }], stem: 'x' }),
180     /MP3 or AAC/);
181 });
182 
183 test('subtitles of another book are refused', async () => {
184   await assert.rejects(
185     syncedEpub({ book: await book(), cues: [{ text: 'Call me Ishmael. Some years ago', start: 0, end: 2 }], parts, stem: 'x' }),
186     /None of the subtitles/);
187 });
188 
189 test('a short line is not looked for far ahead', async () => {
190   // "Yes." is next only in chapter two... but here chapter one's is near, so it is found there.
191   const r = await syncedEpub({ book: await book(), cues: [{ text: 'Yes.', start: 0, end: 1 }], parts, stem: 'x' });
192   assert.equal(r.located, 1);
193 });
194 
195 test('a zip that is written reads back the same, whatever is in it', async () => {
196   let seed = 7;
197   const random = () => (seed = (seed * 1103515245 + 12345) >>> 0) / 2 ** 32;
198   for (let round = 0; round < 25; round++) {
199     const files = Array.from({ length: 1 + Math.floor(random() * 6) }, (_, i) => {
200       const data = Uint8Array.from({ length: Math.floor(random() * 3000) }, () => Math.floor(random() * 256));
201       return { name: `dir ${i}/ファイル-${round}-${i}.bin`, data: i % 2 ? new Blob([data]) : data, bytes: data };
202     });
203     const back = zipEntries(new Uint8Array(await (await writeZip(files)).arrayBuffer()));
204     assert.deepEqual([...back.keys()], files.map((f) => f.name));
205     for (const f of files) assert.deepEqual(await back.get(f.name)(), f.bytes);
206   }
207   // And a standard tool agrees about the checksums.
208   const dir = mkdtempSync(path.join(tmpdir(), 'subread-'));
209   try {
210     const out = path.join(dir, 'a.zip');
211     writeFileSync(out, new Uint8Array(await (await writeZip([{ name: 'a.txt', data: enc.encode('hello') }])).arrayBuffer()));
212     const r = spawnSync('python', ['-c', 'import sys,zipfile; z=zipfile.ZipFile(sys.argv[1]); assert z.testzip() is None; print(z.read("a.txt").decode())', out], { encoding: 'utf8' });
213     if (!r.error) assert.equal(r.stdout.trim(), 'hello', r.stderr);
214   } finally { rmSync(dir, { recursive: true, force: true }); }
215 });
216 
217 test('relative paths', () => {
218   assert.equal(relative('OEBPS/subread', 'OEBPS/text/a b.xhtml'), '../text/a%20b.xhtml');
219   assert.equal(relative('OEBPS/text', 'OEBPS/text/two.xhtml'), 'two.xhtml');
220   assert.equal(relative('', 'subread/overlay.css'), 'subread/overlay.css');
221   assert.equal(relative('a/b', 'c.css'), '../../c.css');
222 });
223 
224 /* A real book, when it is on this machine (it is in copyright, so it is not in the repository). */
225 const local = path.join(path.dirname(new URL(import.meta.url).pathname.replace(/^\/(\w:)/, '$1')), 'excerpt-local');
226 const realBook = path.join(local, 'moskva.epub'), realAudio = path.join(local, 'ru.mp3');
227 test('a real book: nearly each line is found, and the validator accepts the result',
228   { skip: !(existsSync(realBook) && existsSync(realAudio)) && 'no local book' }, async () => {
229     const fixture = JSON.parse(readFileSync(path.join(local, 'moskva_device.json'), 'utf8'));
230     const bookFile = new File([readFileSync(realBook)], 'moskva.epub');
231     const { paragraphs } = await readBook(bookFile);
232     const { cues } = E.alignBook(fixture.transcript, paragraphs, E.language(fixture.language));
233     const duration = fixture.transcript.at(-1).end + 1;
234     const r = await syncedEpub({
235       book: bookFile, cues, stem: 'moskva',
236       parts: [{ file: new File([readFileSync(realAudio)], 'ru.mp3'), duration, codec: 'mp3' }],
237     });
238     console.log(`located ${r.located} of ${r.of} cues; ${r.file.size} bytes`);
239     assert.ok(r.located >= 0.95 * r.of, `${r.located} of ${r.of}`);
240     assert.deepEqual((await readBook(r.file)).paragraphs, paragraphs);
241 
242     const out = process.env.SUBREAD_KEEP_EPUB;
243     if (out) writeFileSync(out, new Uint8Array(await r.file.arrayBuffer()));
244     const jar = process.env.EPUBCHECK;
245     if (jar && existsSync(jar) && out) {
246       const before = spawnSync('java', ['-jar', jar, realBook], { encoding: 'utf8' });
247       const after = spawnSync('java', ['-jar', jar, out], { encoding: 'utf8' });
248       const errors = (s) => Number(/(\d+) errors?/.exec(s.stdout + s.stderr)?.[1] ?? -1);
249       console.log(`epubcheck errors: ${errors(before)} in the source, ${errors(after)} in the result`);
250       console.log((after.stdout + after.stderr).split('\n').filter((l) => /ERROR|FATAL/.test(l)).slice(0, 12).join('\n'));
251     }
252   });