Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


frontend/engine/epub.js (17282 bytes)

1 /* A read-along book: the epub, with the narration inside it and each line of
2  * text tied to its stretch of audio (EPUB 3 Media Overlays). Thorium,
3  * Storyteller and other EPUB 3 readers play it and highlight the text.
4  *
5  * The cues say what was read and when, but not where in the book's pages the
6  * words are. That is found again by text: a cue's words, without white space,
7  * are looked for in the book's words, from the end of the cue before. Each
8  * find is wrapped in <span id>. A cue that crosses markup (<em>, ruby, the end
9  * of a paragraph) gets several spans, and its time is divided among them by
10  * their length.
11  *
12  * Made in the browser, like everything else: nothing is uploaded.
13  */
14 import { zipEntries, leafBlocks, page, resolve, decode } from './book.js';
15 import { UNMATCHED } from './align.js';
16 
17 const XHTML = 'http://www.w3.org/1999/xhtml';
18 const OPF = 'http://www.idpf.org/2007/opf';
19 const ACTIVE = '-epub-media-overlay-active';
20 const DIR = 'subread';                       // what this adds, beside the package file
21 const SHOW_TEXT = 4;
22 
23 // A jump this far ahead is believed only from a cue this long: a short line
24 // ("Yes.") is found by chance somewhere in any book.
25 const FAR = 4000;
26 const LONG_ENOUGH = 12;
27 
28 /** Audio an EPUB 3 reader must play, by codec: file extension and media type. */
29 const AUDIO = { mp3: ['mp3', 'audio/mpeg'], aac: ['m4a', 'audio/mp4'] };
30 
31 export class NotAnEpub extends Error {}
32 
33 /**
34  * @param book   the epub, a File
35  * @param cues   [{ text, start, end }], seconds on the clock of all the audio
36  * @param parts  [{ file, duration, codec }], in playback order
37  * @returns {{ file: File, located: number, of: number }}
38  */
39 export async function syncedEpub({ book, cues, parts, stem }) {
40   for (const p of parts) {
41     if (!AUDIO[p.codec]) throw new Error(`A read-along book needs MP3 or AAC (m4a, m4b) audio; this is ${p.codec}.`);
42   }
43   const entries = zipEntries(new Uint8Array(await book.arrayBuffer()));
44   const text = async (name) => decode(await entries.get(name)());
45   const xml = (s) => new DOMParser().parseFromString(s, 'application/xml');
46 
47   if (!entries.has('META-INF/container.xml')) throw new NotAnEpub('This epub has no package file.');
48   const opfPath = xml(await text('META-INF/container.xml')).querySelector('rootfile')?.getAttribute('full-path');
49   if (!opfPath || !entries.has(opfPath)) throw new NotAnEpub('This epub has no package file.');
50   const opf = xml(await text(opfPath));
51   if (opf.querySelector('parsererror')) throw new NotAnEpub('The package file of this epub cannot be read.');
52   const base = opfPath.includes('/') ? opfPath.slice(0, opfPath.lastIndexOf('/')) : '';
53   const inBase = (p) => (base ? `${base}/${p}` : p);
54 
55   const items = new Map();                   // zip path -> manifest item
56   for (const item of opf.querySelectorAll('manifest > item')) {
57     items.set(resolve(base, decodeURIComponent((item.getAttribute('href') ?? '').split('#')[0])), item);
58   }
59   const byId = new Map([...items].map(([p, item]) => [item.getAttribute('id'), p]));
60   const order = [...opf.querySelectorAll('spine > itemref')]
61     .filter((r) => r.getAttribute('linear') !== 'no')
62     .map((r) => byId.get(r.getAttribute('idref')))
63     .filter((p) => p && entries.has(p));
64   if (!order.length) throw new NotAnEpub('This epub has no reading order.');
65 
66   // The book's words, without white space, and where each one is.
67   const pages = [];
68   const runs = [];
69   const words = [];
70   let g = 0;
71   for (const path of order) {
72     const doc = page(await text(path));
73     const pg = { path, doc, spans: [] };
74     pages.push(pg);
75     for (const node of textNodes(doc)) {
76       const offsets = [];
77       for (let i = 0; i < node.data.length; i++) if (/\S/.test(node.data[i])) offsets.push(i);
78       if (!offsets.length) continue;
79       runs.push({ page: pg, node, g0: g, offsets, wraps: [] });
80       words.push(offsets.map((i) => node.data[i]).join(''));
81       g += offsets.length;
82     }
83   }
84   const all = words.join('');
85 
86   // Where each cue is.
87   let cursor = 0, r = 0, located = 0, of = 0;
88   cues.forEach((cue, index) => {
89     if (cue.text.startsWith(UNMATCHED) || !(cue.end > cue.start)) return;
90     const want = cue.text.replace(/\s+/g, '');
91     if (!want) return;
92     of++;
93     const at = all.indexOf(want, cursor);
94     if (at < 0 || (at - cursor > FAR && want.length < LONG_ENOUGH)) return;
95     located++;
96     cursor = at + want.length;
97 
98     while (runs[r].g0 + runs[r].offsets.length <= at) r++;
99     const pieces = [];
100     for (let k = r; k < runs.length && runs[k].g0 < cursor; k++) {
101       const run = runs[k];
102       const s = Math.max(at, run.g0) - run.g0, e = Math.min(cursor, run.g0 + run.offsets.length) - run.g0;
103       pieces.push({ run, from: run.offsets[s], to: run.offsets[e - 1] + 1, chars: e - s });
104     }
105     let t = cue.start;
106     pieces.forEach((piece, k) => {
107       const id = `subread-${index}${k ? `-${k}` : ''}`;
108       const until = k === pieces.length - 1 ? cue.end : t + (cue.end - cue.start) * piece.chars / want.length;
109       piece.run.wraps.push({ from: piece.from, to: piece.to, id });
110       piece.run.page.spans.push({ id, start: t, end: until });
111       t = until;
112     });
113   });
114   if (!located) throw new Error('None of the subtitles could be found in the pages of this epub.');
115 
116   for (const run of runs) if (run.wraps.length) wrap(run);
117 
118   // The new files.
119   const out = new Map();                     // zip path -> Uint8Array | Blob
120   const manifest = opf.querySelector('manifest');
121   const metadata = opf.querySelector('metadata');
122   const add = (parent, name, attributes, content) => {
123     const el = opf.createElementNS(OPF, name);
124     for (const [k, v] of Object.entries(attributes)) el.setAttribute(k, v);
125     if (content != null) el.textContent = content;
126     parent.appendChild(el);
127     return el;
128   };
129   const encoder = new TextEncoder();
130 
131   const audio = parts.map((p, i) => {
132     const [extension, type] = AUDIO[p.codec];
133     const path = `${DIR}/audio-${i + 1}.${extension}`;
134     out.set(inBase(path), p.file);
135     add(manifest, 'item', { id: `subread-audio-${i + 1}`, href: path, 'media-type': type });
136     return path;
137   });
138   out.set(inBase(`${DIR}/overlay.css`), encoder.encode(`.${ACTIVE} { background: #ffe08a; color: #000; }\n`));
139   add(manifest, 'item', { id: 'subread-css', href: `${DIR}/overlay.css`, 'media-type': 'text/css' });
140 
141   let total = 0;
142   pages.forEach((pg, n) => {
143     if (!pg.spans.length) return;
144     const smilPath = `${DIR}/page-${n + 1}.smil`;
145     const smilDir = inBase(DIR);
146     const lines = pg.spans.map((s, k) => {
147       const clip = clipOf(s, parts);
148       total += clip.end - clip.begin;
149       return `<par id="par-${k + 1}"><text src="${attr(relative(smilDir, pg.path))}#${s.id}"/>` +
150         `<audio src="${attr(relative(smilDir, inBase(audio[clip.part])))}" clipBegin="${clip.begin.toFixed(3)}s" clipEnd="${clip.end.toFixed(3)}s"/></par>`;
151     });
152     out.set(inBase(smilPath), encoder.encode(
153       '<?xml version="1.0" encoding="utf-8"?>\n' +
154       '<smil xmlns="http://www.w3.org/ns/SMIL" xmlns:epub="http://www.idpf.org/2007/ops" version="3.0">\n<body>\n' +
155       `<seq id="seq" epub:textref="${attr(relative(smilDir, pg.path))}" epub:type="bodymatter">\n${lines.join('\n')}\n</seq>\n</body>\n</smil>\n`));
156 
157     const id = `subread-smil-${n + 1}`;
158     add(manifest, 'item', { id, href: smilPath, 'media-type': 'application/smil+xml' });
159     items.get(pg.path).setAttribute('media-overlay', id);
160     const seconds = pg.spans.reduce((sum, s) => { const c = clipOf(s, parts); return sum + c.end - c.begin; }, 0);
161     add(metadata, 'meta', { property: 'media:duration', refines: `#${id}` }, clock(seconds));
162 
163     const head = pg.doc.querySelector('head');
164     if (head) {
165       const link = pg.doc.createElementNS(XHTML, 'link');
166       link.setAttribute('rel', 'stylesheet');
167       link.setAttribute('type', 'text/css');
168       link.setAttribute('href', relative(dirOf(pg.path), inBase(`${DIR}/overlay.css`)));
169       head.appendChild(link);
170     }
171     out.set(pg.path, encoder.encode(serialize(pg.doc)));
172   });
173   add(metadata, 'meta', { property: 'media:duration' }, clock(total));
174   add(metadata, 'meta', { property: 'media:active-class' }, ACTIVE);
175 
176   await toEpub3(opf, { entries, items, order, pages, base, out, add, text });
177   out.set(opfPath, encoder.encode(serialize(opf)));
178 
179   // mimetype first and not compressed: that is how a reader knows an epub.
180   const files = [{ name: 'mimetype', data: encoder.encode('application/epub+zip') }];
181   for (const [name, read] of entries) {
182     if (name !== 'mimetype' && !out.has(name)) files.push({ name, data: await read() });
183   }
184   for (const [name, data] of out) files.push({ name, data });
185   const zip = await writeZip(files);
186   return { file: new File([zip], `${stem}.read-along.epub`, { type: 'application/epub+zip' }), located, of };
187 }
188 
189 /* --------------------------------------------------------------------- pages */
190 
191 /** The text nodes that readBook() reads, in its order: leaf blocks, without furigana. */
192 function textNodes(doc) {
193   const body = doc.querySelector('body') ?? doc.documentElement;
194   const out = [];
195   for (const b of leafBlocks(body)) {
196     const walker = doc.createTreeWalker(b, SHOW_TEXT);
197     for (let n = walker.nextNode(); n; n = walker.nextNode()) {
198       if (!n.parentElement?.closest('rt, rp')) out.push(n);
199     }
200   }
201   return out;
202 }
203 
204 /** Replaces the text node with its pieces: plain text, and a <span id> for each find. */
205 function wrap({ node, wraps }) {
206   const doc = node.ownerDocument, parent = node.parentNode, data = node.data;
207   let at = 0;
208   for (const w of wraps) {
209     if (w.from > at) parent.insertBefore(doc.createTextNode(data.slice(at, w.from)), node);
210     const span = doc.createElementNS(XHTML, 'span');
211     span.setAttribute('id', w.id);
212     span.textContent = data.slice(w.from, w.to);
213     parent.insertBefore(span, node);
214     at = w.to;
215   }
216   if (at < data.length) parent.insertBefore(doc.createTextNode(data.slice(at)), node);
217   parent.removeChild(node);
218 }
219 
220 function serialize(doc) {
221   const s = new XMLSerializer().serializeToString(doc);
222   return s.startsWith('<?xml') ? s : `<?xml version="1.0" encoding="utf-8"?>\n${s}`;
223 }
224 
225 /** The stretch of one audio file that a span is read in. */
226 function clipOf(span, parts) {
227   let base = 0, part = 0;
228   while (part < parts.length - 1 && span.start >= base + parts[part].duration) base += parts[part++].duration;
229   const begin = Math.max(0, span.start - base);
230   // A line that runs past the end of its file stops there.
231   const end = Math.max(begin + 0.001, Math.min(span.end - base, parts[part].duration || Infinity));
232   return { part, begin, end };
233 }
234 
235 export function clock(seconds) {
236   const ms = Math.round(seconds * 1000);
237   const p = (n, w = 2) => String(n).padStart(w, '0');
238   return `${Math.floor(ms / 3600000)}:${p(Math.floor(ms / 60000) % 60)}:${p(Math.floor(ms / 1000) % 60)}.${p(ms % 1000, 3)}`;
239 }
240 
241 const dirOf = (p) => (p.includes('/') ? p.slice(0, p.lastIndexOf('/')) : '');
242 const attr = (s) => s.replace(/&/g, '&amp;').replace(/"/g, '&quot;').replace(/</g, '&lt;');
243 
244 /** The way from a directory to a file, both as zip paths. */
245 export function relative(fromDir, to) {
246   const a = fromDir ? fromDir.split('/') : [], b = to.split('/');
247   while (a.length && b.length > 1 && a[0] === b[0]) { a.shift(); b.shift(); }
248   return encodeURI('../'.repeat(a.length) + b.join('/'));
249 }
250 
251 /* ----------------------------------------------------------- EPUB 2 to EPUB 3 */
252 
253 /**
254  * Media overlays are EPUB 3. Most epubs in the wild are EPUB 2, and an EPUB 3
255  * package must have two things that those do not: a modification date and a
256  * navigation document. The old table of contents (NCX) is kept too.
257  */
258 async function toEpub3(opf, { entries, items, order, pages, base, out, add, text }) {
259   const pkg = opf.documentElement;
260   const metadata = opf.querySelector('metadata');
261   if (!(parseFloat(pkg.getAttribute('version')) >= 3)) {
262     pkg.setAttribute('version', '3.0');
263     // EPUB 2 attributes that EPUB 3 does not have.
264     for (const el of metadata.querySelectorAll('*')) {
265       for (const a of [...el.attributes]) if (a.name.startsWith('opf:')) el.removeAttribute(a.name);
266     }
267     const dates = [...metadata.children].filter((el) => el.localName === 'date');
268     dates.slice(1).forEach((el) => el.remove());
269     // EPUB 3 wants to be told which pages hold these.
270     for (const pg of pages) {
271       const has = { svg: 'svg', mathml: 'math', scripted: 'script' };
272       const found = Object.keys(has).filter((k) => pg.doc.getElementsByTagName(has[k]).length);
273       const item = items.get(pg.path);
274       const was = (item.getAttribute('properties') ?? '').split(/\s+/).filter(Boolean);
275       const now = [...new Set([...was, ...found])];
276       if (now.length) item.setAttribute('properties', now.join(' '));
277     }
278   }
279   if (![...metadata.querySelectorAll('meta')].some((m) => m.getAttribute('property') === 'dcterms:modified')) {
280     add(metadata, 'meta', { property: 'dcterms:modified' }, new Date().toISOString().replace(/\.\d+Z$/, 'Z'));
281   }
282   if ([...items.values()].some((i) => (i.getAttribute('properties') ?? '').split(/\s+/).includes('nav'))) return;
283 
284   // The contents: from the NCX when there is one, else one line for each page.
285   let contents = [];
286   const ncxPath = [...items].find(([, i]) => i.getAttribute('media-type') === 'application/x-dtbncx+xml')?.[0];
287   if (ncxPath && entries.has(ncxPath)) {
288     const ncx = new DOMParser().parseFromString(await text(ncxPath), 'application/xml');
289     contents = [...ncx.querySelectorAll('navPoint')].map((p) => ({
290       label: p.querySelector('navLabel > text')?.textContent.trim(),
291       path: resolve(dirOf(ncxPath), decodeURIComponent(p.querySelector('content')?.getAttribute('src') ?? '')),
292     })).filter((c) => c.label && c.path);
293   }
294   if (!contents.length) contents = order.map((path, n) => ({ label: `Part ${n + 1}`, path }));
295 
296   const navPath = `${DIR}/nav.xhtml`;
297   const dir = base ? `${base}/${DIR}` : DIR;
298   const esc = (s) => s.replace(/&/g, '&amp;').replace(/</g, '&lt;');
299   out.set(`${dir}/nav.xhtml`, new TextEncoder().encode(
300     '<?xml version="1.0" encoding="utf-8"?>\n' +
301     `<html xmlns="${XHTML}" xmlns:epub="http://www.idpf.org/2007/ops"><head><title>Contents</title></head><body>\n` +
302     '<nav epub:type="toc"><h1>Contents</h1><ol>\n' +
303     contents.map((c) => {
304       const [file, fragment] = c.path.split('#');
305       return `<li><a href="${attr(relative(dir, file))}${fragment ? `#${attr(fragment)}` : ''}">${esc(c.label)}</a></li>`;
306     }).join('\n') +
307     '\n</ol></nav>\n</body></html>\n'));
308   add(opf.querySelector('manifest'), 'item',
309     { id: 'subread-nav', href: navPath, 'media-type': 'application/xhtml+xml', properties: 'nav' });
310 }
311 
312 /* ----------------------------------------------------------------------- zip */
313 
314 const CRC = (() => {
315   const t = new Uint32Array(256);
316   for (let n = 0; n < 256; n++) {
317     let c = n;
318     for (let k = 0; k < 8; k++) c = c & 1 ? 0xedb88320 ^ (c >>> 1) : c >>> 1;
319     t[n] = c >>> 0;
320   }
321   return t;
322 })();
323 
324 function crc32(bytes, crc = 0) {
325   let c = ~crc;
326   for (let i = 0; i < bytes.length; i++) c = CRC[(c ^ bytes[i]) & 0xff] ^ (c >>> 8);
327   return ~c >>> 0;
328 }
329 
330 /**
331  * A zip with nothing compressed: the audio, which is nearly all of it, is
332  * compressed already. A Blob of parts, so a book of audio is never copied
333  * into memory.
334  *
335  * @param files [{ name, data: Uint8Array | Blob }]
336  */
337 export async function writeZip(files) {
338   const parts = [], directory = [];
339   let offset = 0;
340   for (const { name, data } of files) {
341     const size = data instanceof Blob ? data.size : data.length;
342     let crc = 0;
343     if (data instanceof Blob) {
344       const reader = data.stream().getReader();
345       for (let chunk = await reader.read(); !chunk.done; chunk = await reader.read()) crc = crc32(chunk.value, crc);
346     } else crc = crc32(data);
347 
348     const nameBytes = new TextEncoder().encode(name);
349     const header = (signature, extra) => {
350       const b = new DataView(new ArrayBuffer(extra ? 46 : 30));
351       let o = 0;
352       const u16 = (v) => { b.setUint16(o, v, true); o += 2; };
353       const u32 = (v) => { b.setUint32(o, v, true); o += 4; };
354       u32(signature);
355       if (extra) u16(20);                    // version made by
356       u16(20); u16(0x0800); u16(0);          // version needed; UTF-8 names; stored
357       u16(0); u16(0x21);                     // 1980-01-01 00:00
358       u32(crc); u32(size); u32(size);
359       u16(nameBytes.length); u16(0);
360       if (extra) { u16(0); u16(0); u16(0); u32(0); u32(offset); }
361       return new Uint8Array(b.buffer);
362     };
363     directory.push(header(0x02014b50, true), nameBytes);
364     parts.push(header(0x04034b50, false), nameBytes, data);
365     offset += 30 + nameBytes.length + size;
366     if (offset > 0xffffffff) throw new Error('The book and its audio are over 4 GB, which is more than an epub can hold.');
367   }
368   const directorySize = directory.reduce((n, p) => n + p.length, 0);
369   const end = new DataView(new ArrayBuffer(22));
370   end.setUint32(0, 0x06054b50, true);
371   end.setUint16(8, files.length, true);
372   end.setUint16(10, files.length, true);
373   end.setUint32(12, directorySize, true);
374   end.setUint32(16, offset, true);
375   return new Blob([...parts, ...directory, new Uint8Array(end.buffer)]);
376 }