frontend/engine/epub.js (17282 bytes)
1 /* A read-along book: the epub, with the narration inside it and each line of 2 * text tied to its stretch of audio (EPUB 3 Media Overlays). Thorium, 3 * Storyteller and other EPUB 3 readers play it and highlight the text. 4 * 5 * The cues say what was read and when, but not where in the book's pages the 6 * words are. That is found again by text: a cue's words, without white space, 7 * are looked for in the book's words, from the end of the cue before. Each 8 * find is wrapped in <span id>. A cue that crosses markup (<em>, ruby, the end 9 * of a paragraph) gets several spans, and its time is divided among them by 10 * their length. 11 * 12 * Made in the browser, like everything else: nothing is uploaded. 13 */ 14 import { zipEntries, leafBlocks, page, resolve, decode } from './book.js'; 15 import { UNMATCHED } from './align.js'; 16 17 const XHTML = 'http://www.w3.org/1999/xhtml'; 18 const OPF = 'http://www.idpf.org/2007/opf'; 19 const ACTIVE = '-epub-media-overlay-active'; 20 const DIR = 'subread'; // what this adds, beside the package file 21 const SHOW_TEXT = 4; 22 23 // A jump this far ahead is believed only from a cue this long: a short line 24 // ("Yes.") is found by chance somewhere in any book. 25 const FAR = 4000; 26 const LONG_ENOUGH = 12; 27 28 /** Audio an EPUB 3 reader must play, by codec: file extension and media type. */ 29 const AUDIO = { mp3: ['mp3', 'audio/mpeg'], aac: ['m4a', 'audio/mp4'] }; 30 31 export class NotAnEpub extends Error {} 32 33 /** 34 * @param book the epub, a File 35 * @param cues [{ text, start, end }], seconds on the clock of all the audio 36 * @param parts [{ file, duration, codec }], in playback order 37 * @returns {{ file: File, located: number, of: number }} 38 */ 39 export async function syncedEpub({ book, cues, parts, stem }) { 40 for (const p of parts) { 41 if (!AUDIO[p.codec]) throw new Error(`A read-along book needs MP3 or AAC (m4a, m4b) audio; this is ${p.codec}.`); 42 } 43 const entries = zipEntries(new Uint8Array(await book.arrayBuffer())); 44 const text = async (name) => decode(await entries.get(name)()); 45 const xml = (s) => new DOMParser().parseFromString(s, 'application/xml'); 46 47 if (!entries.has('META-INF/container.xml')) throw new NotAnEpub('This epub has no package file.'); 48 const opfPath = xml(await text('META-INF/container.xml')).querySelector('rootfile')?.getAttribute('full-path'); 49 if (!opfPath || !entries.has(opfPath)) throw new NotAnEpub('This epub has no package file.'); 50 const opf = xml(await text(opfPath)); 51 if (opf.querySelector('parsererror')) throw new NotAnEpub('The package file of this epub cannot be read.'); 52 const base = opfPath.includes('/') ? opfPath.slice(0, opfPath.lastIndexOf('/')) : ''; 53 const inBase = (p) => (base ? `${base}/${p}` : p); 54 55 const items = new Map(); // zip path -> manifest item 56 for (const item of opf.querySelectorAll('manifest > item')) { 57 items.set(resolve(base, decodeURIComponent((item.getAttribute('href') ?? '').split('#')[0])), item); 58 } 59 const byId = new Map([...items].map(([p, item]) => [item.getAttribute('id'), p])); 60 const order = [...opf.querySelectorAll('spine > itemref')] 61 .filter((r) => r.getAttribute('linear') !== 'no') 62 .map((r) => byId.get(r.getAttribute('idref'))) 63 .filter((p) => p && entries.has(p)); 64 if (!order.length) throw new NotAnEpub('This epub has no reading order.'); 65 66 // The book's words, without white space, and where each one is. 67 const pages = []; 68 const runs = []; 69 const words = []; 70 let g = 0; 71 for (const path of order) { 72 const doc = page(await text(path)); 73 const pg = { path, doc, spans: [] }; 74 pages.push(pg); 75 for (const node of textNodes(doc)) { 76 const offsets = []; 77 for (let i = 0; i < node.data.length; i++) if (/\S/.test(node.data[i])) offsets.push(i); 78 if (!offsets.length) continue; 79 runs.push({ page: pg, node, g0: g, offsets, wraps: [] }); 80 words.push(offsets.map((i) => node.data[i]).join('')); 81 g += offsets.length; 82 } 83 } 84 const all = words.join(''); 85 86 // Where each cue is. 87 let cursor = 0, r = 0, located = 0, of = 0; 88 cues.forEach((cue, index) => { 89 if (cue.text.startsWith(UNMATCHED) || !(cue.end > cue.start)) return; 90 const want = cue.text.replace(/\s+/g, ''); 91 if (!want) return; 92 of++; 93 const at = all.indexOf(want, cursor); 94 if (at < 0 || (at - cursor > FAR && want.length < LONG_ENOUGH)) return; 95 located++; 96 cursor = at + want.length; 97 98 while (runs[r].g0 + runs[r].offsets.length <= at) r++; 99 const pieces = []; 100 for (let k = r; k < runs.length && runs[k].g0 < cursor; k++) { 101 const run = runs[k]; 102 const s = Math.max(at, run.g0) - run.g0, e = Math.min(cursor, run.g0 + run.offsets.length) - run.g0; 103 pieces.push({ run, from: run.offsets[s], to: run.offsets[e - 1] + 1, chars: e - s }); 104 } 105 let t = cue.start; 106 pieces.forEach((piece, k) => { 107 const id = `subread-${index}${k ? `-${k}` : ''}`; 108 const until = k === pieces.length - 1 ? cue.end : t + (cue.end - cue.start) * piece.chars / want.length; 109 piece.run.wraps.push({ from: piece.from, to: piece.to, id }); 110 piece.run.page.spans.push({ id, start: t, end: until }); 111 t = until; 112 }); 113 }); 114 if (!located) throw new Error('None of the subtitles could be found in the pages of this epub.'); 115 116 for (const run of runs) if (run.wraps.length) wrap(run); 117 118 // The new files. 119 const out = new Map(); // zip path -> Uint8Array | Blob 120 const manifest = opf.querySelector('manifest'); 121 const metadata = opf.querySelector('metadata'); 122 const add = (parent, name, attributes, content) => { 123 const el = opf.createElementNS(OPF, name); 124 for (const [k, v] of Object.entries(attributes)) el.setAttribute(k, v); 125 if (content != null) el.textContent = content; 126 parent.appendChild(el); 127 return el; 128 }; 129 const encoder = new TextEncoder(); 130 131 const audio = parts.map((p, i) => { 132 const [extension, type] = AUDIO[p.codec]; 133 const path = `${DIR}/audio-${i + 1}.${extension}`; 134 out.set(inBase(path), p.file); 135 add(manifest, 'item', { id: `subread-audio-${i + 1}`, href: path, 'media-type': type }); 136 return path; 137 }); 138 out.set(inBase(`${DIR}/overlay.css`), encoder.encode(`.${ACTIVE} { background: #ffe08a; color: #000; }\n`)); 139 add(manifest, 'item', { id: 'subread-css', href: `${DIR}/overlay.css`, 'media-type': 'text/css' }); 140 141 let total = 0; 142 pages.forEach((pg, n) => { 143 if (!pg.spans.length) return; 144 const smilPath = `${DIR}/page-${n + 1}.smil`; 145 const smilDir = inBase(DIR); 146 const lines = pg.spans.map((s, k) => { 147 const clip = clipOf(s, parts); 148 total += clip.end - clip.begin; 149 return `<par id="par-${k + 1}"><text src="${attr(relative(smilDir, pg.path))}#${s.id}"/>` + 150 `<audio src="${attr(relative(smilDir, inBase(audio[clip.part])))}" clipBegin="${clip.begin.toFixed(3)}s" clipEnd="${clip.end.toFixed(3)}s"/></par>`; 151 }); 152 out.set(inBase(smilPath), encoder.encode( 153 '<?xml version="1.0" encoding="utf-8"?>\n' + 154 '<smil xmlns="http://www.w3.org/ns/SMIL" xmlns:epub="http://www.idpf.org/2007/ops" version="3.0">\n<body>\n' + 155 `<seq id="seq" epub:textref="${attr(relative(smilDir, pg.path))}" epub:type="bodymatter">\n${lines.join('\n')}\n</seq>\n</body>\n</smil>\n`)); 156 157 const id = `subread-smil-${n + 1}`; 158 add(manifest, 'item', { id, href: smilPath, 'media-type': 'application/smil+xml' }); 159 items.get(pg.path).setAttribute('media-overlay', id); 160 const seconds = pg.spans.reduce((sum, s) => { const c = clipOf(s, parts); return sum + c.end - c.begin; }, 0); 161 add(metadata, 'meta', { property: 'media:duration', refines: `#${id}` }, clock(seconds)); 162 163 const head = pg.doc.querySelector('head'); 164 if (head) { 165 const link = pg.doc.createElementNS(XHTML, 'link'); 166 link.setAttribute('rel', 'stylesheet'); 167 link.setAttribute('type', 'text/css'); 168 link.setAttribute('href', relative(dirOf(pg.path), inBase(`${DIR}/overlay.css`))); 169 head.appendChild(link); 170 } 171 out.set(pg.path, encoder.encode(serialize(pg.doc))); 172 }); 173 add(metadata, 'meta', { property: 'media:duration' }, clock(total)); 174 add(metadata, 'meta', { property: 'media:active-class' }, ACTIVE); 175 176 await toEpub3(opf, { entries, items, order, pages, base, out, add, text }); 177 out.set(opfPath, encoder.encode(serialize(opf))); 178 179 // mimetype first and not compressed: that is how a reader knows an epub. 180 const files = [{ name: 'mimetype', data: encoder.encode('application/epub+zip') }]; 181 for (const [name, read] of entries) { 182 if (name !== 'mimetype' && !out.has(name)) files.push({ name, data: await read() }); 183 } 184 for (const [name, data] of out) files.push({ name, data }); 185 const zip = await writeZip(files); 186 return { file: new File([zip], `${stem}.read-along.epub`, { type: 'application/epub+zip' }), located, of }; 187 } 188 189 /* --------------------------------------------------------------------- pages */ 190 191 /** The text nodes that readBook() reads, in its order: leaf blocks, without furigana. */ 192 function textNodes(doc) { 193 const body = doc.querySelector('body') ?? doc.documentElement; 194 const out = []; 195 for (const b of leafBlocks(body)) { 196 const walker = doc.createTreeWalker(b, SHOW_TEXT); 197 for (let n = walker.nextNode(); n; n = walker.nextNode()) { 198 if (!n.parentElement?.closest('rt, rp')) out.push(n); 199 } 200 } 201 return out; 202 } 203 204 /** Replaces the text node with its pieces: plain text, and a <span id> for each find. */ 205 function wrap({ node, wraps }) { 206 const doc = node.ownerDocument, parent = node.parentNode, data = node.data; 207 let at = 0; 208 for (const w of wraps) { 209 if (w.from > at) parent.insertBefore(doc.createTextNode(data.slice(at, w.from)), node); 210 const span = doc.createElementNS(XHTML, 'span'); 211 span.setAttribute('id', w.id); 212 span.textContent = data.slice(w.from, w.to); 213 parent.insertBefore(span, node); 214 at = w.to; 215 } 216 if (at < data.length) parent.insertBefore(doc.createTextNode(data.slice(at)), node); 217 parent.removeChild(node); 218 } 219 220 function serialize(doc) { 221 const s = new XMLSerializer().serializeToString(doc); 222 return s.startsWith('<?xml') ? s : `<?xml version="1.0" encoding="utf-8"?>\n${s}`; 223 } 224 225 /** The stretch of one audio file that a span is read in. */ 226 function clipOf(span, parts) { 227 let base = 0, part = 0; 228 while (part < parts.length - 1 && span.start >= base + parts[part].duration) base += parts[part++].duration; 229 const begin = Math.max(0, span.start - base); 230 // A line that runs past the end of its file stops there. 231 const end = Math.max(begin + 0.001, Math.min(span.end - base, parts[part].duration || Infinity)); 232 return { part, begin, end }; 233 } 234 235 export function clock(seconds) { 236 const ms = Math.round(seconds * 1000); 237 const p = (n, w = 2) => String(n).padStart(w, '0'); 238 return `${Math.floor(ms / 3600000)}:${p(Math.floor(ms / 60000) % 60)}:${p(Math.floor(ms / 1000) % 60)}.${p(ms % 1000, 3)}`; 239 } 240 241 const dirOf = (p) => (p.includes('/') ? p.slice(0, p.lastIndexOf('/')) : ''); 242 const attr = (s) => s.replace(/&/g, '&').replace(/"/g, '"').replace(/</g, '<'); 243 244 /** The way from a directory to a file, both as zip paths. */ 245 export function relative(fromDir, to) { 246 const a = fromDir ? fromDir.split('/') : [], b = to.split('/'); 247 while (a.length && b.length > 1 && a[0] === b[0]) { a.shift(); b.shift(); } 248 return encodeURI('../'.repeat(a.length) + b.join('/')); 249 } 250 251 /* ----------------------------------------------------------- EPUB 2 to EPUB 3 */ 252 253 /** 254 * Media overlays are EPUB 3. Most epubs in the wild are EPUB 2, and an EPUB 3 255 * package must have two things that those do not: a modification date and a 256 * navigation document. The old table of contents (NCX) is kept too. 257 */ 258 async function toEpub3(opf, { entries, items, order, pages, base, out, add, text }) { 259 const pkg = opf.documentElement; 260 const metadata = opf.querySelector('metadata'); 261 if (!(parseFloat(pkg.getAttribute('version')) >= 3)) { 262 pkg.setAttribute('version', '3.0'); 263 // EPUB 2 attributes that EPUB 3 does not have. 264 for (const el of metadata.querySelectorAll('*')) { 265 for (const a of [...el.attributes]) if (a.name.startsWith('opf:')) el.removeAttribute(a.name); 266 } 267 const dates = [...metadata.children].filter((el) => el.localName === 'date'); 268 dates.slice(1).forEach((el) => el.remove()); 269 // EPUB 3 wants to be told which pages hold these. 270 for (const pg of pages) { 271 const has = { svg: 'svg', mathml: 'math', scripted: 'script' }; 272 const found = Object.keys(has).filter((k) => pg.doc.getElementsByTagName(has[k]).length); 273 const item = items.get(pg.path); 274 const was = (item.getAttribute('properties') ?? '').split(/\s+/).filter(Boolean); 275 const now = [...new Set([...was, ...found])]; 276 if (now.length) item.setAttribute('properties', now.join(' ')); 277 } 278 } 279 if (![...metadata.querySelectorAll('meta')].some((m) => m.getAttribute('property') === 'dcterms:modified')) { 280 add(metadata, 'meta', { property: 'dcterms:modified' }, new Date().toISOString().replace(/\.\d+Z$/, 'Z')); 281 } 282 if ([...items.values()].some((i) => (i.getAttribute('properties') ?? '').split(/\s+/).includes('nav'))) return; 283 284 // The contents: from the NCX when there is one, else one line for each page. 285 let contents = []; 286 const ncxPath = [...items].find(([, i]) => i.getAttribute('media-type') === 'application/x-dtbncx+xml')?.[0]; 287 if (ncxPath && entries.has(ncxPath)) { 288 const ncx = new DOMParser().parseFromString(await text(ncxPath), 'application/xml'); 289 contents = [...ncx.querySelectorAll('navPoint')].map((p) => ({ 290 label: p.querySelector('navLabel > text')?.textContent.trim(), 291 path: resolve(dirOf(ncxPath), decodeURIComponent(p.querySelector('content')?.getAttribute('src') ?? '')), 292 })).filter((c) => c.label && c.path); 293 } 294 if (!contents.length) contents = order.map((path, n) => ({ label: `Part ${n + 1}`, path })); 295 296 const navPath = `${DIR}/nav.xhtml`; 297 const dir = base ? `${base}/${DIR}` : DIR; 298 const esc = (s) => s.replace(/&/g, '&').replace(/</g, '<'); 299 out.set(`${dir}/nav.xhtml`, new TextEncoder().encode( 300 '<?xml version="1.0" encoding="utf-8"?>\n' + 301 `<html xmlns="${XHTML}" xmlns:epub="http://www.idpf.org/2007/ops"><head><title>Contents</title></head><body>\n` + 302 '<nav epub:type="toc"><h1>Contents</h1><ol>\n' + 303 contents.map((c) => { 304 const [file, fragment] = c.path.split('#'); 305 return `<li><a href="${attr(relative(dir, file))}${fragment ? `#${attr(fragment)}` : ''}">${esc(c.label)}</a></li>`; 306 }).join('\n') + 307 '\n</ol></nav>\n</body></html>\n')); 308 add(opf.querySelector('manifest'), 'item', 309 { id: 'subread-nav', href: navPath, 'media-type': 'application/xhtml+xml', properties: 'nav' }); 310 } 311 312 /* ----------------------------------------------------------------------- zip */ 313 314 const CRC = (() => { 315 const t = new Uint32Array(256); 316 for (let n = 0; n < 256; n++) { 317 let c = n; 318 for (let k = 0; k < 8; k++) c = c & 1 ? 0xedb88320 ^ (c >>> 1) : c >>> 1; 319 t[n] = c >>> 0; 320 } 321 return t; 322 })(); 323 324 function crc32(bytes, crc = 0) { 325 let c = ~crc; 326 for (let i = 0; i < bytes.length; i++) c = CRC[(c ^ bytes[i]) & 0xff] ^ (c >>> 8); 327 return ~c >>> 0; 328 } 329 330 /** 331 * A zip with nothing compressed: the audio, which is nearly all of it, is 332 * compressed already. A Blob of parts, so a book of audio is never copied 333 * into memory. 334 * 335 * @param files [{ name, data: Uint8Array | Blob }] 336 */ 337 export async function writeZip(files) { 338 const parts = [], directory = []; 339 let offset = 0; 340 for (const { name, data } of files) { 341 const size = data instanceof Blob ? data.size : data.length; 342 let crc = 0; 343 if (data instanceof Blob) { 344 const reader = data.stream().getReader(); 345 for (let chunk = await reader.read(); !chunk.done; chunk = await reader.read()) crc = crc32(chunk.value, crc); 346 } else crc = crc32(data); 347 348 const nameBytes = new TextEncoder().encode(name); 349 const header = (signature, extra) => { 350 const b = new DataView(new ArrayBuffer(extra ? 46 : 30)); 351 let o = 0; 352 const u16 = (v) => { b.setUint16(o, v, true); o += 2; }; 353 const u32 = (v) => { b.setUint32(o, v, true); o += 4; }; 354 u32(signature); 355 if (extra) u16(20); // version made by 356 u16(20); u16(0x0800); u16(0); // version needed; UTF-8 names; stored 357 u16(0); u16(0x21); // 1980-01-01 00:00 358 u32(crc); u32(size); u32(size); 359 u16(nameBytes.length); u16(0); 360 if (extra) { u16(0); u16(0); u16(0); u32(0); u32(offset); } 361 return new Uint8Array(b.buffer); 362 }; 363 directory.push(header(0x02014b50, true), nameBytes); 364 parts.push(header(0x04034b50, false), nameBytes, data); 365 offset += 30 + nameBytes.length + size; 366 if (offset > 0xffffffff) throw new Error('The book and its audio are over 4 GB, which is more than an epub can hold.'); 367 } 368 const directorySize = directory.reduce((n, p) => n + p.length, 0); 369 const end = new DataView(new ArrayBuffer(22)); 370 end.setUint32(0, 0x06054b50, true); 371 end.setUint16(8, files.length, true); 372 end.setUint16(10, files.length, true); 373 end.setUint32(12, directorySize, true); 374 end.setUint32(16, offset, true); 375 return new Blob([...parts, ...directory, new Uint8Array(end.buffer)]); 376 }