frontend/engine/book.js (12649 bytes)
1 /* A book as a flat list of paragraphs in reading order - read in the browser, 2 * never uploaded. 3 * 4 * epub (a zip of XHTML), Kindle (MOBI and KF8, through foliate-js), fb2 (XML), 5 * Aozora Bunko (Shift_JIS text with ruby markup, usually zipped) and plain 6 * text. No zip library: the platform can inflate (DecompressionStream), and 7 * the little that is left of the format is a directory at the end of the file. 8 */ 9 10 const BLOCKS = 'p, li, blockquote, h1, h2, h3, h4, h5, h6'; 11 12 export async function readBook(file) { 13 const name = file.name.toLowerCase(); 14 if (/\.(mobi|azw|azw3|prc)$/.test(name)) return mobi(file); 15 const bytes = new Uint8Array(await file.arrayBuffer()); 16 if (name.endsWith('.epub')) return epub(bytes); 17 if (name.endsWith('.fb2.zip') || name.endsWith('.fbz')) return fb2(decode(await firstEntry(bytes, /\.fb2$/i))); 18 if (name.endsWith('.zip')) return aozora(decode(await firstEntry(bytes, /\.txt$/i))); 19 if (name.endsWith('.fb2')) return fb2(decode(bytes)); 20 return plain(decode(bytes)); 21 } 22 23 /* ----------------------------------------------------------------------- zip */ 24 25 const u16 = (b, o) => b[o] | (b[o + 1] << 8); 26 const u32 = (b, o) => (b[o] | (b[o + 1] << 8) | (b[o + 2] << 16) | (b[o + 3] << 24)) >>> 0; 27 28 /** name -> () => Promise<Uint8Array>, in directory order. */ 29 export function zipEntries(bytes) { 30 let eocd = -1; 31 for (let i = bytes.length - 22; i >= Math.max(0, bytes.length - 65557); i--) { 32 if (u32(bytes, i) === 0x06054b50) { eocd = i; break; } 33 } 34 if (eocd < 0) throw new Error('This file is not a readable zip archive.'); 35 const entries = new Map(); 36 let o = u32(bytes, eocd + 16); 37 for (let n = u16(bytes, eocd + 10); n > 0 && u32(bytes, o) === 0x02014b50; n--) { 38 const method = u16(bytes, o + 10), size = u32(bytes, o + 20); 39 const nameLen = u16(bytes, o + 28), extraLen = u16(bytes, o + 30), commentLen = u16(bytes, o + 32); 40 const local = u32(bytes, o + 42); 41 const name = new TextDecoder().decode(bytes.subarray(o + 46, o + 46 + nameLen)); 42 if (!name.endsWith('/')) { 43 entries.set(name, async () => { 44 const start = local + 30 + u16(bytes, local + 26) + u16(bytes, local + 28); 45 const raw = bytes.subarray(start, start + size); 46 if (method === 0) return raw; 47 if (method !== 8) throw new Error(`Unsupported zip compression in ${name}.`); 48 const stream = new Blob([raw]).stream().pipeThrough(new DecompressionStream('deflate-raw')); 49 return new Uint8Array(await new Response(stream).arrayBuffer()); 50 }); 51 } 52 o += 46 + nameLen + extraLen + commentLen; 53 } 54 return entries; 55 } 56 57 async function firstEntry(bytes, pattern) { 58 for (const [name, read] of zipEntries(bytes)) if (pattern.test(name)) return read(); 59 throw new Error('Nothing readable was found inside that archive.'); 60 } 61 62 /** UTF-8 unless it plainly is not; then whatever the file declares, or Shift_JIS. */ 63 export function decode(bytes) { 64 try { 65 return new TextDecoder('utf-8', { fatal: true }).decode(bytes).replace(/^/, ''); 66 } catch { /* not UTF-8 */ } 67 const head = new TextDecoder('latin1').decode(bytes.subarray(0, 200)); 68 const declared = /encoding=["']([\w-]+)["']/i.exec(head)?.[1]; 69 for (const label of [declared, 'shift_jis', 'windows-1251']) { 70 if (!label) continue; 71 try { return new TextDecoder(label).decode(bytes); } catch { /* unknown label */ } 72 } 73 return new TextDecoder().decode(bytes); 74 } 75 76 /* ---------------------------------------------------------------------- epub */ 77 78 export function resolve(base, href) { 79 const parts = base ? base.split('/') : []; 80 for (const seg of href.split('/')) { 81 if (seg === '' || seg === '.') continue; 82 if (seg === '..') parts.pop(); else parts.push(seg); 83 } 84 return parts.join('/'); 85 } 86 87 /** 88 * An epub page is XHTML, and has to be parsed as XML first: to an HTML parser 89 * a self-closing <title/> never closes, and swallows the whole book as its text. 90 * Plenty of epubs are not well-formed, though, so HTML is the fallback. 91 */ 92 /** 93 * The blocks of text of a page, in reading order. A quote holding paragraphs 94 * would yield its text twice, so only the leaves count. The sort is for DOM 95 * implementations that give the matches of a selector list out of order. 96 */ 97 export function leafBlocks(body) { 98 const blocks = [...body.querySelectorAll(BLOCKS)].filter((b) => !b.querySelector(BLOCKS)) 99 .sort((x, y) => (x.compareDocumentPosition(y) & 4 ? -1 : 1)); 100 return blocks.length ? blocks : [body]; 101 } 102 103 export function page(source) { 104 const xml = new DOMParser().parseFromString(source, 'application/xhtml+xml'); 105 if (!xml.querySelector('parsererror')) return xml; 106 return new DOMParser().parseFromString(source, 'text/html'); 107 } 108 109 async function epub(bytes) { 110 const entries = zipEntries(bytes); 111 const text = async (name) => decode(await entries.get(name)()); 112 const xml = (s) => new DOMParser().parseFromString(s, 'application/xml'); 113 const isPage = (n) => /\.(xhtml|html|htm)$/i.test(n); 114 115 let order = null; 116 try { 117 const opfPath = xml(await text('META-INF/container.xml')).querySelector('rootfile').getAttribute('full-path'); 118 const opf = xml(await text(opfPath)); 119 const base = opfPath.includes('/') ? opfPath.slice(0, opfPath.lastIndexOf('/')) : ''; 120 const hrefs = new Map([...opf.querySelectorAll('manifest > item')].map((i) => [i.getAttribute('id'), i.getAttribute('href')])); 121 order = [...opf.querySelectorAll('spine > itemref')] 122 // linear="no" is auxiliary content outside the reading order. 123 .filter((r) => r.getAttribute('linear') !== 'no') 124 .map((r) => hrefs.get(r.getAttribute('idref'))) 125 .filter(Boolean) 126 .map((h) => resolve(base, decodeURIComponent(h.split('#')[0]))) 127 .filter((p) => entries.has(p)); 128 } catch { /* malformed package file: fall through */ } 129 if (!order?.length) order = [...entries.keys()].filter(isPage).sort(); 130 131 const out = []; 132 let cover = null; 133 for (const path of order) { 134 const doc = page(await text(path)); 135 const body = doc.querySelector('body') ?? doc.documentElement; 136 // Furigana would otherwise be read twice: once as kanji, once as kana. 137 doc.querySelectorAll('rt, rp').forEach((n) => n.remove()); 138 for (const b of leafBlocks(body)) { 139 const t = b.textContent.replace(/\s+/g, ' ').trim(); 140 if (t) out.push(t); 141 } 142 } 143 144 // The largest image is almost always the cover, and this works on the 145 // malformed epubs that converted books usually are. 146 const images = [...entries.keys()].filter((n) => /\.(jpe?g|png|webp)$/i.test(n)); 147 if (images.length) { 148 const sized = await Promise.all(images.map(async (n) => [n, (await entries.get(n)()).length])); 149 const [name, size] = sized.sort((a, b) => b[1] - a[1])[0]; 150 if (size >= 1024) cover = { name, bytes: await entries.get(name)() }; 151 } 152 return { paragraphs: out, cover }; 153 } 154 155 /* -------------------------------------------------------------------- kindle */ 156 157 /** Inflate with the platform: foliate-js asks for this only for embedded fonts. */ 158 async function unzlib(bytes) { 159 const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream('deflate')); 160 return new Uint8Array(await new Response(stream).arrayBuffer()); 161 } 162 163 /** 164 * MOBI and KF8 (azw3), the Kindle formats. foliate-js takes the file apart 165 * and gives each section back as a document; the blocks are read the same 166 * way as an epub page. 167 */ 168 async function mobi(file) { 169 const { MOBI } = await import('../vendor/foliate/mobi.js'); 170 const book = await new MOBI({ unzlib }).open(file); 171 const out = []; 172 for (const section of book.sections) { 173 let doc = await section.createDocument(); 174 if (doc.querySelector('parsererror')) { 175 // A KF8 section that is not well-formed XHTML: read it as HTML instead. 176 doc = page(await (await fetch(await section.load())).text()); 177 } 178 const body = doc.querySelector('body') ?? doc.documentElement; 179 doc.querySelectorAll('rt, rp').forEach((n) => n.remove()); 180 for (const b of leafBlocks(body)) { 181 const t = b.textContent.replace(/\s+/g, ' ').trim(); 182 if (t) out.push(t); 183 } 184 } 185 let cover = null; 186 const blob = await book.getCover?.(); 187 if (blob?.size >= 1024) cover = { name: 'cover', bytes: new Uint8Array(await blob.arrayBuffer()) }; 188 return { paragraphs: out, cover }; 189 } 190 191 /* ------------------------------------------------------------ other formats */ 192 193 function fb2(source) { 194 const doc = new DOMParser().parseFromString(source, 'application/xml'); 195 const bodies = [...doc.getElementsByTagName('body')] 196 // Footnotes live in a second body; nobody narrates them. 197 .filter((b) => (b.getAttribute('name') || '') !== 'notes'); 198 const out = []; 199 for (const body of bodies) { 200 for (const p of body.querySelectorAll('p, v, subtitle, text-author')) { 201 const t = p.textContent.replace(/\s+/g, ' ').trim(); 202 if (t) out.push(t); 203 } 204 } 205 return { paragraphs: out, cover: null }; 206 } 207 208 /** Aozora Bunko: strip the legend, the colophon, ruby readings and editorial notes. */ 209 export function aozora(raw) { 210 let lines = raw.replace(/\r\n/g, '\n').split('\n'); 211 const rules = lines.flatMap((l, i) => (l.startsWith('-----') ? [i] : [])); 212 if (rules.length >= 2) lines = lines.slice(rules[1] + 1); 213 const colophon = lines.findIndex((l) => l.startsWith('底本:')); 214 if (colophon >= 0) lines = lines.slice(0, colophon); 215 const paragraphs = lines 216 .map((l) => l.replace(/[#[^]]*]/g, '').replace(/《[^》]*》/g, '').replaceAll('|', '').trim()) 217 .filter(Boolean); 218 return { paragraphs, cover: null }; 219 } 220 221 function plain(text) { 222 if (text.includes('《') && text.includes('底本:')) return aozora(text); 223 return { paragraphs: text.replace(/\r\n/g, '\n').split('\n').map((l) => l.trim()).filter(Boolean), cover: null }; 224 } 225 226 /* ------------------------------------------------------------------ language */ 227 228 // The commonest little words of each language. Crude, and enough: a book is 229 // tens of thousands of words, and the visitor can always overrule the guess. 230 const STOPWORDS = { 231 en: 'the and of to in that was his he it with as for had you not be her on at by which have from this', 232 es: 'de la que el en y a los del se las por un para con no una su al lo como más pero sus le ya', 233 pt: 'de a o que e do da em um para é com não uma os no se na por mais as dos como mas foi ao ele', 234 fr: 'de la le et les des en un du une que est pour qui dans a par plus pas au sur ne se ce il sont', 235 de: 'der die und in den von zu das mit sich des auf für ist im dem nicht ein eine als auch es an er', 236 it: 'di e il la che in a per un è del non le si con i da una dei più al come ma lo gli nel alla', 237 nl: 'de van het een en in is dat op te zijn voor met die niet aan er om ook als dan maar bij hij', 238 sv: 'och i att det som en på är av för med till den har de inte om ett han men var jag sig från', 239 fi: 'ja on ei että oli hän se en mutta niin kun kuin joka hänen ole sen olen mitä minä jos nyt vain', 240 pl: 'i w nie na z że się do to jest jak a o po ale co tak za od go był przez już tylko jego', 241 tr: 'bir ve bu da de için ile ne o ben gibi çok daha ama kadar sonra en var mi diye her olan', 242 ru: 'и в не на я что он с как а то это все она так его но да ты к у же вы за бы по только', 243 uk: 'і в не на я що він з як а то це все вона так його але та ти до у ж ви за би по тільки', 244 }; 245 246 /** Best guess at a Whisper language code for `paragraphs`, or null. */ 247 export function detectLanguage(paragraphs) { 248 const sample = paragraphs.join(' ').slice(0, 60000); 249 const count = (re) => (sample.match(re) || []).length; 250 const kana = count(/[-ヿ]/g), han = count(/[一-鿿]/g), hangul = count(/[가-]/g); 251 if (kana > 50) return 'ja'; 252 if (hangul > 50) return 'ko'; 253 if (han > 200) return 'zh'; 254 if (count(/[Ͱ-Ͽ]/g) > 200) return 'el'; 255 if (count(/[-]/g) > 200) return 'he'; 256 if (count(/[-ۿ]/g) > 200) return 'ar'; 257 if (count(/[-]/g) > 200) return 'th'; 258 259 const words = sample.toLowerCase().match(/\p{L}+/gu) || []; 260 if (words.length < 50) return null; 261 const tally = new Map(); 262 for (const w of words) tally.set(w, (tally.get(w) || 0) + 1); 263 let best = null, bestScore = 0; 264 for (const [code, list] of Object.entries(STOPWORDS)) { 265 const score = list.split(' ').reduce((n, w) => n + (tally.get(w) || 0), 0); 266 if (score > bestScore) { best = code; bestScore = score; } 267 } 268 // Fewer than one word in twelve a stopword of the winner: not one of ours. 269 return bestScore / words.length > 0.08 ? best : null; 270 }