Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


frontend/engine/book.js (12649 bytes)

1 /* A book as a flat list of paragraphs in reading order - read in the browser,
2  * never uploaded.
3  *
4  * epub (a zip of XHTML), Kindle (MOBI and KF8, through foliate-js), fb2 (XML),
5  * Aozora Bunko (Shift_JIS text with ruby markup, usually zipped) and plain
6  * text. No zip library: the platform can inflate (DecompressionStream), and
7  * the little that is left of the format is a directory at the end of the file.
8  */
9 
10 const BLOCKS = 'p, li, blockquote, h1, h2, h3, h4, h5, h6';
11 
12 export async function readBook(file) {
13   const name = file.name.toLowerCase();
14   if (/\.(mobi|azw|azw3|prc)$/.test(name)) return mobi(file);
15   const bytes = new Uint8Array(await file.arrayBuffer());
16   if (name.endsWith('.epub')) return epub(bytes);
17   if (name.endsWith('.fb2.zip') || name.endsWith('.fbz')) return fb2(decode(await firstEntry(bytes, /\.fb2$/i)));
18   if (name.endsWith('.zip')) return aozora(decode(await firstEntry(bytes, /\.txt$/i)));
19   if (name.endsWith('.fb2')) return fb2(decode(bytes));
20   return plain(decode(bytes));
21 }
22 
23 /* ----------------------------------------------------------------------- zip */
24 
25 const u16 = (b, o) => b[o] | (b[o + 1] << 8);
26 const u32 = (b, o) => (b[o] | (b[o + 1] << 8) | (b[o + 2] << 16) | (b[o + 3] << 24)) >>> 0;
27 
28 /** name -> () => Promise<Uint8Array>, in directory order. */
29 export function zipEntries(bytes) {
30   let eocd = -1;
31   for (let i = bytes.length - 22; i >= Math.max(0, bytes.length - 65557); i--) {
32     if (u32(bytes, i) === 0x06054b50) { eocd = i; break; }
33   }
34   if (eocd < 0) throw new Error('This file is not a readable zip archive.');
35   const entries = new Map();
36   let o = u32(bytes, eocd + 16);
37   for (let n = u16(bytes, eocd + 10); n > 0 && u32(bytes, o) === 0x02014b50; n--) {
38     const method = u16(bytes, o + 10), size = u32(bytes, o + 20);
39     const nameLen = u16(bytes, o + 28), extraLen = u16(bytes, o + 30), commentLen = u16(bytes, o + 32);
40     const local = u32(bytes, o + 42);
41     const name = new TextDecoder().decode(bytes.subarray(o + 46, o + 46 + nameLen));
42     if (!name.endsWith('/')) {
43       entries.set(name, async () => {
44         const start = local + 30 + u16(bytes, local + 26) + u16(bytes, local + 28);
45         const raw = bytes.subarray(start, start + size);
46         if (method === 0) return raw;
47         if (method !== 8) throw new Error(`Unsupported zip compression in ${name}.`);
48         const stream = new Blob([raw]).stream().pipeThrough(new DecompressionStream('deflate-raw'));
49         return new Uint8Array(await new Response(stream).arrayBuffer());
50       });
51     }
52     o += 46 + nameLen + extraLen + commentLen;
53   }
54   return entries;
55 }
56 
57 async function firstEntry(bytes, pattern) {
58   for (const [name, read] of zipEntries(bytes)) if (pattern.test(name)) return read();
59   throw new Error('Nothing readable was found inside that archive.');
60 }
61 
62 /** UTF-8 unless it plainly is not; then whatever the file declares, or Shift_JIS. */
63 export function decode(bytes) {
64   try {
65     return new TextDecoder('utf-8', { fatal: true }).decode(bytes).replace(/^/, '');
66   } catch { /* not UTF-8 */ }
67   const head = new TextDecoder('latin1').decode(bytes.subarray(0, 200));
68   const declared = /encoding=["']([\w-]+)["']/i.exec(head)?.[1];
69   for (const label of [declared, 'shift_jis', 'windows-1251']) {
70     if (!label) continue;
71     try { return new TextDecoder(label).decode(bytes); } catch { /* unknown label */ }
72   }
73   return new TextDecoder().decode(bytes);
74 }
75 
76 /* ---------------------------------------------------------------------- epub */
77 
78 export function resolve(base, href) {
79   const parts = base ? base.split('/') : [];
80   for (const seg of href.split('/')) {
81     if (seg === '' || seg === '.') continue;
82     if (seg === '..') parts.pop(); else parts.push(seg);
83   }
84   return parts.join('/');
85 }
86 
87 /**
88  * An epub page is XHTML, and has to be parsed as XML first: to an HTML parser
89  * a self-closing <title/> never closes, and swallows the whole book as its text.
90  * Plenty of epubs are not well-formed, though, so HTML is the fallback.
91  */
92 /**
93  * The blocks of text of a page, in reading order. A quote holding paragraphs
94  * would yield its text twice, so only the leaves count. The sort is for DOM
95  * implementations that give the matches of a selector list out of order.
96  */
97 export function leafBlocks(body) {
98   const blocks = [...body.querySelectorAll(BLOCKS)].filter((b) => !b.querySelector(BLOCKS))
99     .sort((x, y) => (x.compareDocumentPosition(y) & 4 ? -1 : 1));
100   return blocks.length ? blocks : [body];
101 }
102 
103 export function page(source) {
104   const xml = new DOMParser().parseFromString(source, 'application/xhtml+xml');
105   if (!xml.querySelector('parsererror')) return xml;
106   return new DOMParser().parseFromString(source, 'text/html');
107 }
108 
109 async function epub(bytes) {
110   const entries = zipEntries(bytes);
111   const text = async (name) => decode(await entries.get(name)());
112   const xml = (s) => new DOMParser().parseFromString(s, 'application/xml');
113   const isPage = (n) => /\.(xhtml|html|htm)$/i.test(n);
114 
115   let order = null;
116   try {
117     const opfPath = xml(await text('META-INF/container.xml')).querySelector('rootfile').getAttribute('full-path');
118     const opf = xml(await text(opfPath));
119     const base = opfPath.includes('/') ? opfPath.slice(0, opfPath.lastIndexOf('/')) : '';
120     const hrefs = new Map([...opf.querySelectorAll('manifest > item')].map((i) => [i.getAttribute('id'), i.getAttribute('href')]));
121     order = [...opf.querySelectorAll('spine > itemref')]
122       // linear="no" is auxiliary content outside the reading order.
123       .filter((r) => r.getAttribute('linear') !== 'no')
124       .map((r) => hrefs.get(r.getAttribute('idref')))
125       .filter(Boolean)
126       .map((h) => resolve(base, decodeURIComponent(h.split('#')[0])))
127       .filter((p) => entries.has(p));
128   } catch { /* malformed package file: fall through */ }
129   if (!order?.length) order = [...entries.keys()].filter(isPage).sort();
130 
131   const out = [];
132   let cover = null;
133   for (const path of order) {
134     const doc = page(await text(path));
135     const body = doc.querySelector('body') ?? doc.documentElement;
136     // Furigana would otherwise be read twice: once as kanji, once as kana.
137     doc.querySelectorAll('rt, rp').forEach((n) => n.remove());
138     for (const b of leafBlocks(body)) {
139       const t = b.textContent.replace(/\s+/g, ' ').trim();
140       if (t) out.push(t);
141     }
142   }
143 
144   // The largest image is almost always the cover, and this works on the
145   // malformed epubs that converted books usually are.
146   const images = [...entries.keys()].filter((n) => /\.(jpe?g|png|webp)$/i.test(n));
147   if (images.length) {
148     const sized = await Promise.all(images.map(async (n) => [n, (await entries.get(n)()).length]));
149     const [name, size] = sized.sort((a, b) => b[1] - a[1])[0];
150     if (size >= 1024) cover = { name, bytes: await entries.get(name)() };
151   }
152   return { paragraphs: out, cover };
153 }
154 
155 /* -------------------------------------------------------------------- kindle */
156 
157 /** Inflate with the platform: foliate-js asks for this only for embedded fonts. */
158 async function unzlib(bytes) {
159   const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream('deflate'));
160   return new Uint8Array(await new Response(stream).arrayBuffer());
161 }
162 
163 /**
164  * MOBI and KF8 (azw3), the Kindle formats. foliate-js takes the file apart
165  * and gives each section back as a document; the blocks are read the same
166  * way as an epub page.
167  */
168 async function mobi(file) {
169   const { MOBI } = await import('../vendor/foliate/mobi.js');
170   const book = await new MOBI({ unzlib }).open(file);
171   const out = [];
172   for (const section of book.sections) {
173     let doc = await section.createDocument();
174     if (doc.querySelector('parsererror')) {
175       // A KF8 section that is not well-formed XHTML: read it as HTML instead.
176       doc = page(await (await fetch(await section.load())).text());
177     }
178     const body = doc.querySelector('body') ?? doc.documentElement;
179     doc.querySelectorAll('rt, rp').forEach((n) => n.remove());
180     for (const b of leafBlocks(body)) {
181       const t = b.textContent.replace(/\s+/g, ' ').trim();
182       if (t) out.push(t);
183     }
184   }
185   let cover = null;
186   const blob = await book.getCover?.();
187   if (blob?.size >= 1024) cover = { name: 'cover', bytes: new Uint8Array(await blob.arrayBuffer()) };
188   return { paragraphs: out, cover };
189 }
190 
191 /* ------------------------------------------------------------ other formats */
192 
193 function fb2(source) {
194   const doc = new DOMParser().parseFromString(source, 'application/xml');
195   const bodies = [...doc.getElementsByTagName('body')]
196     // Footnotes live in a second body; nobody narrates them.
197     .filter((b) => (b.getAttribute('name') || '') !== 'notes');
198   const out = [];
199   for (const body of bodies) {
200     for (const p of body.querySelectorAll('p, v, subtitle, text-author')) {
201       const t = p.textContent.replace(/\s+/g, ' ').trim();
202       if (t) out.push(t);
203     }
204   }
205   return { paragraphs: out, cover: null };
206 }
207 
208 /** Aozora Bunko: strip the legend, the colophon, ruby readings and editorial notes. */
209 export function aozora(raw) {
210   let lines = raw.replace(/\r\n/g, '\n').split('\n');
211   const rules = lines.flatMap((l, i) => (l.startsWith('-----') ? [i] : []));
212   if (rules.length >= 2) lines = lines.slice(rules[1] + 1);
213   const colophon = lines.findIndex((l) => l.startsWith('底本:'));
214   if (colophon >= 0) lines = lines.slice(0, colophon);
215   const paragraphs = lines
216     .map((l) => l.replace(/[#[^]]*]/g, '').replace(/《[^》]*》/g, '').replaceAll('|', '').trim())
217     .filter(Boolean);
218   return { paragraphs, cover: null };
219 }
220 
221 function plain(text) {
222   if (text.includes('《') && text.includes('底本:')) return aozora(text);
223   return { paragraphs: text.replace(/\r\n/g, '\n').split('\n').map((l) => l.trim()).filter(Boolean), cover: null };
224 }
225 
226 /* ------------------------------------------------------------------ language */
227 
228 // The commonest little words of each language. Crude, and enough: a book is
229 // tens of thousands of words, and the visitor can always overrule the guess.
230 const STOPWORDS = {
231   en: 'the and of to in that was his he it with as for had you not be her on at by which have from this',
232   es: 'de la que el en y a los del se las por un para con no una su al lo como más pero sus le ya',
233   pt: 'de a o que e do da em um para é com não uma os no se na por mais as dos como mas foi ao ele',
234   fr: 'de la le et les des en un du une que est pour qui dans a par plus pas au sur ne se ce il sont',
235   de: 'der die und in den von zu das mit sich des auf für ist im dem nicht ein eine als auch es an er',
236   it: 'di e il la che in a per un è del non le si con i da una dei più al come ma lo gli nel alla',
237   nl: 'de van het een en in is dat op te zijn voor met die niet aan er om ook als dan maar bij hij',
238   sv: 'och i att det som en på är av för med till den har de inte om ett han men var jag sig från',
239   fi: 'ja on ei että oli hän se en mutta niin kun kuin joka hänen ole sen olen mitä minä jos nyt vain',
240   pl: 'i w nie na z że się do to jest jak a o po ale co tak za od go był przez już tylko jego',
241   tr: 'bir ve bu da de için ile ne o ben gibi çok daha ama kadar sonra en var mi diye her olan',
242   ru: 'и в не на я что он с как а то это все она так его но да ты к у же вы за бы по только',
243   uk: 'і в не на я що він з як а то це все вона так його але та ти до у ж ви за би по тільки',
244 };
245 
246 /** Best guess at a Whisper language code for `paragraphs`, or null. */
247 export function detectLanguage(paragraphs) {
248   const sample = paragraphs.join(' ').slice(0, 60000);
249   const count = (re) => (sample.match(re) || []).length;
250   const kana = count(/[぀-ヿ]/g), han = count(/[一-鿿]/g), hangul = count(/[가-힯]/g);
251   if (kana > 50) return 'ja';
252   if (hangul > 50) return 'ko';
253   if (han > 200) return 'zh';
254   if (count(/[Ͱ-Ͽ]/g) > 200) return 'el';
255   if (count(/[֐-׿]/g) > 200) return 'he';
256   if (count(/[؀-ۿ]/g) > 200) return 'ar';
257   if (count(/[฀-๿]/g) > 200) return 'th';
258 
259   const words = sample.toLowerCase().match(/\p{L}+/gu) || [];
260   if (words.length < 50) return null;
261   const tally = new Map();
262   for (const w of words) tally.set(w, (tally.get(w) || 0) + 1);
263   let best = null, bestScore = 0;
264   for (const [code, list] of Object.entries(STOPWORDS)) {
265     const score = list.split(' ').reduce((n, w) => n + (tally.get(w) || 0), 0);
266     if (score > bestScore) { best = code; bestScore = score; }
267   }
268   // Fewer than one word in twelve a stopword of the winner: not one of ours.
269   return bestScore / words.length > 0.08 ? best : null;
270 }