commit f9ace057f8b888cd5112fd3845a062f6ff8f2c7e equwal <truex@equwal.com> 2026-09-21 17:27:45 -0700 Read Kindle books in the browser A mobi, azw, azw3 or prc book went to POST /api/convert, the one step of a conversion in the tab that still needed the server. book.js now reads those formats with foliate-js (MIT, one file, fetched and pinned by tools/fetch_vendor.py) and takes the paragraphs from each section the same way as from an epub page. The cover comes from the file too. The test reads Project Gutenberg #1952 as a Kindle file and as an epub and checks that the same story comes out, in order.
README.md | 9 ++++-- frontend/app.js | 13 +------- frontend/engine/book.js | 45 +++++++++++++++++++++++++--- tests/engine/fixtures/yellow-wallpaper.epub | Bin 0 -> 87911 bytes tests/engine/fixtures/yellow-wallpaper.mobi | Bin 0 -> 121373 bytes tests/engine/mobi.test.mjs | 45 ++++++++++++++++++++++++++++ tools/fetch_vendor.py | 4 +++ 7 files changed, 97 insertions(+), 19 deletions(-)
diff --git a/README.md b/README.md index 845dcf8..e53249a 100644 --- a/README.md +++ b/README.md @@ -62,10 +62,13 @@ A single file, or **a folder of per-chapter files**: drop all 44 mp3s and they are merged into one chaptered file, in natural order (`9.mp3` before `10.mp3`). Files can arrive one drop at a time; the upload starts once both halves are in. -**Book** — `epub`, `txt`, `srt`, `vtt`, `ass` directly; `fb2`, `fb2.zip`, -`mobi`, `azw3`, `azw` and `prc` are converted on upload. +**Book** — `epub`, `txt`, `srt`, `vtt`, `ass`, `fb2`, `fb2.zip`, and the +Kindle formats `mobi`, `azw3`, `azw` and `prc`. In the browser every one of +them is read in the tab (`frontend/engine/book.js`; the Kindle formats through +[foliate-js](https://github.com/johnfactotum/foliate-js), fetched by +`tools/fetch_vendor.py`). A server job converts the same formats on upload. -Conversion delegates rather than parsing ebook formats by hand +Conversion on the server delegates rather than parsing ebook formats by hand (`backend/convert.py`): | From | How | diff --git a/frontend/app.js b/frontend/app.js index 93e127d..31af395 100644 --- a/frontend/app.js +++ b/frontend/app.js @@ -248,7 +248,6 @@ document.querySelectorAll('.slot-x').forEach((btn) => { /* Nothing is uploaded. Once both halves are here the book is read in this tab, its language guessed, and the visitor asked to confirm before hours of work. */ -const NEEDS_CONVERTING = /\.(mobi|azw|azw3|prc)$/i; const natural = new Intl.Collator(undefined, { numeric: true, sensitivity: 'base' }); async function prepare() { @@ -262,17 +261,7 @@ async function prepare() { try { const { readBook, detectLanguage } = await import('/engine/book.js'); - let book = staged.text; - if (NEEDS_CONVERTING.test(book.name)) { - // The one thing the browser cannot do itself: Kindle formats need a real - // parser. A book is small; the audio still never leaves this machine. - el.uptext.textContent = 'Converting the book to epub…'; - const form = new FormData(); - form.append('file', book, book.name); - const res = await fetch('/api/convert', { method: 'POST', body: form, credentials: 'same-origin' }); - if (!res.ok) throw new Error((await res.json().catch(() => ({}))).detail || 'Could not convert that book.'); - book = new File([await res.blob()], res.headers.get('X-Filename') || 'book.epub'); - } + const book = staged.text; const parsed = await readBook(book); if (!parsed.paragraphs.length) { throw new Error(`No text could be read from ${book.name}. A scanned, image-only book cannot be aligned.`); diff --git a/frontend/engine/book.js b/frontend/engine/book.js index 9581b9f..07cad8b 100644 --- a/frontend/engine/book.js +++ b/frontend/engine/book.js @@ -1,16 +1,17 @@ /* A book as a flat list of paragraphs in reading order - read in the browser, * never uploaded. * - * epub (a zip of XHTML), fb2 (XML), Aozora Bunko (Shift_JIS text with ruby - * markup, usually zipped) and plain text. No zip library: the platform can - * inflate (DecompressionStream), and the little that is left of the format is - * a directory at the end of the file. + * epub (a zip of XHTML), Kindle (MOBI and KF8, through foliate-js), fb2 (XML), + * Aozora Bunko (Shift_JIS text with ruby markup, usually zipped) and plain + * text. No zip library: the platform can inflate (DecompressionStream), and + * the little that is left of the format is a directory at the end of the file. */ const BLOCKS = 'p, li, blockquote, h1, h2, h3, h4, h5, h6'; export async function readBook(file) { const name = file.name.toLowerCase(); + if (/\.(mobi|azw|azw3|prc)$/.test(name)) return mobi(file); const bytes = new Uint8Array(await file.arrayBuffer()); if (name.endsWith('.epub')) return epub(bytes); if (name.endsWith('.fb2.zip') || name.endsWith('.fbz')) return fb2(decode(await firstEntry(bytes, /\.fb2$/i))); @@ -151,6 +152,42 @@ async function epub(bytes) { return { paragraphs: out, cover }; } +/* -------------------------------------------------------------------- kindle */ + +/** Inflate with the platform: foliate-js asks for this only for embedded fonts. */ +async function unzlib(bytes) { + const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream('deflate')); + return new Uint8Array(await new Response(stream).arrayBuffer()); +} + +/** + * MOBI and KF8 (azw3), the Kindle formats. foliate-js takes the file apart + * and gives each section back as a document; the blocks are read the same + * way as an epub page. + */ +async function mobi(file) { + const { MOBI } = await import('../vendor/foliate/mobi.js'); + const book = await new MOBI({ unzlib }).open(file); + const out = []; + for (const section of book.sections) { + let doc = await section.createDocument(); + if (doc.querySelector('parsererror')) { + // A KF8 section that is not well-formed XHTML: read it as HTML instead. + doc = page(await (await fetch(await section.load())).text()); + } + const body = doc.querySelector('body') ?? doc.documentElement; + doc.querySelectorAll('rt, rp').forEach((n) => n.remove()); + for (const b of leafBlocks(body)) { + const t = b.textContent.replace(/\s+/g, ' ').trim(); + if (t) out.push(t); + } + } + let cover = null; + const blob = await book.getCover?.(); + if (blob?.size >= 1024) cover = { name: 'cover', bytes: new Uint8Array(await blob.arrayBuffer()) }; + return { paragraphs: out, cover }; +} + /* ------------------------------------------------------------ other formats */ function fb2(source) { diff --git a/tests/engine/fixtures/yellow-wallpaper.epub b/tests/engine/fixtures/yellow-wallpaper.epub new file mode 100644 index 0000000..04b6952 Binary files /dev/null and b/tests/engine/fixtures/yellow-wallpaper.epub differ diff --git a/tests/engine/fixtures/yellow-wallpaper.mobi b/tests/engine/fixtures/yellow-wallpaper.mobi new file mode 100644 index 0000000..18a2861 Binary files /dev/null and b/tests/engine/fixtures/yellow-wallpaper.mobi differ diff --git a/tests/engine/mobi.test.mjs b/tests/engine/mobi.test.mjs new file mode 100644 index 0000000..dcf4b34 --- /dev/null +++ b/tests/engine/mobi.test.mjs @@ -0,0 +1,45 @@ +/* Kindle books are read in the browser, by foliate-js. The same public-domain + * book (The Yellow Wallpaper, Project Gutenberg #1952) as a MOBI/KF8 file and + * as an epub must give the same text. + * + * node --test tests/engine + */ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import path from 'node:path'; +import { JSDOM } from 'jsdom'; + +// foliate-js unescapes HTML entities through a <textarea>, so it wants a document too. +const { window } = new JSDOM(''); +Object.assign(globalThis, { DOMParser: window.DOMParser, XMLSerializer: window.XMLSerializer, document: window.document }); + +const { readBook } = await import('../../frontend/engine/book.js'); + +const here = path.dirname(fileURLToPath(import.meta.url)); +const fixture = (name) => new File([readFileSync(path.join(here, 'fixtures', name))], name); +const bare = (s) => s.replace(/\s+/g, '').toLowerCase(); + +test('a Kindle file gives the paragraphs of the book, as the epub does', async () => { + const mobi = await readBook(fixture('yellow-wallpaper.mobi')); + const epub = await readBook(fixture('yellow-wallpaper.epub')); + assert.ok(epub.paragraphs.length > 100, `epub has ${epub.paragraphs.length} paragraphs`); + assert.ok(mobi.paragraphs.length > 100, `mobi has ${mobi.paragraphs.length} paragraphs`); + + // The story itself, not the licence, which each edition words its own way. + const story = (p) => p.filter((t) => t.length > 40); + const inMobi = new Set(story(mobi.paragraphs).map(bare)); + const shared = story(epub.paragraphs).filter((t) => inMobi.has(bare(t))).length; + const share = shared / story(epub.paragraphs).length; + assert.ok(share > 0.9, `${(share * 100).toFixed(0)}% of the epub's paragraphs are in the mobi`); + + // In reading order: the first line of the story before its last. + const first = mobi.paragraphs.findIndex((t) => /very seldom that mere ordinary people/.test(t)); + const last = mobi.paragraphs.findIndex((t) => /had to creep over him every time/.test(t)); + assert.ok(first >= 0 && last > first, `first line at ${first}, last at ${last}`); +}); + +test('a Kindle file that is not one is refused, not read as text', async () => { + await assert.rejects(readBook(new File([new Uint8Array(100)], 'broken.azw3'))); +}); diff --git a/tools/fetch_vendor.py b/tools/fetch_vendor.py index f18e3f7..ae477cc 100644 --- a/tools/fetch_vendor.py +++ b/tools/fetch_vendor.py @@ -45,6 +45,10 @@ PACKAGES = [ for name in ("ort.webgpu.bundle.min.mjs", "ort-wasm-simd-threaded.asyncify.mjs", "ort-wasm-simd-threaded.asyncify.wasm") }), + # Reads Kindle books (MOBI and KF8) in the browser. One file, no imports. + ("foliate-js", "1.0.1", { + "package/mobi.js": "foliate/mobi.js", + }), ]