Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


commit f9ace057f8b888cd5112fd3845a062f6ff8f2c7e
equwal <truex@equwal.com>
2026-09-21 17:27:45 -0700

Read Kindle books in the browser

A mobi, azw, azw3 or prc book went to POST /api/convert, the one step of
a conversion in the tab that still needed the server. book.js now reads
those formats with foliate-js (MIT, one file, fetched and pinned by
tools/fetch_vendor.py) and takes the paragraphs from each section the
same way as from an epub page. The cover comes from the file too.

The test reads Project Gutenberg #1952 as a Kindle file and as an epub
and checks that the same story comes out, in order.

 README.md                                   |   9 ++++--
 frontend/app.js                             |  13 +-------
 frontend/engine/book.js                     |  45 +++++++++++++++++++++++++---
 tests/engine/fixtures/yellow-wallpaper.epub | Bin 0 -> 87911 bytes
 tests/engine/fixtures/yellow-wallpaper.mobi | Bin 0 -> 121373 bytes
 tests/engine/mobi.test.mjs                  |  45 ++++++++++++++++++++++++++++
 tools/fetch_vendor.py                       |   4 +++
 7 files changed, 97 insertions(+), 19 deletions(-)
diff --git a/README.md b/README.md
index 845dcf8..e53249a 100644
--- a/README.md
+++ b/README.md
@@ -62,10 +62,13 @@ A single file, or **a folder of per-chapter files**: drop all 44 mp3s and they
 are merged into one chaptered file, in natural order (`9.mp3` before `10.mp3`).
 Files can arrive one drop at a time; the upload starts once both halves are in.
 
-**Book** — `epub`, `txt`, `srt`, `vtt`, `ass` directly; `fb2`, `fb2.zip`,
-`mobi`, `azw3`, `azw` and `prc` are converted on upload.
+**Book** — `epub`, `txt`, `srt`, `vtt`, `ass`, `fb2`, `fb2.zip`, and the
+Kindle formats `mobi`, `azw3`, `azw` and `prc`. In the browser every one of
+them is read in the tab (`frontend/engine/book.js`; the Kindle formats through
+[foliate-js](https://github.com/johnfactotum/foliate-js), fetched by
+`tools/fetch_vendor.py`). A server job converts the same formats on upload.
 
-Conversion delegates rather than parsing ebook formats by hand
+Conversion on the server delegates rather than parsing ebook formats by hand
 (`backend/convert.py`):
 
 | From | How |
diff --git a/frontend/app.js b/frontend/app.js
index 93e127d..31af395 100644
--- a/frontend/app.js
+++ b/frontend/app.js
@@ -248,7 +248,6 @@ document.querySelectorAll('.slot-x').forEach((btn) => {
 
 /* Nothing is uploaded. Once both halves are here the book is read in this tab,
    its language guessed, and the visitor asked to confirm before hours of work. */
-const NEEDS_CONVERTING = /\.(mobi|azw|azw3|prc)$/i;
 const natural = new Intl.Collator(undefined, { numeric: true, sensitivity: 'base' });
 
 async function prepare() {
@@ -262,17 +261,7 @@ async function prepare() {
 
   try {
     const { readBook, detectLanguage } = await import('/engine/book.js');
-    let book = staged.text;
-    if (NEEDS_CONVERTING.test(book.name)) {
-      // The one thing the browser cannot do itself: Kindle formats need a real
-      // parser. A book is small; the audio still never leaves this machine.
-      el.uptext.textContent = 'Converting the book to epub…';
-      const form = new FormData();
-      form.append('file', book, book.name);
-      const res = await fetch('/api/convert', { method: 'POST', body: form, credentials: 'same-origin' });
-      if (!res.ok) throw new Error((await res.json().catch(() => ({}))).detail || 'Could not convert that book.');
-      book = new File([await res.blob()], res.headers.get('X-Filename') || 'book.epub');
-    }
+    const book = staged.text;
     const parsed = await readBook(book);
     if (!parsed.paragraphs.length) {
       throw new Error(`No text could be read from ${book.name}. A scanned, image-only book cannot be aligned.`);
diff --git a/frontend/engine/book.js b/frontend/engine/book.js
index 9581b9f..07cad8b 100644
--- a/frontend/engine/book.js
+++ b/frontend/engine/book.js
@@ -1,16 +1,17 @@
 /* A book as a flat list of paragraphs in reading order - read in the browser,
  * never uploaded.
  *
- * epub (a zip of XHTML), fb2 (XML), Aozora Bunko (Shift_JIS text with ruby
- * markup, usually zipped) and plain text. No zip library: the platform can
- * inflate (DecompressionStream), and the little that is left of the format is
- * a directory at the end of the file.
+ * epub (a zip of XHTML), Kindle (MOBI and KF8, through foliate-js), fb2 (XML),
+ * Aozora Bunko (Shift_JIS text with ruby markup, usually zipped) and plain
+ * text. No zip library: the platform can inflate (DecompressionStream), and
+ * the little that is left of the format is a directory at the end of the file.
  */
 
 const BLOCKS = 'p, li, blockquote, h1, h2, h3, h4, h5, h6';
 
 export async function readBook(file) {
   const name = file.name.toLowerCase();
+  if (/\.(mobi|azw|azw3|prc)$/.test(name)) return mobi(file);
   const bytes = new Uint8Array(await file.arrayBuffer());
   if (name.endsWith('.epub')) return epub(bytes);
   if (name.endsWith('.fb2.zip') || name.endsWith('.fbz')) return fb2(decode(await firstEntry(bytes, /\.fb2$/i)));
@@ -151,6 +152,42 @@ async function epub(bytes) {
   return { paragraphs: out, cover };
 }
 
+/* -------------------------------------------------------------------- kindle */
+
+/** Inflate with the platform: foliate-js asks for this only for embedded fonts. */
+async function unzlib(bytes) {
+  const stream = new Blob([bytes]).stream().pipeThrough(new DecompressionStream('deflate'));
+  return new Uint8Array(await new Response(stream).arrayBuffer());
+}
+
+/**
+ * MOBI and KF8 (azw3), the Kindle formats. foliate-js takes the file apart
+ * and gives each section back as a document; the blocks are read the same
+ * way as an epub page.
+ */
+async function mobi(file) {
+  const { MOBI } = await import('../vendor/foliate/mobi.js');
+  const book = await new MOBI({ unzlib }).open(file);
+  const out = [];
+  for (const section of book.sections) {
+    let doc = await section.createDocument();
+    if (doc.querySelector('parsererror')) {
+      // A KF8 section that is not well-formed XHTML: read it as HTML instead.
+      doc = page(await (await fetch(await section.load())).text());
+    }
+    const body = doc.querySelector('body') ?? doc.documentElement;
+    doc.querySelectorAll('rt, rp').forEach((n) => n.remove());
+    for (const b of leafBlocks(body)) {
+      const t = b.textContent.replace(/\s+/g, ' ').trim();
+      if (t) out.push(t);
+    }
+  }
+  let cover = null;
+  const blob = await book.getCover?.();
+  if (blob?.size >= 1024) cover = { name: 'cover', bytes: new Uint8Array(await blob.arrayBuffer()) };
+  return { paragraphs: out, cover };
+}
+
 /* ------------------------------------------------------------ other formats */
 
 function fb2(source) {
diff --git a/tests/engine/fixtures/yellow-wallpaper.epub b/tests/engine/fixtures/yellow-wallpaper.epub
new file mode 100644
index 0000000..04b6952
Binary files /dev/null and b/tests/engine/fixtures/yellow-wallpaper.epub differ
diff --git a/tests/engine/fixtures/yellow-wallpaper.mobi b/tests/engine/fixtures/yellow-wallpaper.mobi
new file mode 100644
index 0000000..18a2861
Binary files /dev/null and b/tests/engine/fixtures/yellow-wallpaper.mobi differ
diff --git a/tests/engine/mobi.test.mjs b/tests/engine/mobi.test.mjs
new file mode 100644
index 0000000..dcf4b34
--- /dev/null
+++ b/tests/engine/mobi.test.mjs
@@ -0,0 +1,45 @@
+/* Kindle books are read in the browser, by foliate-js. The same public-domain
+ * book (The Yellow Wallpaper, Project Gutenberg #1952) as a MOBI/KF8 file and
+ * as an epub must give the same text.
+ *
+ *   node --test tests/engine
+ */
+import test from 'node:test';
+import assert from 'node:assert/strict';
+import { readFileSync } from 'node:fs';
+import { fileURLToPath } from 'node:url';
+import path from 'node:path';
+import { JSDOM } from 'jsdom';
+
+// foliate-js unescapes HTML entities through a <textarea>, so it wants a document too.
+const { window } = new JSDOM('');
+Object.assign(globalThis, { DOMParser: window.DOMParser, XMLSerializer: window.XMLSerializer, document: window.document });
+
+const { readBook } = await import('../../frontend/engine/book.js');
+
+const here = path.dirname(fileURLToPath(import.meta.url));
+const fixture = (name) => new File([readFileSync(path.join(here, 'fixtures', name))], name);
+const bare = (s) => s.replace(/\s+/g, '').toLowerCase();
+
+test('a Kindle file gives the paragraphs of the book, as the epub does', async () => {
+  const mobi = await readBook(fixture('yellow-wallpaper.mobi'));
+  const epub = await readBook(fixture('yellow-wallpaper.epub'));
+  assert.ok(epub.paragraphs.length > 100, `epub has ${epub.paragraphs.length} paragraphs`);
+  assert.ok(mobi.paragraphs.length > 100, `mobi has ${mobi.paragraphs.length} paragraphs`);
+
+  // The story itself, not the licence, which each edition words its own way.
+  const story = (p) => p.filter((t) => t.length > 40);
+  const inMobi = new Set(story(mobi.paragraphs).map(bare));
+  const shared = story(epub.paragraphs).filter((t) => inMobi.has(bare(t))).length;
+  const share = shared / story(epub.paragraphs).length;
+  assert.ok(share > 0.9, `${(share * 100).toFixed(0)}% of the epub's paragraphs are in the mobi`);
+
+  // In reading order: the first line of the story before its last.
+  const first = mobi.paragraphs.findIndex((t) => /very seldom that mere ordinary people/.test(t));
+  const last = mobi.paragraphs.findIndex((t) => /had to creep over him every time/.test(t));
+  assert.ok(first >= 0 && last > first, `first line at ${first}, last at ${last}`);
+});
+
+test('a Kindle file that is not one is refused, not read as text', async () => {
+  await assert.rejects(readBook(new File([new Uint8Array(100)], 'broken.azw3')));
+});
diff --git a/tools/fetch_vendor.py b/tools/fetch_vendor.py
index f18e3f7..ae477cc 100644
--- a/tools/fetch_vendor.py
+++ b/tools/fetch_vendor.py
@@ -45,6 +45,10 @@ PACKAGES = [
         for name in ("ort.webgpu.bundle.min.mjs", "ort-wasm-simd-threaded.asyncify.mjs",
                      "ort-wasm-simd-threaded.asyncify.wasm")
     }),
+    # Reads Kindle books (MOBI and KF8) in the browser. One file, no imports.
+    ("foliate-js", "1.0.1", {
+        "package/mobi.js": "foliate/mobi.js",
+    }),
 ]