Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


commit 7e1fe5c203ee7184fea1591ec27ac6271af522d2
equwal <truex@equwal.com>
2026-09-20 17:22:47 -0700

The whole conversion, in the browser

engine/media.js   ffmpeg.wasm: reads any audio format straight off the
                  visitor's disk (WORKERFS, no copy) two minutes at a time,
                  and muxes the videos
engine/asr.js     Whisper-tiny through transformers.js; WebGPU, else WASM
                  threads. Language is always explicit - left to itself the
                  model translates instead of transcribing
engine/book.js    epub / fb2 / Aozora / text in the browser, with a zip
                  reader built on DecompressionStream; language detection
engine/job.js     the job: resumable (IndexedDB) transcription, alignment in
                  a worker, videos on request. A still picture is encoded
                  for one minute and repeated by stream copy, so a video
                  costs seconds rather than an hour of WebAssembly x264

tools/fetch_vendor.py pins and checksums the three libraries and serves them
from our own origin; the page is cross-origin isolated so the runtime can
use threads. lab.html exercises each piece; verified on a real sample: a
clip from the middle of a novel finds its place in the whole epub, and both
videos come out with the right streams.

 .gitignore                      |   3 +
 backend/main.py                 |  22 ++++
 frontend/engine/align.worker.js |  11 ++
 frontend/engine/asr.js          |  71 ++++++++++++
 frontend/engine/book.js         | 224 ++++++++++++++++++++++++++++++++++++++
 frontend/engine/job.js          | 234 ++++++++++++++++++++++++++++++++++++++++
 frontend/engine/media.js        | 111 +++++++++++++++++++
 frontend/lab.html               |  93 ++++++++++++++++
 tools/fetch_vendor.py           |  88 +++++++++++++++
 9 files changed, 857 insertions(+)
diff --git a/.gitignore b/.gitignore
index 6c69c5d..354a1ba 100644
--- a/.gitignore
+++ b/.gitignore
@@ -24,3 +24,6 @@ Thumbs.db
 
 # Whole-book test input; local only
 tests/engine/golden-local/
+
+# Browser-side libraries; tools/fetch_vendor.py downloads them, pinned
+frontend/vendor/
diff --git a/backend/main.py b/backend/main.py
index 51a36b5..3fd7bc7 100644
--- a/backend/main.py
+++ b/backend/main.py
@@ -9,6 +9,7 @@ CORS config and no second server.
 from __future__ import annotations
 
 import logging
+import mimetypes
 from contextlib import asynccontextmanager
 
 from fastapi import FastAPI
@@ -29,6 +30,12 @@ log = logging.getLogger("subplz.web")
 
 FRONTEND = ROOT / "frontend"
 
+# Python takes these from the OS, and Windows gets both wrong. A module served
+# as text/plain is refused outright, and WebAssembly will not stream-compile.
+mimetypes.add_type("text/javascript", ".js")
+mimetypes.add_type("text/javascript", ".mjs")
+mimetypes.add_type("application/wasm", ".wasm")
+
 
 def _requeue_interrupted() -> None:
     """Recover jobs that were mid-flight when the server last stopped.
@@ -79,6 +86,21 @@ app = FastAPI(
 app.include_router(router)
 
 
+@app.middleware("http")
+async def cross_origin_isolation(request, call_next):
+    """Let the page use threads.
+
+    The speech model runs in the browser on WebAssembly threads, which need
+    SharedArrayBuffer, which browsers only hand to a cross-origin-isolated
+    page. "credentialless" rather than "require-corp": the model weights come
+    from a third-party host that sends CORS headers but not CORP ones.
+    """
+    response = await call_next(request)
+    response.headers["Cross-Origin-Opener-Policy"] = "same-origin"
+    response.headers["Cross-Origin-Embedder-Policy"] = "credentialless"
+    return response
+
+
 @app.get("/healthz")
 def healthz():
     return {
diff --git a/frontend/engine/align.worker.js b/frontend/engine/align.worker.js
new file mode 100644
index 0000000..fb2fd4f
--- /dev/null
+++ b/frontend/engine/align.worker.js
@@ -0,0 +1,11 @@
+/* Alignment off the main thread: a long book is a few seconds of solid work. */
+import { alignBook, language, writeSrt } from './align.js';
+
+self.onmessage = ({ data }) => {
+  try {
+    const result = alignBook(data.transcript, data.paragraphs, language(data.language));
+    self.postMessage({ ok: true, ...result, srt: writeSrt(result.cues) });
+  } catch (e) {
+    self.postMessage({ ok: false, error: String(e?.message || e) });
+  }
+};
diff --git a/frontend/engine/asr.js b/frontend/engine/asr.js
new file mode 100644
index 0000000..6a48b9d
--- /dev/null
+++ b/frontend/engine/asr.js
@@ -0,0 +1,71 @@
+/* Speech recognition in the browser: Whisper-tiny on the ONNX runtime, through
+ * transformers.js. WebGPU where the browser has it (many times faster than real
+ * time on an ordinary laptop), WebAssembly threads where it does not.
+ *
+ * The transcript only has to be good enough to line up against a book we
+ * already have the words of, which is why the smallest model will do.
+ */
+import { pipeline, env } from '/vendor/transformers/transformers.min.js';
+
+// Everything from our own origin except the model weights themselves.
+env.backends.onnx.wasm.wasmPaths = new URL('/vendor/transformers/', import.meta.url).href;
+env.allowLocalModels = false;
+
+const MODEL = 'onnx-community/whisper-tiny';
+
+export async function hasWebGpu() {
+  try { return !!(navigator.gpu && await navigator.gpu.requestAdapter()); } catch { return false; }
+}
+
+export class Recogniser {
+  #pipe;
+  device;
+
+  /** @param onProgress ({file, loaded, total}) while the ~40 MB of weights download (cached after). */
+  static async open(onProgress) {
+    const r = new Recogniser();
+    r.device = (await hasWebGpu()) ? 'webgpu' : 'wasm';
+    const options = (device) => ({
+      device,
+      // Full-precision encoder: quantising it is what ruins accuracy. The
+      // decoder tolerates 4-bit well and is where the time goes.
+      dtype: device === 'webgpu'
+        ? { encoder_model: 'fp32', decoder_model_merged: 'q4' }
+        : { encoder_model: 'fp32', decoder_model_merged: 'q8' },
+      progress_callback: (p) => { if (p.status === 'progress') onProgress?.(p); },
+    });
+    try {
+      r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options(r.device));
+    } catch (e) {
+      if (r.device !== 'webgpu') throw e;
+      // A GPU that is listed but cannot run the model: fall back rather than fail.
+      r.device = 'wasm';
+      r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options('wasm'));
+    }
+    return r;
+  }
+
+  /**
+   * @param pcm 16 kHz mono Float32Array
+   * @param language a Whisper code ("ja", "ru", ...), or null to let it decide
+   * @returns [{text, start, end}] with times offset by `offset` seconds
+   */
+  async transcribe(pcm, language, offset = 0) {
+    const out = await this.#pipe(pcm, {
+      return_timestamps: true,
+      chunk_length_s: 30,
+      stride_length_s: 5,
+      task: 'transcribe',
+      ...(language ? { language } : {}),
+    });
+    const segments = [];
+    for (const c of out.chunks ?? []) {
+      const text = (c.text ?? '').trim();
+      const [s, e] = c.timestamp ?? [];
+      if (!text || s == null) continue;
+      // The last segment of a buffer can come back without an end.
+      segments.push({ text, start: offset + s, end: offset + (e ?? pcm.length / 16000) });
+    }
+    return segments;
+  }
+}
diff --git a/frontend/engine/book.js b/frontend/engine/book.js
new file mode 100644
index 0000000..3387daf
--- /dev/null
+++ b/frontend/engine/book.js
@@ -0,0 +1,224 @@
+/* A book as a flat list of paragraphs in reading order - read in the browser,
+ * never uploaded.
+ *
+ * epub (a zip of XHTML), fb2 (XML), Aozora Bunko (Shift_JIS text with ruby
+ * markup, usually zipped) and plain text. No zip library: the platform can
+ * inflate (DecompressionStream), and the little that is left of the format is
+ * a directory at the end of the file.
+ */
+
+const BLOCKS = 'p, li, blockquote, h1, h2, h3, h4, h5, h6';
+
+export async function readBook(file) {
+  const name = file.name.toLowerCase();
+  const bytes = new Uint8Array(await file.arrayBuffer());
+  if (name.endsWith('.epub')) return epub(bytes);
+  if (name.endsWith('.fb2.zip') || name.endsWith('.fbz')) return fb2(decode(await firstEntry(bytes, /\.fb2$/i)));
+  if (name.endsWith('.zip')) return aozora(decode(await firstEntry(bytes, /\.txt$/i)));
+  if (name.endsWith('.fb2')) return fb2(decode(bytes));
+  return plain(decode(bytes));
+}
+
+/* ----------------------------------------------------------------------- zip */
+
+const u16 = (b, o) => b[o] | (b[o + 1] << 8);
+const u32 = (b, o) => (b[o] | (b[o + 1] << 8) | (b[o + 2] << 16) | (b[o + 3] << 24)) >>> 0;
+
+/** name -> () => Promise<Uint8Array>, in directory order. */
+export function zipEntries(bytes) {
+  let eocd = -1;
+  for (let i = bytes.length - 22; i >= Math.max(0, bytes.length - 65557); i--) {
+    if (u32(bytes, i) === 0x06054b50) { eocd = i; break; }
+  }
+  if (eocd < 0) throw new Error('This file is not a readable zip archive.');
+  const entries = new Map();
+  let o = u32(bytes, eocd + 16);
+  for (let n = u16(bytes, eocd + 10); n > 0 && u32(bytes, o) === 0x02014b50; n--) {
+    const method = u16(bytes, o + 10), size = u32(bytes, o + 20);
+    const nameLen = u16(bytes, o + 28), extraLen = u16(bytes, o + 30), commentLen = u16(bytes, o + 32);
+    const local = u32(bytes, o + 42);
+    const name = new TextDecoder().decode(bytes.subarray(o + 46, o + 46 + nameLen));
+    if (!name.endsWith('/')) {
+      entries.set(name, async () => {
+        const start = local + 30 + u16(bytes, local + 26) + u16(bytes, local + 28);
+        const raw = bytes.subarray(start, start + size);
+        if (method === 0) return raw;
+        if (method !== 8) throw new Error(`Unsupported zip compression in ${name}.`);
+        const stream = new Blob([raw]).stream().pipeThrough(new DecompressionStream('deflate-raw'));
+        return new Uint8Array(await new Response(stream).arrayBuffer());
+      });
+    }
+    o += 46 + nameLen + extraLen + commentLen;
+  }
+  return entries;
+}
+
+async function firstEntry(bytes, pattern) {
+  for (const [name, read] of zipEntries(bytes)) if (pattern.test(name)) return read();
+  throw new Error('Nothing readable was found inside that archive.');
+}
+
+/** UTF-8 unless it plainly is not; then whatever the file declares, or Shift_JIS. */
+function decode(bytes) {
+  try {
+    return new TextDecoder('utf-8', { fatal: true }).decode(bytes).replace(/^/, '');
+  } catch { /* not UTF-8 */ }
+  const head = new TextDecoder('latin1').decode(bytes.subarray(0, 200));
+  const declared = /encoding=["']([\w-]+)["']/i.exec(head)?.[1];
+  for (const label of [declared, 'shift_jis', 'windows-1251']) {
+    if (!label) continue;
+    try { return new TextDecoder(label).decode(bytes); } catch { /* unknown label */ }
+  }
+  return new TextDecoder().decode(bytes);
+}
+
+/* ---------------------------------------------------------------------- epub */
+
+function resolve(base, href) {
+  const parts = base ? base.split('/') : [];
+  for (const seg of href.split('/')) {
+    if (seg === '' || seg === '.') continue;
+    if (seg === '..') parts.pop(); else parts.push(seg);
+  }
+  return parts.join('/');
+}
+
+/**
+ * An epub page is XHTML, and has to be parsed as XML first: to an HTML parser
+ * a self-closing <title/> never closes, and swallows the whole book as its text.
+ * Plenty of epubs are not well-formed, though, so HTML is the fallback.
+ */
+function page(source) {
+  const xml = new DOMParser().parseFromString(source, 'application/xhtml+xml');
+  if (!xml.querySelector('parsererror')) return xml;
+  return new DOMParser().parseFromString(source, 'text/html');
+}
+
+async function epub(bytes) {
+  const entries = zipEntries(bytes);
+  const text = async (name) => decode(await entries.get(name)());
+  const xml = (s) => new DOMParser().parseFromString(s, 'application/xml');
+  const isPage = (n) => /\.(xhtml|html|htm)$/i.test(n);
+
+  let order = null;
+  try {
+    const opfPath = xml(await text('META-INF/container.xml')).querySelector('rootfile').getAttribute('full-path');
+    const opf = xml(await text(opfPath));
+    const base = opfPath.includes('/') ? opfPath.slice(0, opfPath.lastIndexOf('/')) : '';
+    const hrefs = new Map([...opf.querySelectorAll('manifest > item')].map((i) => [i.getAttribute('id'), i.getAttribute('href')]));
+    order = [...opf.querySelectorAll('spine > itemref')]
+      // linear="no" is auxiliary content outside the reading order.
+      .filter((r) => r.getAttribute('linear') !== 'no')
+      .map((r) => hrefs.get(r.getAttribute('idref')))
+      .filter(Boolean)
+      .map((h) => resolve(base, decodeURIComponent(h.split('#')[0])))
+      .filter((p) => entries.has(p));
+  } catch { /* malformed package file: fall through */ }
+  if (!order?.length) order = [...entries.keys()].filter(isPage).sort();
+
+  const out = [];
+  let cover = null;
+  for (const path of order) {
+    const doc = page(await text(path));
+    const body = doc.querySelector('body') ?? doc.documentElement;
+    // Furigana would otherwise be read twice: once as kanji, once as kana.
+    doc.querySelectorAll('rt, rp').forEach((n) => n.remove());
+    // A quote holding paragraphs would yield its text twice; keep the leaves.
+    const blocks = [...body.querySelectorAll(BLOCKS)].filter((b) => !b.querySelector(BLOCKS));
+    for (const b of blocks.length ? blocks : [body]) {
+      const t = b.textContent.replace(/\s+/g, ' ').trim();
+      if (t) out.push(t);
+    }
+  }
+
+  // The largest image is almost always the cover, and this works on the
+  // malformed epubs that converted books usually are.
+  const images = [...entries.keys()].filter((n) => /\.(jpe?g|png|webp)$/i.test(n));
+  if (images.length) {
+    const sized = await Promise.all(images.map(async (n) => [n, (await entries.get(n)()).length]));
+    const [name, size] = sized.sort((a, b) => b[1] - a[1])[0];
+    if (size >= 1024) cover = { name, bytes: await entries.get(name)() };
+  }
+  return { paragraphs: out, cover };
+}
+
+/* ------------------------------------------------------------ other formats */
+
+function fb2(source) {
+  const doc = new DOMParser().parseFromString(source, 'application/xml');
+  const bodies = [...doc.getElementsByTagName('body')]
+    // Footnotes live in a second body; nobody narrates them.
+    .filter((b) => (b.getAttribute('name') || '') !== 'notes');
+  const out = [];
+  for (const body of bodies) {
+    for (const p of body.querySelectorAll('p, v, subtitle, text-author')) {
+      const t = p.textContent.replace(/\s+/g, ' ').trim();
+      if (t) out.push(t);
+    }
+  }
+  return { paragraphs: out, cover: null };
+}
+
+/** Aozora Bunko: strip the legend, the colophon, ruby readings and editorial notes. */
+export function aozora(raw) {
+  let lines = raw.replace(/\r\n/g, '\n').split('\n');
+  const rules = lines.flatMap((l, i) => (l.startsWith('-----') ? [i] : []));
+  if (rules.length >= 2) lines = lines.slice(rules[1] + 1);
+  const colophon = lines.findIndex((l) => l.startsWith('底本:'));
+  if (colophon >= 0) lines = lines.slice(0, colophon);
+  const paragraphs = lines
+    .map((l) => l.replace(/[#[^]]*]/g, '').replace(/《[^》]*》/g, '').replaceAll('|', '').trim())
+    .filter(Boolean);
+  return { paragraphs, cover: null };
+}
+
+function plain(text) {
+  if (text.includes('《') && text.includes('底本:')) return aozora(text);
+  return { paragraphs: text.replace(/\r\n/g, '\n').split('\n').map((l) => l.trim()).filter(Boolean), cover: null };
+}
+
+/* ------------------------------------------------------------------ language */
+
+// The commonest little words of each language. Crude, and enough: a book is
+// tens of thousands of words, and the visitor can always overrule the guess.
+const STOPWORDS = {
+  en: 'the and of to in that was his he it with as for had you not be her on at by which have from this',
+  es: 'de la que el en y a los del se las por un para con no una su al lo como más pero sus le ya',
+  pt: 'de a o que e do da em um para é com não uma os no se na por mais as dos como mas foi ao ele',
+  fr: 'de la le et les des en un du une que est pour qui dans a par plus pas au sur ne se ce il sont',
+  de: 'der die und in den von zu das mit sich des auf für ist im dem nicht ein eine als auch es an er',
+  it: 'di e il la che in a per un è del non le si con i da una dei più al come ma lo gli nel alla',
+  nl: 'de van het een en in is dat op te zijn voor met die niet aan er om ook als dan maar bij hij',
+  sv: 'och i att det som en på är av för med till den har de inte om ett han men var jag sig från',
+  fi: 'ja on ei että oli hän se en mutta niin kun kuin joka hänen ole sen olen mitä minä jos nyt vain',
+  pl: 'i w nie na z że się do to jest jak a o po ale co tak za od go był przez już tylko jego',
+  tr: 'bir ve bu da de için ile ne o ben gibi çok daha ama kadar sonra en var mi diye her olan',
+  ru: 'и в не на я что он с как а то это все она так его но да ты к у же вы за бы по только',
+  uk: 'і в не на я що він з як а то це все вона так його але та ти до у ж ви за би по тільки',
+};
+
+/** Best guess at a Whisper language code for `paragraphs`, or null. */
+export function detectLanguage(paragraphs) {
+  const sample = paragraphs.join(' ').slice(0, 60000);
+  const count = (re) => (sample.match(re) || []).length;
+  const kana = count(/[぀-ヿ]/g), han = count(/[一-鿿]/g), hangul = count(/[가-힯]/g);
+  if (kana > 50) return 'ja';
+  if (hangul > 50) return 'ko';
+  if (han > 200) return 'zh';
+  if (count(/[Ͱ-Ͽ]/g) > 200) return 'el';
+  if (count(/[֐-׿]/g) > 200) return 'he';
+  if (count(/[؀-ۿ]/g) > 200) return 'ar';
+  if (count(/[฀-๿]/g) > 200) return 'th';
+
+  const words = sample.toLowerCase().match(/\p{L}+/gu) || [];
+  if (words.length < 50) return null;
+  const tally = new Map();
+  for (const w of words) tally.set(w, (tally.get(w) || 0) + 1);
+  let best = null, bestScore = 0;
+  for (const [code, list] of Object.entries(STOPWORDS)) {
+    const score = list.split(' ').reduce((n, w) => n + (tally.get(w) || 0), 0);
+    if (score > bestScore) { best = code; bestScore = score; }
+  }
+  // Fewer than one word in twelve a stopword of the winner: not one of ours.
+  return bestScore / words.length > 0.08 ? best : null;
+}
diff --git a/frontend/engine/job.js b/frontend/engine/job.js
new file mode 100644
index 0000000..c2c6c16
--- /dev/null
+++ b/frontend/engine/job.js
@@ -0,0 +1,234 @@
+/* One conversion, start to finish, in this tab.
+ *
+ *   audio --ffmpeg--> two-minute chunks --whisper--> rough transcript
+ *   book  ----------> paragraphs
+ *   transcript + paragraphs --align--> subtitles (.srt)
+ *   cover + audio (+ subtitles) --ffmpeg--> video, on request
+ *
+ * Nothing leaves the machine. The transcript is saved to IndexedDB after every
+ * chunk, because transcribing a book takes hours and tabs get closed: opening
+ * the same files again carries on from the last chunk, and aligning the same
+ * audio against a different edition of the book costs seconds.
+ */
+import { Media, quietestPoint, SAMPLE_RATE } from './media.js';
+import { Recogniser } from './asr.js';
+import { readBook } from './book.js';
+
+const CHUNK_SECONDS = 120;
+
+export class Cancelled extends Error { constructor() { super('Stopped.'); } }
+
+/* ------------------------------------------------------------ saved progress */
+
+const db = () => new Promise((resolve, reject) => {
+  const open = indexedDB.open('subread', 1);
+  open.onupgradeneeded = () => open.result.createObjectStore('transcripts');
+  open.onsuccess = () => resolve(open.result);
+  open.onerror = () => reject(open.error);
+});
+
+async function store(mode, fn) {
+  const d = await db();
+  return new Promise((resolve, reject) => {
+    const tx = d.transaction('transcripts', mode);
+    const req = fn(tx.objectStore('transcripts'));
+    tx.oncomplete = () => resolve(req?.result);
+    tx.onerror = () => reject(tx.error);
+  });
+}
+
+/** The same files picked again resume, whatever route they were picked by. */
+const keyFor = (files, language) => files.map((f) => `${f.name}|${f.size}`).join('//') + `#${language}`;
+
+export async function savedProgress(files, language) {
+  try { return (await store('readonly', (s) => s.get(keyFor(files, language)))) ?? null; } catch { return null; }
+}
+
+/* ------------------------------------------------------------------- the job */
+
+export class Job {
+  #media = null;
+  #paths = [];
+  #cancelled = false;
+  #book = null;
+
+  constructor({ audio, book, language, onStatus }) {
+    this.audio = audio;          // File[], in playback order
+    this.bookFile = book;        // File
+    this.language = language;    // Whisper code; always explicit, see asr.js
+    this.onStatus = onStatus ?? (() => {});
+    this.result = null;
+  }
+
+  cancel() { this.#cancelled = true; }
+  #check() { if (this.#cancelled) throw new Cancelled(); }
+  #status(phase, detail, fraction, extra = {}) { this.onStatus({ phase, detail, fraction, ...extra }); }
+
+  async run() {
+    this.#status('preparing', 'Reading the book', 0);
+    this.#book = await readBook(this.bookFile);
+    if (!this.#book.paragraphs.length) throw new Error(`No text could be read from ${this.bookFile.name}.`);
+
+    this.#status('preparing', 'Starting the audio decoder', 0);
+    this.#media = await Media.open();
+    const parts = [];
+    for (const [i, file] of this.audio.entries()) {
+      const path = await this.#media.mount(file, `/in${i}`);
+      parts.push({ path, ...(await this.#media.probe(path)) });
+      this.#paths.push(path);
+    }
+    const total = parts.reduce((n, p) => n + p.duration, 0);
+
+    const transcript = await this.#transcribe(parts, total);
+    this.#check();
+
+    this.#status('aligning', 'Matching the book to the narration', 0.97);
+    const aligned = await alignInWorker(transcript, this.#book.paragraphs, this.language);
+    const stem = (this.audio.length > 1 ? this.bookFile : this.audio[0]).name.replace(/\.[^.]+$/, '');
+    this.result = {
+      ...aligned, stem, duration: total, parts,
+      srtName: `${stem}.${this.language}.srt`,
+      segments: transcript.length,
+    };
+    this.#status('done', 'Done', 1);
+    return this.result;
+  }
+
+  async #transcribe(parts, total) {
+    const key = keyFor(this.audio, this.language);
+    const saved = (await savedProgress(this.audio, this.language)) ?? { doneUntil: 0, segments: [], complete: false };
+    if (saved.complete) return saved.segments;
+
+    this.#status('preparing', 'Loading the speech model', 0, { indeterminate: true });
+    const asr = await Recogniser.open((p) => {
+      if (p.total) this.#status('preparing', 'Downloading the speech model (once)', p.loaded / p.total * 0.02);
+    });
+    this.device = asr.device;
+
+    const resumedAt = saved.doneUntil, started = performance.now();
+    let base = 0;   // where the current part starts on the whole book's clock
+    for (const part of parts) {
+      let pos = Math.max(0, saved.doneUntil - base);
+      while (pos < part.duration - 0.05) {
+        this.#check();
+        let pcm = await this.#media.pcm(part.path, pos, CHUNK_SECONDS);
+        if (!pcm.length) break;
+        const last = pos + pcm.length / SAMPLE_RATE >= part.duration - 0.5;
+        // Cut where the narrator pauses, so no word straddles two chunks.
+        if (!last) pcm = pcm.subarray(0, quietestPoint(pcm));
+
+        const at = base + pos;
+        saved.segments.push(...await asr.transcribe(pcm, this.language, at));
+        pos += pcm.length / SAMPLE_RATE;
+        saved.doneUntil = base + pos;
+        await store('readwrite', (s) => s.put(saved, key)).catch(() => {});
+
+        const elapsed = (performance.now() - started) / 1000;
+        const speed = (saved.doneUntil - resumedAt) / elapsed;
+        this.#status('transcribing', `Listening: ${clock(saved.doneUntil)} of ${clock(total)}`,
+          0.02 + 0.94 * saved.doneUntil / total,
+          { eta: elapsed > 10 ? (total - saved.doneUntil) / speed : null, speed: elapsed > 10 ? speed : null });
+      }
+      base += part.duration;
+    }
+    saved.complete = true;
+    await store('readwrite', (s) => s.put(saved, key)).catch(() => {});
+    return saved.segments;
+  }
+
+  /**
+   * The video, made on request: a still image over the audio.
+   * kind "mkv" carries the subtitles inside the file (MPV, VLC);
+   * kind "mp4" is clean, for YouTube, where the .srt is uploaded alongside.
+   *
+   * A still picture does not need encoding for ten hours. One minute of it is
+   * encoded once and then repeated by stream copy, so the cost is that of
+   * copying the audio - seconds, not an hour.
+   */
+  async video(kind, { coverFile, onProgress } = {}) {
+    const m = this.#media, r = this.result;
+    const still = await canvasPng(coverFile ?? this.#book.cover);
+    await m.write('canvas.png', still);
+    await m.run(['-loop', '1', '-framerate', '1', '-i', 'canvas.png', '-t', '60',
+      '-c:v', 'libx264', '-preset', 'veryfast', '-tune', 'stillimage', '-pix_fmt', 'yuv420p',
+      '-r', '1', '-g', '60', 'still.mp4']);
+
+    let audioIn;
+    if (this.#paths.length === 1) audioIn = ['-i', this.#paths[0]];
+    else {
+      const list = this.#paths.map((p) => `file '${p.replaceAll("'", "'\\''")}'`).join('\n');
+      await m.write('parts.txt', new TextEncoder().encode(list));
+      audioIn = ['-f', 'concat', '-safe', '0', '-i', 'parts.txt'];
+    }
+    // Copy the audio when the container can hold it; re-encoding a whole book
+    // in WebAssembly is the one slow thing here, so only when there is no choice.
+    const copyable = kind === 'mkv'
+      ? ['aac', 'mp3', 'opus', 'vorbis', 'flac', 'ac3', 'alac']
+      : ['aac', 'mp3', 'ac3', 'alac'];
+    const codecs = new Set(r.parts.map((p) => p.codec));
+    const audioCodec = codecs.size === 1 && copyable.includes([...codecs][0]) ? ['-c:a', 'copy'] : ['-c:a', 'aac', '-b:a', '128k'];
+
+    const out = `out.${kind}`;
+    const args = ['-stream_loop', '-1', '-i', 'still.mp4', ...audioIn];
+    if (kind === 'mkv') {
+      await m.write('subs.srt', new TextEncoder().encode(r.srt));
+      args.push('-f', 'srt', '-i', 'subs.srt', '-map', '0:v', '-map', '1:a:0', '-map', '2:s',
+        '-c:v', 'copy', ...audioCodec, '-c:s', 'srt', '-disposition:s:0', 'default',
+        '-metadata:s:s:0', 'title=Aligned subtitles');
+    } else {
+      args.push('-map', '0:v', '-map', '1:a:0', '-c:v', 'copy', ...audioCodec);
+    }
+    args.push('-map_chapters', '1', '-t', r.duration.toFixed(3), out);
+
+    try {
+      await m.run(args, onProgress);
+      const bytes = await m.read(out);
+      return new File([bytes], `${r.stem}.${this.language}.${kind}`,
+        { type: kind === 'mkv' ? 'video/x-matroska' : 'video/mp4' });
+    } finally {
+      for (const f of [out, 'still.mp4', 'canvas.png', 'subs.srt', 'parts.txt']) await m.remove(f);
+    }
+  }
+
+  close() { this.#media?.close(); }
+}
+
+function alignInWorker(transcript, paragraphs, language) {
+  return new Promise((resolve, reject) => {
+    const worker = new Worker(new URL('./align.worker.js', import.meta.url), { type: 'module' });
+    worker.onmessage = ({ data }) => {
+      worker.terminate();
+      if (data.ok) resolve(data); else reject(new Error(data.error));
+    };
+    worker.onerror = (e) => { worker.terminate(); reject(new Error(e.message || 'The aligner crashed.')); };
+    worker.postMessage({ transcript, paragraphs, language });
+  });
+}
+
+/** The cover, letterboxed onto a 1920x1080 card; a plain dark card when there is none. */
+async function canvasPng(cover) {
+  const W = 1920, H = 1080;
+  const canvas = new OffscreenCanvas(W, H);
+  const ctx = canvas.getContext('2d');
+  ctx.fillStyle = '#16161a';
+  ctx.fillRect(0, 0, W, H);
+  if (cover) {
+    try {
+      const blob = cover instanceof Blob ? cover : new Blob([cover.bytes]);
+      const img = await createImageBitmap(blob);
+      const scale = Math.min(W / img.width, H / img.height);
+      const w = img.width * scale, h = img.height * scale;
+      ctx.fillStyle = '#000';
+      ctx.fillRect(0, 0, W, H);
+      ctx.drawImage(img, (W - w) / 2, (H - h) / 2, w, h);
+    } catch { /* an image the browser cannot decode: keep the plain card */ }
+  }
+  const png = await canvas.convertToBlob({ type: 'image/png' });
+  return new Uint8Array(await png.arrayBuffer());
+}
+
+export function clock(seconds) {
+  const s = Math.max(0, Math.floor(seconds)), pad = (v) => String(v).padStart(2, '0');
+  return s >= 3600 ? `${Math.floor(s / 3600)}:${pad(Math.floor(s / 60) % 60)}:${pad(s % 60)}`
+    : `${Math.floor(s / 60)}:${pad(s % 60)}`;
+}
diff --git a/frontend/engine/media.js b/frontend/engine/media.js
new file mode 100644
index 0000000..7ad378c
--- /dev/null
+++ b/frontend/engine/media.js
@@ -0,0 +1,111 @@
+/* Reading and writing media in the browser, with ffmpeg compiled to WebAssembly.
+ *
+ * The visitor's file is never copied anywhere: it is mounted read-only into
+ * ffmpeg's filesystem (WORKERFS), which reads straight from the File on disk.
+ * That is what makes a 600 MB audiobook workable - the only audio ever held in
+ * memory is the couple of minutes being transcribed.
+ *
+ * ffmpeg runs in its own worker; everything here is async message-passing.
+ */
+import { FFmpeg } from '/vendor/ffmpeg/index.js';
+
+const CORE = new URL('/vendor/ffmpeg/ffmpeg-core.js', import.meta.url).href;
+const RATE = 16000;
+
+export class Media {
+  #ffmpeg = new FFmpeg();
+  #log = [];
+  #mounted = [];
+
+  /** Bytes fetched so far while the 32 MB core downloads, for a progress bar. */
+  static async open(onLog) {
+    const m = new Media();
+    m.#ffmpeg.on('log', ({ message }) => {
+      m.#log.push(message);
+      if (m.#log.length > 400) m.#log.shift();
+      onLog?.(message);
+    });
+    await m.#ffmpeg.load({ coreURL: CORE });
+    return m;
+  }
+
+  /** Make `file` readable by ffmpeg at the returned path, without copying it. */
+  async mount(file, dir) {
+    await this.#ffmpeg.createDir(dir).catch(() => {});
+    await this.#ffmpeg.mount('WORKERFS', { files: [file] }, dir);
+    this.#mounted.push(dir);
+    return `${dir}/${file.name}`;
+  }
+
+  /** Duration in seconds and chapter marks, read from ffmpeg's own banner. */
+  async probe(path) {
+    this.#log.length = 0;
+    await this.#ffmpeg.exec(['-hide_banner', '-i', path]);   // "fails": no output given. Expected.
+    const text = this.#log.join('\n');
+    const d = /Duration:\s*(\d+):(\d+):(\d+(?:\.\d+)?)/.exec(text);
+    if (!d) throw new Error('This file does not look like audio ffmpeg can read.');
+    const audio = /Stream #\d+:\d+.*?: Audio: ([a-z0-9_]+)/i.exec(text);
+    return {
+      duration: (+d[1]) * 3600 + (+d[2]) * 60 + (+d[3]),
+      codec: audio ? audio[1].toLowerCase() : null,
+      chapters: [...text.matchAll(/Chapter #\d+:\d+: start ([\d.]+), end ([\d.]+)/g)]
+        .map((c) => ({ start: +c[1], end: +c[2] })),
+    };
+  }
+
+  /** `seconds` of audio from `start`, as the 16 kHz mono float PCM the speech model takes. */
+  async pcm(path, start, seconds) {
+    const out = 'chunk.f32';
+    const code = await this.#ffmpeg.exec([
+      '-hide_banner', '-v', 'error',
+      // Tolerate damaged files: drop what cannot be decoded, keep going.
+      '-err_detect', 'ignore_err', '-fflags', '+discardcorrupt',
+      '-ss', String(start), '-t', String(seconds), '-i', path,
+      '-map', '0:a:0', '-vn', '-sn', '-ac', '1', '-ar', String(RATE), '-f', 'f32le', out,
+    ]);
+    if (code !== 0) throw new Error(`Could not decode the audio at ${Math.round(start)}s.`);
+    const bytes = await this.#ffmpeg.readFile(out);
+    await this.#ffmpeg.deleteFile(out);
+    return new Float32Array(bytes.buffer, bytes.byteOffset, bytes.byteLength >> 2);
+  }
+
+  async write(name, data) { await this.#ffmpeg.writeFile(name, data); }
+  async read(name) { return this.#ffmpeg.readFile(name); }
+  async remove(name) { await this.#ffmpeg.deleteFile(name).catch(() => {}); }
+
+  /** Run ffmpeg; throws with the tail of its log when it fails. */
+  async run(args, onProgress) {
+    this.#log.length = 0;
+    const handler = onProgress && (({ progress }) => onProgress(Math.min(Math.max(progress, 0), 1)));
+    if (handler) this.#ffmpeg.on('progress', handler);
+    try {
+      const code = await this.#ffmpeg.exec(['-hide_banner', '-y', ...args]);
+      if (code !== 0) throw new Error(this.#log.slice(-3).join(' | ') || `ffmpeg exited with ${code}`);
+    } finally {
+      if (handler) this.#ffmpeg.off('progress', handler);
+    }
+  }
+
+  close() { this.#ffmpeg.terminate(); }
+}
+
+/**
+ * Where to cut a chunk: the quietest quarter second in its last few seconds,
+ * so that no word is split across two transcriptions.
+ * Returns a sample index into `pcm`.
+ */
+export function quietestPoint(pcm, searchSeconds = 12, windowSeconds = 0.25) {
+  const window = Math.floor(windowSeconds * RATE);
+  const from = Math.max(0, pcm.length - searchSeconds * RATE);
+  if (pcm.length - from <= window * 2) return pcm.length;
+  let energy = 0;
+  for (let i = from; i < from + window; i++) energy += Math.abs(pcm[i]);
+  let best = energy, bestAt = from;
+  for (let i = from; i + window < pcm.length; i++) {
+    energy += Math.abs(pcm[i + window]) - Math.abs(pcm[i]);
+    if (energy < best) { best = energy; bestAt = i + 1; }
+  }
+  return bestAt + (window >> 1);
+}
+
+export const SAMPLE_RATE = RATE;
diff --git a/frontend/lab.html b/frontend/lab.html
new file mode 100644
index 0000000..24bd965
--- /dev/null
+++ b/frontend/lab.html
@@ -0,0 +1,93 @@
+<!DOCTYPE html>
+<meta charset="utf-8">
+<title>engine lab</title>
+<body style="font:14px/1.5 ui-monospace,monospace;padding:20px;max-width:900px">
+<h3>Browser engine lab</h3>
+<p>Proves the client-side pieces one at a time. Not linked from the site.</p>
+<input type="file" id="f"> <button id="go">run on /vendor/lab/ru.mp3</button>
+<pre id="out"></pre>
+<script type="module">
+import { Media, quietestPoint } from '/engine/media.js';
+import { Recogniser, hasWebGpu } from '/engine/asr.js';
+
+const out = document.getElementById('out');
+const log = (...a) => { out.textContent += a.join(' ') + '\n'; };
+window.lab = { done: false, error: null, results: {} };
+
+async function run(file) {
+  try {
+    log('crossOriginIsolated:', crossOriginIsolated, '| threads:', navigator.hardwareConcurrency,
+        '| webgpu:', await hasWebGpu());
+    let t = performance.now();
+    const media = await Media.open();
+    log(`ffmpeg loaded in ${((performance.now() - t) / 1000).toFixed(1)}s`);
+
+    const path = await media.mount(file, '/in');
+    const info = await media.probe(path);
+    log('probe:', JSON.stringify(info));
+
+    t = performance.now();
+    const pcm = await media.pcm(path, 0, 60);
+    log(`decoded ${(pcm.length / 16000).toFixed(1)}s of audio in ${((performance.now() - t) / 1000).toFixed(2)}s;`,
+        'cut at', (quietestPoint(pcm) / 16000).toFixed(2) + 's');
+
+    t = performance.now();
+    const asr = await Recogniser.open((p) => { window.lab.results.download = `${p.file} ${Math.round(p.progress)}%`; });
+    log(`model ready on ${asr.device} in ${((performance.now() - t) / 1000).toFixed(1)}s`);
+
+    const lang = new URLSearchParams(location.search).get('lang') || null;
+    t = performance.now();
+    await asr.transcribe(pcm.subarray(0, 16000 * 20), lang, 0);
+    log(`warm-up pass (shader compile) ${((performance.now() - t) / 1000).toFixed(1)}s, language=${lang}`);
+    t = performance.now();
+    const segments = await asr.transcribe(pcm, lang, 0);
+    const took = (performance.now() - t) / 1000;
+    log(`transcribed in ${took.toFixed(1)}s = ${(pcm.length / 16000 / took).toFixed(1)}x real time, ${segments.length} segments`);
+    for (const s of segments.slice(0, 8)) log(`  ${s.start.toFixed(2)}-${s.end.toFixed(2)}  ${s.text}`);
+    window.lab.results = { info, device: asr.device, took, segments };
+  } catch (e) {
+    log('FAILED:', e?.stack || e);
+    window.lab.error = String(e?.message || e);
+  }
+  window.lab.done = true;
+}
+
+document.getElementById('go').onclick = async () => {
+  const blob = await (await fetch('/vendor/lab/ru.mp3')).blob();
+  run(new File([blob], 'ru.mp3'));
+};
+// The whole job: decode, transcribe, align, then both videos.
+window.fullJob = async () => {
+  window.lab = { done: false, error: null, results: {} };
+  out.textContent = '';
+  try {
+    const { Job } = await import('/engine/job.js');
+    const { detectLanguage, readBook } = await import('/engine/book.js');
+    const get = async (n) => new File([await (await fetch('/vendor/lab/' + n)).blob()], n);
+    const audio = await get('ru.mp3'), book = await get('moskva.epub');
+    const parsed = await readBook(book);
+    const lang = detectLanguage(parsed.paragraphs);
+    log(`book: ${parsed.paragraphs.length} paragraphs, cover ${parsed.cover ? parsed.cover.name : 'none'}, detected language ${lang}`);
+    let last = '';
+    const job = new Job({ audio: [audio], book, language: lang, onStatus: (st) => {
+      const line = `${st.phase}: ${st.detail}`;
+      if (line !== last) { last = line; log('  ' + line, st.speed ? `(${st.speed.toFixed(1)}x)` : ''); }
+    }});
+    let t = performance.now();
+    const r = await job.run();
+    log(`job done in ${((performance.now() - t) / 1000).toFixed(1)}s on ${job.device}: ${r.cues.length} cues, ` +
+        `${(r.matchRate * 100).toFixed(0)}% matched, ${r.paragraphsDropped}/${r.paragraphsDropped + r.paragraphsUsed} paragraphs dropped`);
+    log(r.srt.split('\n').slice(0, 16).join('\n'));
+    for (const kind of ['mkv', 'mp4']) {
+      t = performance.now();
+      const file = await job.video(kind);
+      log(`${kind}: ${file.name} ${(file.size / 1e6).toFixed(2)} MB in ${((performance.now() - t) / 1000).toFixed(1)}s`);
+      window.lab.results[kind] = file;
+    }
+    window.lab.results.srt = r.srt;
+  } catch (e) { log('FAILED:', e?.stack || e); window.lab.error = String(e?.message || e); }
+  window.lab.done = true;
+};
+
+document.getElementById('f').onchange = (e) => run(e.target.files[0]);
+</script>
diff --git a/tools/fetch_vendor.py b/tools/fetch_vendor.py
new file mode 100644
index 0000000..1cdf69a
--- /dev/null
+++ b/tools/fetch_vendor.py
@@ -0,0 +1,88 @@
+"""Download the browser-side libraries into frontend/vendor/.
+
+Processing happens in the visitor's browser, which needs three things that are
+far too big to commit: ffmpeg compiled to WebAssembly (reads any audio format,
+muxes the video), the ONNX runtime, and transformers.js to drive Whisper on it.
+They are pinned here by exact version and unpacked from the npm registry, then
+served from our own origin - no CDN in the page, so nothing third-party has to
+be trusted or kept alive, and cross-origin isolation stays simple.
+
+    python tools/fetch_vendor.py          # idempotent; run on every deploy
+
+Standard library only, on purpose: this runs on the server before anything else
+is installed.
+"""
+
+from __future__ import annotations
+
+import hashlib
+import io
+import json
+import sys
+import tarfile
+import urllib.request
+from pathlib import Path
+
+VENDOR = Path(__file__).resolve().parent.parent / "frontend" / "vendor"
+
+# package, version, {path inside the tarball: path under vendor/}
+PACKAGES = [
+    ("@ffmpeg/ffmpeg", "0.12.15", {
+        f"package/dist/esm/{name}": f"ffmpeg/{name}"
+        for name in ("index.js", "classes.js", "const.js", "errors.js", "types.js", "utils.js", "worker.js")
+    }),
+    ("@ffmpeg/core", "0.12.10", {
+        "package/dist/esm/ffmpeg-core.js": "ffmpeg/ffmpeg-core.js",
+        "package/dist/esm/ffmpeg-core.wasm": "ffmpeg/ffmpeg-core.wasm",
+    }),
+    ("@huggingface/transformers", "4.3.0", {
+        "package/dist/transformers.min.js": "transformers/transformers.min.js",
+    }),
+    # Must be exactly the build transformers.js was made against.
+    ("onnxruntime-web", "1.31.0-dev.20260914-8d85527a0", {
+        f"package/dist/{name}": f"transformers/{name}"
+        for name in ("ort.webgpu.bundle.min.mjs", "ort-wasm-simd-threaded.asyncify.mjs",
+                     "ort-wasm-simd-threaded.asyncify.wasm")
+    }),
+]
+
+
+def fetch(package: str, version: str) -> bytes:
+    meta_url = f"https://registry.npmjs.org/{package.replace('/', '%2F')}/{version}"
+    with urllib.request.urlopen(meta_url, timeout=60) as r:
+        dist = json.load(r)["dist"]
+    with urllib.request.urlopen(dist["tarball"], timeout=600) as r:
+        blob = r.read()
+    # The registry's own checksum: a tampered or truncated download stops here.
+    if hashlib.sha1(blob).hexdigest() != dist["shasum"]:
+        sys.exit(f"{package}@{version}: checksum mismatch")
+    return blob
+
+
+def main() -> None:
+    stamp = VENDOR / "versions.json"
+    want = {p: v for p, v, _ in PACKAGES}
+    have = json.loads(stamp.read_text()) if stamp.exists() else {}
+
+    for package, version, files in PACKAGES:
+        targets = [VENDOR / dest for dest in files.values()]
+        if have.get(package) == version and all(t.exists() for t in targets):
+            print(f"ok       {package}@{version}")
+            continue
+        print(f"fetching {package}@{version}")
+        with tarfile.open(fileobj=io.BytesIO(fetch(package, version)), mode="r:gz") as tar:
+            for src, dest in files.items():
+                member = tar.extractfile(src)
+                if member is None:
+                    sys.exit(f"{package}@{version}: {src} is not in the package")
+                out = VENDOR / dest
+                out.parent.mkdir(parents=True, exist_ok=True)
+                out.write_bytes(member.read())
+
+    stamp.write_text(json.dumps(want, indent=2))
+    total = sum(f.stat().st_size for f in VENDOR.rglob("*") if f.is_file())
+    print(f"vendor/ is {total / 1e6:.0f} MB")
+
+
+if __name__ == "__main__":
+    main()