commit 7e1fe5c203ee7184fea1591ec27ac6271af522d2
equwal <truex@equwal.com>
2026-09-20 17:22:47 -0700
The whole conversion, in the browser
engine/media.js ffmpeg.wasm: reads any audio format straight off the
visitor's disk (WORKERFS, no copy) two minutes at a time,
and muxes the videos
engine/asr.js Whisper-tiny through transformers.js; WebGPU, else WASM
threads. Language is always explicit - left to itself the
model translates instead of transcribing
engine/book.js epub / fb2 / Aozora / text in the browser, with a zip
reader built on DecompressionStream; language detection
engine/job.js the job: resumable (IndexedDB) transcription, alignment in
a worker, videos on request. A still picture is encoded
for one minute and repeated by stream copy, so a video
costs seconds rather than an hour of WebAssembly x264
tools/fetch_vendor.py pins and checksums the three libraries and serves them
from our own origin; the page is cross-origin isolated so the runtime can
use threads. lab.html exercises each piece; verified on a real sample: a
clip from the middle of a novel finds its place in the whole epub, and both
videos come out with the right streams.
.gitignore | 3 +
backend/main.py | 22 ++++
frontend/engine/align.worker.js | 11 ++
frontend/engine/asr.js | 71 ++++++++++++
frontend/engine/book.js | 224 ++++++++++++++++++++++++++++++++++++++
frontend/engine/job.js | 234 ++++++++++++++++++++++++++++++++++++++++
frontend/engine/media.js | 111 +++++++++++++++++++
frontend/lab.html | 93 ++++++++++++++++
tools/fetch_vendor.py | 88 +++++++++++++++
9 files changed, 857 insertions(+)
diff --git a/.gitignore b/.gitignore
index 6c69c5d..354a1ba 100644
--- a/.gitignore
+++ b/.gitignore
@@ -24,3 +24,6 @@ Thumbs.db
# Whole-book test input; local only
tests/engine/golden-local/
+
+# Browser-side libraries; tools/fetch_vendor.py downloads them, pinned
+frontend/vendor/
diff --git a/backend/main.py b/backend/main.py
index 51a36b5..3fd7bc7 100644
--- a/backend/main.py
+++ b/backend/main.py
@@ -9,6 +9,7 @@ CORS config and no second server.
from __future__ import annotations
import logging
+import mimetypes
from contextlib import asynccontextmanager
from fastapi import FastAPI
@@ -29,6 +30,12 @@ log = logging.getLogger("subplz.web")
FRONTEND = ROOT / "frontend"
+# Python takes these from the OS, and Windows gets both wrong. A module served
+# as text/plain is refused outright, and WebAssembly will not stream-compile.
+mimetypes.add_type("text/javascript", ".js")
+mimetypes.add_type("text/javascript", ".mjs")
+mimetypes.add_type("application/wasm", ".wasm")
+
def _requeue_interrupted() -> None:
"""Recover jobs that were mid-flight when the server last stopped.
@@ -79,6 +86,21 @@ app = FastAPI(
app.include_router(router)
+@app.middleware("http")
+async def cross_origin_isolation(request, call_next):
+ """Let the page use threads.
+
+ The speech model runs in the browser on WebAssembly threads, which need
+ SharedArrayBuffer, which browsers only hand to a cross-origin-isolated
+ page. "credentialless" rather than "require-corp": the model weights come
+ from a third-party host that sends CORS headers but not CORP ones.
+ """
+ response = await call_next(request)
+ response.headers["Cross-Origin-Opener-Policy"] = "same-origin"
+ response.headers["Cross-Origin-Embedder-Policy"] = "credentialless"
+ return response
+
+
@app.get("/healthz")
def healthz():
return {
diff --git a/frontend/engine/align.worker.js b/frontend/engine/align.worker.js
new file mode 100644
index 0000000..fb2fd4f
--- /dev/null
+++ b/frontend/engine/align.worker.js
@@ -0,0 +1,11 @@
+/* Alignment off the main thread: a long book is a few seconds of solid work. */
+import { alignBook, language, writeSrt } from './align.js';
+
+self.onmessage = ({ data }) => {
+ try {
+ const result = alignBook(data.transcript, data.paragraphs, language(data.language));
+ self.postMessage({ ok: true, ...result, srt: writeSrt(result.cues) });
+ } catch (e) {
+ self.postMessage({ ok: false, error: String(e?.message || e) });
+ }
+};
diff --git a/frontend/engine/asr.js b/frontend/engine/asr.js
new file mode 100644
index 0000000..6a48b9d
--- /dev/null
+++ b/frontend/engine/asr.js
@@ -0,0 +1,71 @@
+/* Speech recognition in the browser: Whisper-tiny on the ONNX runtime, through
+ * transformers.js. WebGPU where the browser has it (many times faster than real
+ * time on an ordinary laptop), WebAssembly threads where it does not.
+ *
+ * The transcript only has to be good enough to line up against a book we
+ * already have the words of, which is why the smallest model will do.
+ */
+import { pipeline, env } from '/vendor/transformers/transformers.min.js';
+
+// Everything from our own origin except the model weights themselves.
+env.backends.onnx.wasm.wasmPaths = new URL('/vendor/transformers/', import.meta.url).href;
+env.allowLocalModels = false;
+
+const MODEL = 'onnx-community/whisper-tiny';
+
+export async function hasWebGpu() {
+ try { return !!(navigator.gpu && await navigator.gpu.requestAdapter()); } catch { return false; }
+}
+
+export class Recogniser {
+ #pipe;
+ device;
+
+ /** @param onProgress ({file, loaded, total}) while the ~40 MB of weights download (cached after). */
+ static async open(onProgress) {
+ const r = new Recogniser();
+ r.device = (await hasWebGpu()) ? 'webgpu' : 'wasm';
+ const options = (device) => ({
+ device,
+ // Full-precision encoder: quantising it is what ruins accuracy. The
+ // decoder tolerates 4-bit well and is where the time goes.
+ dtype: device === 'webgpu'
+ ? { encoder_model: 'fp32', decoder_model_merged: 'q4' }
+ : { encoder_model: 'fp32', decoder_model_merged: 'q8' },
+ progress_callback: (p) => { if (p.status === 'progress') onProgress?.(p); },
+ });
+ try {
+ r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options(r.device));
+ } catch (e) {
+ if (r.device !== 'webgpu') throw e;
+ // A GPU that is listed but cannot run the model: fall back rather than fail.
+ r.device = 'wasm';
+ r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options('wasm'));
+ }
+ return r;
+ }
+
+ /**
+ * @param pcm 16 kHz mono Float32Array
+ * @param language a Whisper code ("ja", "ru", ...), or null to let it decide
+ * @returns [{text, start, end}] with times offset by `offset` seconds
+ */
+ async transcribe(pcm, language, offset = 0) {
+ const out = await this.#pipe(pcm, {
+ return_timestamps: true,
+ chunk_length_s: 30,
+ stride_length_s: 5,
+ task: 'transcribe',
+ ...(language ? { language } : {}),
+ });
+ const segments = [];
+ for (const c of out.chunks ?? []) {
+ const text = (c.text ?? '').trim();
+ const [s, e] = c.timestamp ?? [];
+ if (!text || s == null) continue;
+ // The last segment of a buffer can come back without an end.
+ segments.push({ text, start: offset + s, end: offset + (e ?? pcm.length / 16000) });
+ }
+ return segments;
+ }
+}
diff --git a/frontend/engine/book.js b/frontend/engine/book.js
new file mode 100644
index 0000000..3387daf
--- /dev/null
+++ b/frontend/engine/book.js
@@ -0,0 +1,224 @@
+/* A book as a flat list of paragraphs in reading order - read in the browser,
+ * never uploaded.
+ *
+ * epub (a zip of XHTML), fb2 (XML), Aozora Bunko (Shift_JIS text with ruby
+ * markup, usually zipped) and plain text. No zip library: the platform can
+ * inflate (DecompressionStream), and the little that is left of the format is
+ * a directory at the end of the file.
+ */
+
+const BLOCKS = 'p, li, blockquote, h1, h2, h3, h4, h5, h6';
+
+export async function readBook(file) {
+ const name = file.name.toLowerCase();
+ const bytes = new Uint8Array(await file.arrayBuffer());
+ if (name.endsWith('.epub')) return epub(bytes);
+ if (name.endsWith('.fb2.zip') || name.endsWith('.fbz')) return fb2(decode(await firstEntry(bytes, /\.fb2$/i)));
+ if (name.endsWith('.zip')) return aozora(decode(await firstEntry(bytes, /\.txt$/i)));
+ if (name.endsWith('.fb2')) return fb2(decode(bytes));
+ return plain(decode(bytes));
+}
+
+/* ----------------------------------------------------------------------- zip */
+
+const u16 = (b, o) => b[o] | (b[o + 1] << 8);
+const u32 = (b, o) => (b[o] | (b[o + 1] << 8) | (b[o + 2] << 16) | (b[o + 3] << 24)) >>> 0;
+
+/** name -> () => Promise<Uint8Array>, in directory order. */
+export function zipEntries(bytes) {
+ let eocd = -1;
+ for (let i = bytes.length - 22; i >= Math.max(0, bytes.length - 65557); i--) {
+ if (u32(bytes, i) === 0x06054b50) { eocd = i; break; }
+ }
+ if (eocd < 0) throw new Error('This file is not a readable zip archive.');
+ const entries = new Map();
+ let o = u32(bytes, eocd + 16);
+ for (let n = u16(bytes, eocd + 10); n > 0 && u32(bytes, o) === 0x02014b50; n--) {
+ const method = u16(bytes, o + 10), size = u32(bytes, o + 20);
+ const nameLen = u16(bytes, o + 28), extraLen = u16(bytes, o + 30), commentLen = u16(bytes, o + 32);
+ const local = u32(bytes, o + 42);
+ const name = new TextDecoder().decode(bytes.subarray(o + 46, o + 46 + nameLen));
+ if (!name.endsWith('/')) {
+ entries.set(name, async () => {
+ const start = local + 30 + u16(bytes, local + 26) + u16(bytes, local + 28);
+ const raw = bytes.subarray(start, start + size);
+ if (method === 0) return raw;
+ if (method !== 8) throw new Error(`Unsupported zip compression in ${name}.`);
+ const stream = new Blob([raw]).stream().pipeThrough(new DecompressionStream('deflate-raw'));
+ return new Uint8Array(await new Response(stream).arrayBuffer());
+ });
+ }
+ o += 46 + nameLen + extraLen + commentLen;
+ }
+ return entries;
+}
+
+async function firstEntry(bytes, pattern) {
+ for (const [name, read] of zipEntries(bytes)) if (pattern.test(name)) return read();
+ throw new Error('Nothing readable was found inside that archive.');
+}
+
+/** UTF-8 unless it plainly is not; then whatever the file declares, or Shift_JIS. */
+function decode(bytes) {
+ try {
+ return new TextDecoder('utf-8', { fatal: true }).decode(bytes).replace(/^/, '');
+ } catch { /* not UTF-8 */ }
+ const head = new TextDecoder('latin1').decode(bytes.subarray(0, 200));
+ const declared = /encoding=["']([\w-]+)["']/i.exec(head)?.[1];
+ for (const label of [declared, 'shift_jis', 'windows-1251']) {
+ if (!label) continue;
+ try { return new TextDecoder(label).decode(bytes); } catch { /* unknown label */ }
+ }
+ return new TextDecoder().decode(bytes);
+}
+
+/* ---------------------------------------------------------------------- epub */
+
+function resolve(base, href) {
+ const parts = base ? base.split('/') : [];
+ for (const seg of href.split('/')) {
+ if (seg === '' || seg === '.') continue;
+ if (seg === '..') parts.pop(); else parts.push(seg);
+ }
+ return parts.join('/');
+}
+
+/**
+ * An epub page is XHTML, and has to be parsed as XML first: to an HTML parser
+ * a self-closing <title/> never closes, and swallows the whole book as its text.
+ * Plenty of epubs are not well-formed, though, so HTML is the fallback.
+ */
+function page(source) {
+ const xml = new DOMParser().parseFromString(source, 'application/xhtml+xml');
+ if (!xml.querySelector('parsererror')) return xml;
+ return new DOMParser().parseFromString(source, 'text/html');
+}
+
+async function epub(bytes) {
+ const entries = zipEntries(bytes);
+ const text = async (name) => decode(await entries.get(name)());
+ const xml = (s) => new DOMParser().parseFromString(s, 'application/xml');
+ const isPage = (n) => /\.(xhtml|html|htm)$/i.test(n);
+
+ let order = null;
+ try {
+ const opfPath = xml(await text('META-INF/container.xml')).querySelector('rootfile').getAttribute('full-path');
+ const opf = xml(await text(opfPath));
+ const base = opfPath.includes('/') ? opfPath.slice(0, opfPath.lastIndexOf('/')) : '';
+ const hrefs = new Map([...opf.querySelectorAll('manifest > item')].map((i) => [i.getAttribute('id'), i.getAttribute('href')]));
+ order = [...opf.querySelectorAll('spine > itemref')]
+ // linear="no" is auxiliary content outside the reading order.
+ .filter((r) => r.getAttribute('linear') !== 'no')
+ .map((r) => hrefs.get(r.getAttribute('idref')))
+ .filter(Boolean)
+ .map((h) => resolve(base, decodeURIComponent(h.split('#')[0])))
+ .filter((p) => entries.has(p));
+ } catch { /* malformed package file: fall through */ }
+ if (!order?.length) order = [...entries.keys()].filter(isPage).sort();
+
+ const out = [];
+ let cover = null;
+ for (const path of order) {
+ const doc = page(await text(path));
+ const body = doc.querySelector('body') ?? doc.documentElement;
+ // Furigana would otherwise be read twice: once as kanji, once as kana.
+ doc.querySelectorAll('rt, rp').forEach((n) => n.remove());
+ // A quote holding paragraphs would yield its text twice; keep the leaves.
+ const blocks = [...body.querySelectorAll(BLOCKS)].filter((b) => !b.querySelector(BLOCKS));
+ for (const b of blocks.length ? blocks : [body]) {
+ const t = b.textContent.replace(/\s+/g, ' ').trim();
+ if (t) out.push(t);
+ }
+ }
+
+ // The largest image is almost always the cover, and this works on the
+ // malformed epubs that converted books usually are.
+ const images = [...entries.keys()].filter((n) => /\.(jpe?g|png|webp)$/i.test(n));
+ if (images.length) {
+ const sized = await Promise.all(images.map(async (n) => [n, (await entries.get(n)()).length]));
+ const [name, size] = sized.sort((a, b) => b[1] - a[1])[0];
+ if (size >= 1024) cover = { name, bytes: await entries.get(name)() };
+ }
+ return { paragraphs: out, cover };
+}
+
+/* ------------------------------------------------------------ other formats */
+
+function fb2(source) {
+ const doc = new DOMParser().parseFromString(source, 'application/xml');
+ const bodies = [...doc.getElementsByTagName('body')]
+ // Footnotes live in a second body; nobody narrates them.
+ .filter((b) => (b.getAttribute('name') || '') !== 'notes');
+ const out = [];
+ for (const body of bodies) {
+ for (const p of body.querySelectorAll('p, v, subtitle, text-author')) {
+ const t = p.textContent.replace(/\s+/g, ' ').trim();
+ if (t) out.push(t);
+ }
+ }
+ return { paragraphs: out, cover: null };
+}
+
+/** Aozora Bunko: strip the legend, the colophon, ruby readings and editorial notes. */
+export function aozora(raw) {
+ let lines = raw.replace(/\r\n/g, '\n').split('\n');
+ const rules = lines.flatMap((l, i) => (l.startsWith('-----') ? [i] : []));
+ if (rules.length >= 2) lines = lines.slice(rules[1] + 1);
+ const colophon = lines.findIndex((l) => l.startsWith('底本:'));
+ if (colophon >= 0) lines = lines.slice(0, colophon);
+ const paragraphs = lines
+ .map((l) => l.replace(/[#[^]]*]/g, '').replace(/《[^》]*》/g, '').replaceAll('|', '').trim())
+ .filter(Boolean);
+ return { paragraphs, cover: null };
+}
+
+function plain(text) {
+ if (text.includes('《') && text.includes('底本:')) return aozora(text);
+ return { paragraphs: text.replace(/\r\n/g, '\n').split('\n').map((l) => l.trim()).filter(Boolean), cover: null };
+}
+
+/* ------------------------------------------------------------------ language */
+
+// The commonest little words of each language. Crude, and enough: a book is
+// tens of thousands of words, and the visitor can always overrule the guess.
+const STOPWORDS = {
+ en: 'the and of to in that was his he it with as for had you not be her on at by which have from this',
+ es: 'de la que el en y a los del se las por un para con no una su al lo como más pero sus le ya',
+ pt: 'de a o que e do da em um para é com não uma os no se na por mais as dos como mas foi ao ele',
+ fr: 'de la le et les des en un du une que est pour qui dans a par plus pas au sur ne se ce il sont',
+ de: 'der die und in den von zu das mit sich des auf für ist im dem nicht ein eine als auch es an er',
+ it: 'di e il la che in a per un è del non le si con i da una dei più al come ma lo gli nel alla',
+ nl: 'de van het een en in is dat op te zijn voor met die niet aan er om ook als dan maar bij hij',
+ sv: 'och i att det som en på är av för med till den har de inte om ett han men var jag sig från',
+ fi: 'ja on ei että oli hän se en mutta niin kun kuin joka hänen ole sen olen mitä minä jos nyt vain',
+ pl: 'i w nie na z że się do to jest jak a o po ale co tak za od go był przez już tylko jego',
+ tr: 'bir ve bu da de için ile ne o ben gibi çok daha ama kadar sonra en var mi diye her olan',
+ ru: 'и в не на я что он с как а то это все она так его но да ты к у же вы за бы по только',
+ uk: 'і в не на я що він з як а то це все вона так його але та ти до у ж ви за би по тільки',
+};
+
+/** Best guess at a Whisper language code for `paragraphs`, or null. */
+export function detectLanguage(paragraphs) {
+ const sample = paragraphs.join(' ').slice(0, 60000);
+ const count = (re) => (sample.match(re) || []).length;
+ const kana = count(/[-ヿ]/g), han = count(/[一-鿿]/g), hangul = count(/[가-]/g);
+ if (kana > 50) return 'ja';
+ if (hangul > 50) return 'ko';
+ if (han > 200) return 'zh';
+ if (count(/[Ͱ-Ͽ]/g) > 200) return 'el';
+ if (count(/[-]/g) > 200) return 'he';
+ if (count(/[-ۿ]/g) > 200) return 'ar';
+ if (count(/[-]/g) > 200) return 'th';
+
+ const words = sample.toLowerCase().match(/\p{L}+/gu) || [];
+ if (words.length < 50) return null;
+ const tally = new Map();
+ for (const w of words) tally.set(w, (tally.get(w) || 0) + 1);
+ let best = null, bestScore = 0;
+ for (const [code, list] of Object.entries(STOPWORDS)) {
+ const score = list.split(' ').reduce((n, w) => n + (tally.get(w) || 0), 0);
+ if (score > bestScore) { best = code; bestScore = score; }
+ }
+ // Fewer than one word in twelve a stopword of the winner: not one of ours.
+ return bestScore / words.length > 0.08 ? best : null;
+}
diff --git a/frontend/engine/job.js b/frontend/engine/job.js
new file mode 100644
index 0000000..c2c6c16
--- /dev/null
+++ b/frontend/engine/job.js
@@ -0,0 +1,234 @@
+/* One conversion, start to finish, in this tab.
+ *
+ * audio --ffmpeg--> two-minute chunks --whisper--> rough transcript
+ * book ----------> paragraphs
+ * transcript + paragraphs --align--> subtitles (.srt)
+ * cover + audio (+ subtitles) --ffmpeg--> video, on request
+ *
+ * Nothing leaves the machine. The transcript is saved to IndexedDB after every
+ * chunk, because transcribing a book takes hours and tabs get closed: opening
+ * the same files again carries on from the last chunk, and aligning the same
+ * audio against a different edition of the book costs seconds.
+ */
+import { Media, quietestPoint, SAMPLE_RATE } from './media.js';
+import { Recogniser } from './asr.js';
+import { readBook } from './book.js';
+
+const CHUNK_SECONDS = 120;
+
+export class Cancelled extends Error { constructor() { super('Stopped.'); } }
+
+/* ------------------------------------------------------------ saved progress */
+
+const db = () => new Promise((resolve, reject) => {
+ const open = indexedDB.open('subread', 1);
+ open.onupgradeneeded = () => open.result.createObjectStore('transcripts');
+ open.onsuccess = () => resolve(open.result);
+ open.onerror = () => reject(open.error);
+});
+
+async function store(mode, fn) {
+ const d = await db();
+ return new Promise((resolve, reject) => {
+ const tx = d.transaction('transcripts', mode);
+ const req = fn(tx.objectStore('transcripts'));
+ tx.oncomplete = () => resolve(req?.result);
+ tx.onerror = () => reject(tx.error);
+ });
+}
+
+/** The same files picked again resume, whatever route they were picked by. */
+const keyFor = (files, language) => files.map((f) => `${f.name}|${f.size}`).join('//') + `#${language}`;
+
+export async function savedProgress(files, language) {
+ try { return (await store('readonly', (s) => s.get(keyFor(files, language)))) ?? null; } catch { return null; }
+}
+
+/* ------------------------------------------------------------------- the job */
+
+export class Job {
+ #media = null;
+ #paths = [];
+ #cancelled = false;
+ #book = null;
+
+ constructor({ audio, book, language, onStatus }) {
+ this.audio = audio; // File[], in playback order
+ this.bookFile = book; // File
+ this.language = language; // Whisper code; always explicit, see asr.js
+ this.onStatus = onStatus ?? (() => {});
+ this.result = null;
+ }
+
+ cancel() { this.#cancelled = true; }
+ #check() { if (this.#cancelled) throw new Cancelled(); }
+ #status(phase, detail, fraction, extra = {}) { this.onStatus({ phase, detail, fraction, ...extra }); }
+
+ async run() {
+ this.#status('preparing', 'Reading the book', 0);
+ this.#book = await readBook(this.bookFile);
+ if (!this.#book.paragraphs.length) throw new Error(`No text could be read from ${this.bookFile.name}.`);
+
+ this.#status('preparing', 'Starting the audio decoder', 0);
+ this.#media = await Media.open();
+ const parts = [];
+ for (const [i, file] of this.audio.entries()) {
+ const path = await this.#media.mount(file, `/in${i}`);
+ parts.push({ path, ...(await this.#media.probe(path)) });
+ this.#paths.push(path);
+ }
+ const total = parts.reduce((n, p) => n + p.duration, 0);
+
+ const transcript = await this.#transcribe(parts, total);
+ this.#check();
+
+ this.#status('aligning', 'Matching the book to the narration', 0.97);
+ const aligned = await alignInWorker(transcript, this.#book.paragraphs, this.language);
+ const stem = (this.audio.length > 1 ? this.bookFile : this.audio[0]).name.replace(/\.[^.]+$/, '');
+ this.result = {
+ ...aligned, stem, duration: total, parts,
+ srtName: `${stem}.${this.language}.srt`,
+ segments: transcript.length,
+ };
+ this.#status('done', 'Done', 1);
+ return this.result;
+ }
+
+ async #transcribe(parts, total) {
+ const key = keyFor(this.audio, this.language);
+ const saved = (await savedProgress(this.audio, this.language)) ?? { doneUntil: 0, segments: [], complete: false };
+ if (saved.complete) return saved.segments;
+
+ this.#status('preparing', 'Loading the speech model', 0, { indeterminate: true });
+ const asr = await Recogniser.open((p) => {
+ if (p.total) this.#status('preparing', 'Downloading the speech model (once)', p.loaded / p.total * 0.02);
+ });
+ this.device = asr.device;
+
+ const resumedAt = saved.doneUntil, started = performance.now();
+ let base = 0; // where the current part starts on the whole book's clock
+ for (const part of parts) {
+ let pos = Math.max(0, saved.doneUntil - base);
+ while (pos < part.duration - 0.05) {
+ this.#check();
+ let pcm = await this.#media.pcm(part.path, pos, CHUNK_SECONDS);
+ if (!pcm.length) break;
+ const last = pos + pcm.length / SAMPLE_RATE >= part.duration - 0.5;
+ // Cut where the narrator pauses, so no word straddles two chunks.
+ if (!last) pcm = pcm.subarray(0, quietestPoint(pcm));
+
+ const at = base + pos;
+ saved.segments.push(...await asr.transcribe(pcm, this.language, at));
+ pos += pcm.length / SAMPLE_RATE;
+ saved.doneUntil = base + pos;
+ await store('readwrite', (s) => s.put(saved, key)).catch(() => {});
+
+ const elapsed = (performance.now() - started) / 1000;
+ const speed = (saved.doneUntil - resumedAt) / elapsed;
+ this.#status('transcribing', `Listening: ${clock(saved.doneUntil)} of ${clock(total)}`,
+ 0.02 + 0.94 * saved.doneUntil / total,
+ { eta: elapsed > 10 ? (total - saved.doneUntil) / speed : null, speed: elapsed > 10 ? speed : null });
+ }
+ base += part.duration;
+ }
+ saved.complete = true;
+ await store('readwrite', (s) => s.put(saved, key)).catch(() => {});
+ return saved.segments;
+ }
+
+ /**
+ * The video, made on request: a still image over the audio.
+ * kind "mkv" carries the subtitles inside the file (MPV, VLC);
+ * kind "mp4" is clean, for YouTube, where the .srt is uploaded alongside.
+ *
+ * A still picture does not need encoding for ten hours. One minute of it is
+ * encoded once and then repeated by stream copy, so the cost is that of
+ * copying the audio - seconds, not an hour.
+ */
+ async video(kind, { coverFile, onProgress } = {}) {
+ const m = this.#media, r = this.result;
+ const still = await canvasPng(coverFile ?? this.#book.cover);
+ await m.write('canvas.png', still);
+ await m.run(['-loop', '1', '-framerate', '1', '-i', 'canvas.png', '-t', '60',
+ '-c:v', 'libx264', '-preset', 'veryfast', '-tune', 'stillimage', '-pix_fmt', 'yuv420p',
+ '-r', '1', '-g', '60', 'still.mp4']);
+
+ let audioIn;
+ if (this.#paths.length === 1) audioIn = ['-i', this.#paths[0]];
+ else {
+ const list = this.#paths.map((p) => `file '${p.replaceAll("'", "'\\''")}'`).join('\n');
+ await m.write('parts.txt', new TextEncoder().encode(list));
+ audioIn = ['-f', 'concat', '-safe', '0', '-i', 'parts.txt'];
+ }
+ // Copy the audio when the container can hold it; re-encoding a whole book
+ // in WebAssembly is the one slow thing here, so only when there is no choice.
+ const copyable = kind === 'mkv'
+ ? ['aac', 'mp3', 'opus', 'vorbis', 'flac', 'ac3', 'alac']
+ : ['aac', 'mp3', 'ac3', 'alac'];
+ const codecs = new Set(r.parts.map((p) => p.codec));
+ const audioCodec = codecs.size === 1 && copyable.includes([...codecs][0]) ? ['-c:a', 'copy'] : ['-c:a', 'aac', '-b:a', '128k'];
+
+ const out = `out.${kind}`;
+ const args = ['-stream_loop', '-1', '-i', 'still.mp4', ...audioIn];
+ if (kind === 'mkv') {
+ await m.write('subs.srt', new TextEncoder().encode(r.srt));
+ args.push('-f', 'srt', '-i', 'subs.srt', '-map', '0:v', '-map', '1:a:0', '-map', '2:s',
+ '-c:v', 'copy', ...audioCodec, '-c:s', 'srt', '-disposition:s:0', 'default',
+ '-metadata:s:s:0', 'title=Aligned subtitles');
+ } else {
+ args.push('-map', '0:v', '-map', '1:a:0', '-c:v', 'copy', ...audioCodec);
+ }
+ args.push('-map_chapters', '1', '-t', r.duration.toFixed(3), out);
+
+ try {
+ await m.run(args, onProgress);
+ const bytes = await m.read(out);
+ return new File([bytes], `${r.stem}.${this.language}.${kind}`,
+ { type: kind === 'mkv' ? 'video/x-matroska' : 'video/mp4' });
+ } finally {
+ for (const f of [out, 'still.mp4', 'canvas.png', 'subs.srt', 'parts.txt']) await m.remove(f);
+ }
+ }
+
+ close() { this.#media?.close(); }
+}
+
+function alignInWorker(transcript, paragraphs, language) {
+ return new Promise((resolve, reject) => {
+ const worker = new Worker(new URL('./align.worker.js', import.meta.url), { type: 'module' });
+ worker.onmessage = ({ data }) => {
+ worker.terminate();
+ if (data.ok) resolve(data); else reject(new Error(data.error));
+ };
+ worker.onerror = (e) => { worker.terminate(); reject(new Error(e.message || 'The aligner crashed.')); };
+ worker.postMessage({ transcript, paragraphs, language });
+ });
+}
+
+/** The cover, letterboxed onto a 1920x1080 card; a plain dark card when there is none. */
+async function canvasPng(cover) {
+ const W = 1920, H = 1080;
+ const canvas = new OffscreenCanvas(W, H);
+ const ctx = canvas.getContext('2d');
+ ctx.fillStyle = '#16161a';
+ ctx.fillRect(0, 0, W, H);
+ if (cover) {
+ try {
+ const blob = cover instanceof Blob ? cover : new Blob([cover.bytes]);
+ const img = await createImageBitmap(blob);
+ const scale = Math.min(W / img.width, H / img.height);
+ const w = img.width * scale, h = img.height * scale;
+ ctx.fillStyle = '#000';
+ ctx.fillRect(0, 0, W, H);
+ ctx.drawImage(img, (W - w) / 2, (H - h) / 2, w, h);
+ } catch { /* an image the browser cannot decode: keep the plain card */ }
+ }
+ const png = await canvas.convertToBlob({ type: 'image/png' });
+ return new Uint8Array(await png.arrayBuffer());
+}
+
+export function clock(seconds) {
+ const s = Math.max(0, Math.floor(seconds)), pad = (v) => String(v).padStart(2, '0');
+ return s >= 3600 ? `${Math.floor(s / 3600)}:${pad(Math.floor(s / 60) % 60)}:${pad(s % 60)}`
+ : `${Math.floor(s / 60)}:${pad(s % 60)}`;
+}
diff --git a/frontend/engine/media.js b/frontend/engine/media.js
new file mode 100644
index 0000000..7ad378c
--- /dev/null
+++ b/frontend/engine/media.js
@@ -0,0 +1,111 @@
+/* Reading and writing media in the browser, with ffmpeg compiled to WebAssembly.
+ *
+ * The visitor's file is never copied anywhere: it is mounted read-only into
+ * ffmpeg's filesystem (WORKERFS), which reads straight from the File on disk.
+ * That is what makes a 600 MB audiobook workable - the only audio ever held in
+ * memory is the couple of minutes being transcribed.
+ *
+ * ffmpeg runs in its own worker; everything here is async message-passing.
+ */
+import { FFmpeg } from '/vendor/ffmpeg/index.js';
+
+const CORE = new URL('/vendor/ffmpeg/ffmpeg-core.js', import.meta.url).href;
+const RATE = 16000;
+
+export class Media {
+ #ffmpeg = new FFmpeg();
+ #log = [];
+ #mounted = [];
+
+ /** Bytes fetched so far while the 32 MB core downloads, for a progress bar. */
+ static async open(onLog) {
+ const m = new Media();
+ m.#ffmpeg.on('log', ({ message }) => {
+ m.#log.push(message);
+ if (m.#log.length > 400) m.#log.shift();
+ onLog?.(message);
+ });
+ await m.#ffmpeg.load({ coreURL: CORE });
+ return m;
+ }
+
+ /** Make `file` readable by ffmpeg at the returned path, without copying it. */
+ async mount(file, dir) {
+ await this.#ffmpeg.createDir(dir).catch(() => {});
+ await this.#ffmpeg.mount('WORKERFS', { files: [file] }, dir);
+ this.#mounted.push(dir);
+ return `${dir}/${file.name}`;
+ }
+
+ /** Duration in seconds and chapter marks, read from ffmpeg's own banner. */
+ async probe(path) {
+ this.#log.length = 0;
+ await this.#ffmpeg.exec(['-hide_banner', '-i', path]); // "fails": no output given. Expected.
+ const text = this.#log.join('\n');
+ const d = /Duration:\s*(\d+):(\d+):(\d+(?:\.\d+)?)/.exec(text);
+ if (!d) throw new Error('This file does not look like audio ffmpeg can read.');
+ const audio = /Stream #\d+:\d+.*?: Audio: ([a-z0-9_]+)/i.exec(text);
+ return {
+ duration: (+d[1]) * 3600 + (+d[2]) * 60 + (+d[3]),
+ codec: audio ? audio[1].toLowerCase() : null,
+ chapters: [...text.matchAll(/Chapter #\d+:\d+: start ([\d.]+), end ([\d.]+)/g)]
+ .map((c) => ({ start: +c[1], end: +c[2] })),
+ };
+ }
+
+ /** `seconds` of audio from `start`, as the 16 kHz mono float PCM the speech model takes. */
+ async pcm(path, start, seconds) {
+ const out = 'chunk.f32';
+ const code = await this.#ffmpeg.exec([
+ '-hide_banner', '-v', 'error',
+ // Tolerate damaged files: drop what cannot be decoded, keep going.
+ '-err_detect', 'ignore_err', '-fflags', '+discardcorrupt',
+ '-ss', String(start), '-t', String(seconds), '-i', path,
+ '-map', '0:a:0', '-vn', '-sn', '-ac', '1', '-ar', String(RATE), '-f', 'f32le', out,
+ ]);
+ if (code !== 0) throw new Error(`Could not decode the audio at ${Math.round(start)}s.`);
+ const bytes = await this.#ffmpeg.readFile(out);
+ await this.#ffmpeg.deleteFile(out);
+ return new Float32Array(bytes.buffer, bytes.byteOffset, bytes.byteLength >> 2);
+ }
+
+ async write(name, data) { await this.#ffmpeg.writeFile(name, data); }
+ async read(name) { return this.#ffmpeg.readFile(name); }
+ async remove(name) { await this.#ffmpeg.deleteFile(name).catch(() => {}); }
+
+ /** Run ffmpeg; throws with the tail of its log when it fails. */
+ async run(args, onProgress) {
+ this.#log.length = 0;
+ const handler = onProgress && (({ progress }) => onProgress(Math.min(Math.max(progress, 0), 1)));
+ if (handler) this.#ffmpeg.on('progress', handler);
+ try {
+ const code = await this.#ffmpeg.exec(['-hide_banner', '-y', ...args]);
+ if (code !== 0) throw new Error(this.#log.slice(-3).join(' | ') || `ffmpeg exited with ${code}`);
+ } finally {
+ if (handler) this.#ffmpeg.off('progress', handler);
+ }
+ }
+
+ close() { this.#ffmpeg.terminate(); }
+}
+
+/**
+ * Where to cut a chunk: the quietest quarter second in its last few seconds,
+ * so that no word is split across two transcriptions.
+ * Returns a sample index into `pcm`.
+ */
+export function quietestPoint(pcm, searchSeconds = 12, windowSeconds = 0.25) {
+ const window = Math.floor(windowSeconds * RATE);
+ const from = Math.max(0, pcm.length - searchSeconds * RATE);
+ if (pcm.length - from <= window * 2) return pcm.length;
+ let energy = 0;
+ for (let i = from; i < from + window; i++) energy += Math.abs(pcm[i]);
+ let best = energy, bestAt = from;
+ for (let i = from; i + window < pcm.length; i++) {
+ energy += Math.abs(pcm[i + window]) - Math.abs(pcm[i]);
+ if (energy < best) { best = energy; bestAt = i + 1; }
+ }
+ return bestAt + (window >> 1);
+}
+
+export const SAMPLE_RATE = RATE;
diff --git a/frontend/lab.html b/frontend/lab.html
new file mode 100644
index 0000000..24bd965
--- /dev/null
+++ b/frontend/lab.html
@@ -0,0 +1,93 @@
+<!DOCTYPE html>
+<meta charset="utf-8">
+<title>engine lab</title>
+<body style="font:14px/1.5 ui-monospace,monospace;padding:20px;max-width:900px">
+<h3>Browser engine lab</h3>
+<p>Proves the client-side pieces one at a time. Not linked from the site.</p>
+<input type="file" id="f"> <button id="go">run on /vendor/lab/ru.mp3</button>
+<pre id="out"></pre>
+<script type="module">
+import { Media, quietestPoint } from '/engine/media.js';
+import { Recogniser, hasWebGpu } from '/engine/asr.js';
+
+const out = document.getElementById('out');
+const log = (...a) => { out.textContent += a.join(' ') + '\n'; };
+window.lab = { done: false, error: null, results: {} };
+
+async function run(file) {
+ try {
+ log('crossOriginIsolated:', crossOriginIsolated, '| threads:', navigator.hardwareConcurrency,
+ '| webgpu:', await hasWebGpu());
+ let t = performance.now();
+ const media = await Media.open();
+ log(`ffmpeg loaded in ${((performance.now() - t) / 1000).toFixed(1)}s`);
+
+ const path = await media.mount(file, '/in');
+ const info = await media.probe(path);
+ log('probe:', JSON.stringify(info));
+
+ t = performance.now();
+ const pcm = await media.pcm(path, 0, 60);
+ log(`decoded ${(pcm.length / 16000).toFixed(1)}s of audio in ${((performance.now() - t) / 1000).toFixed(2)}s;`,
+ 'cut at', (quietestPoint(pcm) / 16000).toFixed(2) + 's');
+
+ t = performance.now();
+ const asr = await Recogniser.open((p) => { window.lab.results.download = `${p.file} ${Math.round(p.progress)}%`; });
+ log(`model ready on ${asr.device} in ${((performance.now() - t) / 1000).toFixed(1)}s`);
+
+ const lang = new URLSearchParams(location.search).get('lang') || null;
+ t = performance.now();
+ await asr.transcribe(pcm.subarray(0, 16000 * 20), lang, 0);
+ log(`warm-up pass (shader compile) ${((performance.now() - t) / 1000).toFixed(1)}s, language=${lang}`);
+ t = performance.now();
+ const segments = await asr.transcribe(pcm, lang, 0);
+ const took = (performance.now() - t) / 1000;
+ log(`transcribed in ${took.toFixed(1)}s = ${(pcm.length / 16000 / took).toFixed(1)}x real time, ${segments.length} segments`);
+ for (const s of segments.slice(0, 8)) log(` ${s.start.toFixed(2)}-${s.end.toFixed(2)} ${s.text}`);
+ window.lab.results = { info, device: asr.device, took, segments };
+ } catch (e) {
+ log('FAILED:', e?.stack || e);
+ window.lab.error = String(e?.message || e);
+ }
+ window.lab.done = true;
+}
+
+document.getElementById('go').onclick = async () => {
+ const blob = await (await fetch('/vendor/lab/ru.mp3')).blob();
+ run(new File([blob], 'ru.mp3'));
+};
+// The whole job: decode, transcribe, align, then both videos.
+window.fullJob = async () => {
+ window.lab = { done: false, error: null, results: {} };
+ out.textContent = '';
+ try {
+ const { Job } = await import('/engine/job.js');
+ const { detectLanguage, readBook } = await import('/engine/book.js');
+ const get = async (n) => new File([await (await fetch('/vendor/lab/' + n)).blob()], n);
+ const audio = await get('ru.mp3'), book = await get('moskva.epub');
+ const parsed = await readBook(book);
+ const lang = detectLanguage(parsed.paragraphs);
+ log(`book: ${parsed.paragraphs.length} paragraphs, cover ${parsed.cover ? parsed.cover.name : 'none'}, detected language ${lang}`);
+ let last = '';
+ const job = new Job({ audio: [audio], book, language: lang, onStatus: (st) => {
+ const line = `${st.phase}: ${st.detail}`;
+ if (line !== last) { last = line; log(' ' + line, st.speed ? `(${st.speed.toFixed(1)}x)` : ''); }
+ }});
+ let t = performance.now();
+ const r = await job.run();
+ log(`job done in ${((performance.now() - t) / 1000).toFixed(1)}s on ${job.device}: ${r.cues.length} cues, ` +
+ `${(r.matchRate * 100).toFixed(0)}% matched, ${r.paragraphsDropped}/${r.paragraphsDropped + r.paragraphsUsed} paragraphs dropped`);
+ log(r.srt.split('\n').slice(0, 16).join('\n'));
+ for (const kind of ['mkv', 'mp4']) {
+ t = performance.now();
+ const file = await job.video(kind);
+ log(`${kind}: ${file.name} ${(file.size / 1e6).toFixed(2)} MB in ${((performance.now() - t) / 1000).toFixed(1)}s`);
+ window.lab.results[kind] = file;
+ }
+ window.lab.results.srt = r.srt;
+ } catch (e) { log('FAILED:', e?.stack || e); window.lab.error = String(e?.message || e); }
+ window.lab.done = true;
+};
+
+document.getElementById('f').onchange = (e) => run(e.target.files[0]);
+</script>
diff --git a/tools/fetch_vendor.py b/tools/fetch_vendor.py
new file mode 100644
index 0000000..1cdf69a
--- /dev/null
+++ b/tools/fetch_vendor.py
@@ -0,0 +1,88 @@
+"""Download the browser-side libraries into frontend/vendor/.
+
+Processing happens in the visitor's browser, which needs three things that are
+far too big to commit: ffmpeg compiled to WebAssembly (reads any audio format,
+muxes the video), the ONNX runtime, and transformers.js to drive Whisper on it.
+They are pinned here by exact version and unpacked from the npm registry, then
+served from our own origin - no CDN in the page, so nothing third-party has to
+be trusted or kept alive, and cross-origin isolation stays simple.
+
+ python tools/fetch_vendor.py # idempotent; run on every deploy
+
+Standard library only, on purpose: this runs on the server before anything else
+is installed.
+"""
+
+from __future__ import annotations
+
+import hashlib
+import io
+import json
+import sys
+import tarfile
+import urllib.request
+from pathlib import Path
+
+VENDOR = Path(__file__).resolve().parent.parent / "frontend" / "vendor"
+
+# package, version, {path inside the tarball: path under vendor/}
+PACKAGES = [
+ ("@ffmpeg/ffmpeg", "0.12.15", {
+ f"package/dist/esm/{name}": f"ffmpeg/{name}"
+ for name in ("index.js", "classes.js", "const.js", "errors.js", "types.js", "utils.js", "worker.js")
+ }),
+ ("@ffmpeg/core", "0.12.10", {
+ "package/dist/esm/ffmpeg-core.js": "ffmpeg/ffmpeg-core.js",
+ "package/dist/esm/ffmpeg-core.wasm": "ffmpeg/ffmpeg-core.wasm",
+ }),
+ ("@huggingface/transformers", "4.3.0", {
+ "package/dist/transformers.min.js": "transformers/transformers.min.js",
+ }),
+ # Must be exactly the build transformers.js was made against.
+ ("onnxruntime-web", "1.31.0-dev.20260914-8d85527a0", {
+ f"package/dist/{name}": f"transformers/{name}"
+ for name in ("ort.webgpu.bundle.min.mjs", "ort-wasm-simd-threaded.asyncify.mjs",
+ "ort-wasm-simd-threaded.asyncify.wasm")
+ }),
+]
+
+
+def fetch(package: str, version: str) -> bytes:
+ meta_url = f"https://registry.npmjs.org/{package.replace('/', '%2F')}/{version}"
+ with urllib.request.urlopen(meta_url, timeout=60) as r:
+ dist = json.load(r)["dist"]
+ with urllib.request.urlopen(dist["tarball"], timeout=600) as r:
+ blob = r.read()
+ # The registry's own checksum: a tampered or truncated download stops here.
+ if hashlib.sha1(blob).hexdigest() != dist["shasum"]:
+ sys.exit(f"{package}@{version}: checksum mismatch")
+ return blob
+
+
+def main() -> None:
+ stamp = VENDOR / "versions.json"
+ want = {p: v for p, v, _ in PACKAGES}
+ have = json.loads(stamp.read_text()) if stamp.exists() else {}
+
+ for package, version, files in PACKAGES:
+ targets = [VENDOR / dest for dest in files.values()]
+ if have.get(package) == version and all(t.exists() for t in targets):
+ print(f"ok {package}@{version}")
+ continue
+ print(f"fetching {package}@{version}")
+ with tarfile.open(fileobj=io.BytesIO(fetch(package, version)), mode="r:gz") as tar:
+ for src, dest in files.items():
+ member = tar.extractfile(src)
+ if member is None:
+ sys.exit(f"{package}@{version}: {src} is not in the package")
+ out = VENDOR / dest
+ out.parent.mkdir(parents=True, exist_ok=True)
+ out.write_bytes(member.read())
+
+ stamp.write_text(json.dumps(want, indent=2))
+ total = sum(f.stat().st_size for f in VENDOR.rglob("*") if f.is_file())
+ print(f"vendor/ is {total / 1e6:.0f} MB")
+
+
+if __name__ == "__main__":
+ main()