Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


commit 430d2318735dbc0aeb73d3c73180634dbe7244c6
equwal <truex@equwal.com>
2026-09-21 16:41:28 -0700

Serve the speech model from our origin, and run without the API

The page fetched the Whisper-tiny weights from huggingface.co. Now
tools/fetch_vendor.py downloads them once, checks their sha256, and puts
them under frontend/vendor/models/. The page loads them from there and
never contacts another host.

The Start button waited for POST /api/local/jobs and stopped on an error.
The server only keeps the history of books, so the page now starts the
job without an answer, and reports the result only when there is a job
to report to. The whole conversion works on static hosting with no API.

 backend/main.py        |  5 ++--
 frontend/app.js        | 70 +++++++++++++++++++++++---------------------------
 frontend/engine/asr.js | 12 ++++++---
 frontend/index.html    |  4 +--
 tools/fetch_vendor.py  | 50 +++++++++++++++++++++++++++++++-----
 5 files changed, 89 insertions(+), 52 deletions(-)
diff --git a/backend/main.py b/backend/main.py
index b6efead..80063b7 100644
--- a/backend/main.py
+++ b/backend/main.py
@@ -130,8 +130,9 @@ async def cross_origin_isolation(request, call_next):
 
     The speech model runs in the browser on WebAssembly threads, which need
     SharedArrayBuffer, which browsers only hand to a cross-origin-isolated
-    page. "credentialless" rather than "require-corp": the model weights come
-    from a third-party host that sends CORS headers but not CORP ones.
+    page. Every file the page loads, the model weights included, comes from
+    this origin. "credentialless" rather than "require-corp" so that a
+    cross-origin image or link a visitor adds later does not break the page.
     """
     response = await call_next(request)
     response.headers["Cross-Origin-Opener-Policy"] = "same-origin"
diff --git a/frontend/app.js b/frontend/app.js
index 973675e..93e127d 100644
--- a/frontend/app.js
+++ b/frontend/app.js
@@ -427,28 +427,28 @@ el.start.addEventListener('click', async () => {
   clearError();
   const language = el.language.value;
 
-  // The server keeps the list of this visitor's books; a job in this tab costs nothing.
-  let registered;
-  try {
-    registered = await api('/api/local/jobs', {
-      method: 'POST',
-      headers: { 'Content-Type': 'application/json' },
-      body: JSON.stringify({
-        audio_filename: draft.audio.length > 1 ? `${draft.audio[0].name} + ${draft.audio.length - 1} more` : draft.audio[0].name,
-        audio_parts: draft.audio.length,
-        audio_bytes: draft.audio.reduce((n, f) => n + f.size, 0),
-        text_filename: draft.book.name, language,
-      }),
-    });
-  } catch (e) {
-    showError(e.message);
-    el.start.disabled = false;
-    return;
-  }
+  // The server keeps the list of this visitor's books, and nothing more: a job
+  // in this tab costs nothing and needs nothing from it. A server that is down
+  // or missing only loses the history entry.
+  const registered = await api('/api/local/jobs', {
+    method: 'POST',
+    headers: { 'Content-Type': 'application/json' },
+    body: JSON.stringify({
+      audio_filename: draft.audio.length > 1 ? `${draft.audio[0].name} + ${draft.audio.length - 1} more` : draft.audio[0].name,
+      audio_parts: draft.audio.length,
+      audio_bytes: draft.audio.reduce((n, f) => n + f.size, 0),
+      text_filename: draft.book.name, language,
+    }),
+  }).catch(() => null);
+  const report = (path, body) => (registered
+    ? api(`/api/local/jobs/${registered.id}/${path}`, {
+      method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(body),
+    }).catch(() => {})
+    : Promise.resolve());
 
   const { Job, Cancelled } = await import('/engine/job.js');
   const job = new Job({ audio: draft.audio, book: draft.book, language, onStatus: renderWorking });
-  running = { job, serverId: registered.id, cover: draft.cover };
+  running = { job, serverId: registered?.id, cover: draft.cover };
   el.confirm.hidden = true;
   el.working.hidden = false;
   el.results.hidden = true;
@@ -459,28 +459,22 @@ el.start.addEventListener('click', async () => {
 
   try {
     const result = await job.run();
-    await api(`/api/local/jobs/${registered.id}/finish`, {
-      method: 'POST',
-      headers: { 'Content-Type': 'application/json' },
-      body: JSON.stringify({
-        srt: result.srt, filename: result.srtName,
-        metadata: {
-          source: { audio_parts: draft.audio.length, text_filename: draft.book.name,
-                    audio_duration_seconds: result.duration },
-          alignment: { where: 'in the browser', device: job.device ?? 'cached transcript', model: 'whisper-tiny',
-                       language, mode: 'forced alignment against supplied text' },
-          output: { filename: result.srtName, cue_count: result.cues.length,
-                    match_rate: result.matchRate, paragraphs_dropped: result.paragraphsDropped },
-        },
-      }),
-    }).catch(() => { /* the subtitles are still here to download; only the history entry is missing */ });
+    // The subtitles are here to download whatever the server says; only the history entry can go missing.
+    await report('finish', {
+      srt: result.srt, filename: result.srtName,
+      metadata: {
+        source: { audio_parts: draft.audio.length, text_filename: draft.book.name,
+                  audio_duration_seconds: result.duration },
+        alignment: { where: 'in the browser', device: job.device ?? 'cached transcript', model: 'whisper-tiny',
+                     language, mode: 'forced alignment against supplied text' },
+        output: { filename: result.srtName, cue_count: result.cues.length,
+                  match_rate: result.matchRate, paragraphs_dropped: result.paragraphsDropped },
+      },
+    });
     showResults(result);
   } catch (e) {
     const stopped = e instanceof Cancelled;
-    await api(`/api/local/jobs/${registered.id}/fail`, {
-      method: 'POST', headers: { 'Content-Type': 'application/json' },
-      body: JSON.stringify({ error: stopped ? 'Stopped.' : String(e.message || e) }),
-    }).catch(() => {});
+    await report('fail', { error: stopped ? 'Stopped.' : String(e.message || e) });
     job.close();
     running = null;
     el.working.hidden = true;
diff --git a/frontend/engine/asr.js b/frontend/engine/asr.js
index 6a48b9d..3a42bb4 100644
--- a/frontend/engine/asr.js
+++ b/frontend/engine/asr.js
@@ -7,11 +7,15 @@
  */
 import { pipeline, env } from '/vendor/transformers/transformers.min.js';
 
-// Everything from our own origin except the model weights themselves.
+// Everything from our own origin, the model weights included: tools/fetch_vendor.py
+// puts them under /vendor/models/, and nothing is fetched from anywhere else.
 env.backends.onnx.wasm.wasmPaths = new URL('/vendor/transformers/', import.meta.url).href;
-env.allowLocalModels = false;
+// A path, not a full URL: transformers.js treats an http(s) URL as remote.
+env.localModelPath = '/vendor/models/';
+env.allowLocalModels = true;
+env.allowRemoteModels = false;
 
-const MODEL = 'onnx-community/whisper-tiny';
+const MODEL = 'whisper-tiny';
 
 export async function hasWebGpu() {
   try { return !!(navigator.gpu && await navigator.gpu.requestAdapter()); } catch { return false; }
@@ -21,7 +25,7 @@ export class Recogniser {
   #pipe;
   device;
 
-  /** @param onProgress ({file, loaded, total}) while the ~40 MB of weights download (cached after). */
+  /** @param onProgress ({file, loaded, total}) while the ~120 MB of weights download (cached after). */
   static async open(onProgress) {
     const r = new Recogniser();
     r.device = (await hasWebGpu()) ? 'webgpu' : 'wasm';
diff --git a/frontend/index.html b/frontend/index.html
index 67950aa..8a252f1 100644
--- a/frontend/index.html
+++ b/frontend/index.html
@@ -4,7 +4,7 @@
 <meta charset="utf-8">
 <meta name="viewport" content="width=device-width, initial-scale=1">
 <title>SubRead — read along with your audiobook</title>
-<link rel="stylesheet" href="/style.css?v=10">
+<link rel="stylesheet" href="/style.css?v=11">
 <link rel="icon" href="data:image/svg+xml,<svg xmlns='http://www.w3.org/2000/svg' viewBox='0 0 100 100'><text y='.9em' font-size='90'>🎧</text></svg>">
 </head>
 <body>
@@ -233,6 +233,6 @@
 
 <div id="toast" class="toast" hidden role="status"></div>
 
-<script type="module" src="/app.js?v=10"></script>
+<script type="module" src="/app.js?v=11"></script>
 </body>
 </html>
diff --git a/tools/fetch_vendor.py b/tools/fetch_vendor.py
index 1cdf69a..f18e3f7 100644
--- a/tools/fetch_vendor.py
+++ b/tools/fetch_vendor.py
@@ -1,11 +1,12 @@
-"""Download the browser-side libraries into frontend/vendor/.
+"""Download the browser-side libraries and the speech model into frontend/vendor/.
 
-Processing happens in the visitor's browser, which needs three things that are
+Processing happens in the visitor's browser, which needs four things that are
 far too big to commit: ffmpeg compiled to WebAssembly (reads any audio format,
-muxes the video), the ONNX runtime, and transformers.js to drive Whisper on it.
-They are pinned here by exact version and unpacked from the npm registry, then
-served from our own origin - no CDN in the page, so nothing third-party has to
-be trusted or kept alive, and cross-origin isolation stays simple.
+muxes the video), the ONNX runtime, transformers.js to drive Whisper on it, and
+the Whisper-tiny weights themselves. They are pinned here by exact version or
+hash, taken from the npm registry and the Hugging Face hub, then served from
+our own origin - no CDN and no third-party host in the page, so nothing outside
+has to be trusted or kept alive, and cross-origin isolation stays simple.
 
     python tools/fetch_vendor.py          # idempotent; run on every deploy
 
@@ -47,6 +48,25 @@ PACKAGES = [
 ]
 
 
+# The speech model, as transformers.js loads it: the config and tokenizer files,
+# the full-precision encoder, and one decoder for each device (4-bit on WebGPU,
+# 8-bit on WebAssembly). The hash is the sha256 of the file on the hub, so a
+# changed or truncated download stops here.
+MODEL_REPO = "onnx-community/whisper-tiny"
+MODEL_REVISION = "main"
+MODEL_DIR = "models/whisper-tiny"
+MODEL_FILES = {
+    "config.json": None,
+    "preprocessor_config.json": None,
+    "tokenizer_config.json": None,
+    "generation_config.json": None,
+    "tokenizer.json": None,
+    "onnx/encoder_model.onnx": "6642befb640f950d",
+    "onnx/decoder_model_merged_q4.onnx": "a7573efde84f7d01",
+    "onnx/decoder_model_merged_quantized.onnx": "25e807a962b63493",
+}
+
+
 def fetch(package: str, version: str) -> bytes:
     meta_url = f"https://registry.npmjs.org/{package.replace('/', '%2F')}/{version}"
     with urllib.request.urlopen(meta_url, timeout=60) as r:
@@ -59,6 +79,23 @@ def fetch(package: str, version: str) -> bytes:
     return blob
 
 
+def fetch_model() -> None:
+    """The Whisper weights, once. A file whose hash matches is not fetched again."""
+    for name, prefix in MODEL_FILES.items():
+        out = VENDOR / MODEL_DIR / name
+        if out.exists() and (prefix is None or hashlib.sha256(out.read_bytes()).hexdigest().startswith(prefix)):
+            print(f"ok       {MODEL_REPO}/{name}")
+            continue
+        print(f"fetching {MODEL_REPO}/{name}")
+        url = f"https://huggingface.co/{MODEL_REPO}/resolve/{MODEL_REVISION}/{name}"
+        with urllib.request.urlopen(url, timeout=600) as r:
+            blob = r.read()
+        if prefix is not None and not hashlib.sha256(blob).hexdigest().startswith(prefix):
+            sys.exit(f"{MODEL_REPO}/{name}: checksum mismatch")
+        out.parent.mkdir(parents=True, exist_ok=True)
+        out.write_bytes(blob)
+
+
 def main() -> None:
     stamp = VENDOR / "versions.json"
     want = {p: v for p, v, _ in PACKAGES}
@@ -80,6 +117,7 @@ def main() -> None:
                 out.write_bytes(member.read())
 
     stamp.write_text(json.dumps(want, indent=2))
+    fetch_model()
     total = sum(f.stat().st_size for f in VENDOR.rglob("*") if f.is_file())
     print(f"vendor/ is {total / 1e6:.0f} MB")