Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


frontend/engine/asr.js (2990 bytes)

1 /* Speech recognition in the browser: Whisper-tiny on the ONNX runtime, through
2  * transformers.js. WebGPU where the browser has it (many times faster than real
3  * time on an ordinary laptop), WebAssembly threads where it does not.
4  *
5  * The transcript only has to be good enough to line up against a book we
6  * already have the words of, which is why the smallest model will do.
7  */
8 import { pipeline, env } from '/vendor/transformers/transformers.min.js';
9 
10 // Everything from our own origin, the model weights included: tools/fetch_vendor.py
11 // puts them under /vendor/models/, and nothing is fetched from anywhere else.
12 env.backends.onnx.wasm.wasmPaths = new URL('/vendor/transformers/', import.meta.url).href;
13 // A path, not a full URL: transformers.js treats an http(s) URL as remote.
14 env.localModelPath = '/vendor/models/';
15 env.allowLocalModels = true;
16 env.allowRemoteModels = false;
17 
18 const MODEL = 'whisper-tiny';
19 
20 export async function hasWebGpu() {
21   try { return !!(navigator.gpu && await navigator.gpu.requestAdapter()); } catch { return false; }
22 }
23 
24 export class Recogniser {
25   #pipe;
26   device;
27 
28   /** @param onProgress ({file, loaded, total}) while the ~120 MB of weights download (cached after). */
29   static async open(onProgress) {
30     const r = new Recogniser();
31     r.device = (await hasWebGpu()) ? 'webgpu' : 'wasm';
32     const options = (device) => ({
33       device,
34       // Full-precision encoder: quantising it is what ruins accuracy. The
35       // decoder tolerates 4-bit well and is where the time goes.
36       dtype: device === 'webgpu'
37         ? { encoder_model: 'fp32', decoder_model_merged: 'q4' }
38         : { encoder_model: 'fp32', decoder_model_merged: 'q8' },
39       progress_callback: (p) => { if (p.status === 'progress') onProgress?.(p); },
40     });
41     try {
42       r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options(r.device));
43     } catch (e) {
44       if (r.device !== 'webgpu') throw e;
45       // A GPU that is listed but cannot run the model: fall back rather than fail.
46       r.device = 'wasm';
47       r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options('wasm'));
48     }
49     return r;
50   }
51 
52   /**
53    * @param pcm 16 kHz mono Float32Array
54    * @param language a Whisper code ("ja", "ru", ...), or null to let it decide
55    * @returns [{text, start, end}] with times offset by `offset` seconds
56    */
57   async transcribe(pcm, language, offset = 0) {
58     const out = await this.#pipe(pcm, {
59       return_timestamps: true,
60       chunk_length_s: 30,
61       stride_length_s: 5,
62       task: 'transcribe',
63       ...(language ? { language } : {}),
64     });
65     const segments = [];
66     for (const c of out.chunks ?? []) {
67       const text = (c.text ?? '').trim();
68       const [s, e] = c.timestamp ?? [];
69       if (!text || s == null) continue;
70       // The last segment of a buffer can come back without an end.
71       segments.push({ text, start: offset + s, end: offset + (e ?? pcm.length / 16000) });
72     }
73     return segments;
74   }
75 }