frontend/engine/asr.js (2990 bytes)
1 /* Speech recognition in the browser: Whisper-tiny on the ONNX runtime, through 2 * transformers.js. WebGPU where the browser has it (many times faster than real 3 * time on an ordinary laptop), WebAssembly threads where it does not. 4 * 5 * The transcript only has to be good enough to line up against a book we 6 * already have the words of, which is why the smallest model will do. 7 */ 8 import { pipeline, env } from '/vendor/transformers/transformers.min.js'; 9 10 // Everything from our own origin, the model weights included: tools/fetch_vendor.py 11 // puts them under /vendor/models/, and nothing is fetched from anywhere else. 12 env.backends.onnx.wasm.wasmPaths = new URL('/vendor/transformers/', import.meta.url).href; 13 // A path, not a full URL: transformers.js treats an http(s) URL as remote. 14 env.localModelPath = '/vendor/models/'; 15 env.allowLocalModels = true; 16 env.allowRemoteModels = false; 17 18 const MODEL = 'whisper-tiny'; 19 20 export async function hasWebGpu() { 21 try { return !!(navigator.gpu && await navigator.gpu.requestAdapter()); } catch { return false; } 22 } 23 24 export class Recogniser { 25 #pipe; 26 device; 27 28 /** @param onProgress ({file, loaded, total}) while the ~120 MB of weights download (cached after). */ 29 static async open(onProgress) { 30 const r = new Recogniser(); 31 r.device = (await hasWebGpu()) ? 'webgpu' : 'wasm'; 32 const options = (device) => ({ 33 device, 34 // Full-precision encoder: quantising it is what ruins accuracy. The 35 // decoder tolerates 4-bit well and is where the time goes. 36 dtype: device === 'webgpu' 37 ? { encoder_model: 'fp32', decoder_model_merged: 'q4' } 38 : { encoder_model: 'fp32', decoder_model_merged: 'q8' }, 39 progress_callback: (p) => { if (p.status === 'progress') onProgress?.(p); }, 40 }); 41 try { 42 r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options(r.device)); 43 } catch (e) { 44 if (r.device !== 'webgpu') throw e; 45 // A GPU that is listed but cannot run the model: fall back rather than fail. 46 r.device = 'wasm'; 47 r.#pipe = await pipeline('automatic-speech-recognition', MODEL, options('wasm')); 48 } 49 return r; 50 } 51 52 /** 53 * @param pcm 16 kHz mono Float32Array 54 * @param language a Whisper code ("ja", "ru", ...), or null to let it decide 55 * @returns [{text, start, end}] with times offset by `offset` seconds 56 */ 57 async transcribe(pcm, language, offset = 0) { 58 const out = await this.#pipe(pcm, { 59 return_timestamps: true, 60 chunk_length_s: 30, 61 stride_length_s: 5, 62 task: 'transcribe', 63 ...(language ? { language } : {}), 64 }); 65 const segments = []; 66 for (const c of out.chunks ?? []) { 67 const text = (c.text ?? '').trim(); 68 const [s, e] = c.timestamp ?? []; 69 if (!text || s == null) continue; 70 // The last segment of a buffer can come back without an end. 71 segments.push({ text, start: offset + s, end: offset + (e ?? pcm.length / 16000) }); 72 } 73 return segments; 74 } 75 }