bench.py (4296 bytes)
1 #!/usr/bin/env python3 2 """ 3 bench.py - how fast can this machine transcribe? Picks your --model for you. 4 5 python bench.py # synthetic audio, default model set 6 python bench.py --models small,medium,large-v3-turbo 7 python bench.py --wav selftest.wav # far more accurate: use real speech 8 9 What matters for live captions is the wall-clock time to transcribe one 10 segment, because that is the delay between someone finishing a sentence and 11 the caption appearing. Whisper's encoder always runs over a padded 30s window, 12 so that cost is roughly constant no matter how short the segment is. 13 """ 14 15 import argparse 16 import time 17 import wave 18 from pathlib import Path 19 20 import numpy as np 21 22 HERE = Path(__file__).resolve().parent 23 TARGET_SR = 16000 24 25 26 def load_wav(path): 27 with wave.open(str(path), "rb") as w: 28 if w.getframerate() != TARGET_SR or w.getnchannels() != 1: 29 raise SystemExit("%s must be 16 kHz mono (selftest.wav always is)." % path) 30 raw = w.readframes(w.getnframes()) 31 return np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0 32 33 34 def synth(seconds): 35 """Speech-ish: a few formants, amplitude-modulated at a syllable rate.""" 36 t = np.arange(int(TARGET_SR * seconds)) / TARGET_SR 37 rng = np.random.default_rng(0) 38 sig = np.zeros_like(t) 39 for f in (140, 420, 900, 1800, 2600): 40 sig += np.sin(2 * np.pi * f * t + rng.uniform(0, 6.28)) / f ** 0.35 41 syllables = 0.5 + 0.5 * np.sin(2 * np.pi * 4.5 * t) 42 sig = sig * syllables + 0.01 * rng.standard_normal(len(t)) 43 return (0.3 * sig / np.max(np.abs(sig))).astype(np.float32) 44 45 46 def main(): 47 p = argparse.ArgumentParser() 48 p.add_argument("--models", default="tiny,base,small,large-v3-turbo") 49 p.add_argument("--lang", default="fi") 50 p.add_argument("--compute", default="int8") 51 p.add_argument("--threads", type=int, default=None) 52 p.add_argument("--seconds", type=float, default=6.0) 53 p.add_argument("--runs", type=int, default=3) 54 p.add_argument("--wav", default=None, help="16 kHz mono wav of real speech") 55 args = p.parse_args() 56 57 import os 58 from faster_whisper import WhisperModel 59 60 threads = args.threads or max(2, (os.cpu_count() or 8) - 2) 61 if args.wav: 62 audio = load_wav(Path(args.wav) if Path(args.wav).is_absolute() else HERE / args.wav) 63 source = args.wav 64 else: 65 audio = synth(args.seconds) 66 source = "synthetic (install-free, but optimistic on decode time)" 67 dur = len(audio) / TARGET_SR 68 69 print("\naudio: %s (%.1fs) threads: %d compute: %s\n" 70 % (source, dur, threads, args.compute)) 71 print("%-20s %9s %9s %8s %s" % ("model", "load", "per-seg", "RTF", "verdict")) 72 print("-" * 72) 73 74 for name in [m.strip() for m in args.models.split(",") if m.strip()]: 75 try: 76 t0 = time.time() 77 model = WhisperModel(name, device="cpu", compute_type=args.compute, 78 cpu_threads=threads, num_workers=1) 79 load = time.time() - t0 80 81 times = [] 82 for i in range(args.runs): 83 t0 = time.time() 84 segs, _ = model.transcribe( 85 audio, language=None if args.lang == "auto" else args.lang, 86 beam_size=5, temperature=0.0, condition_on_previous_text=False, 87 vad_filter=False, without_timestamps=True) 88 list(segs) # generator: force the work 89 times.append(time.time() - t0) 90 best = min(times[1:] or times) # first run includes warm-up 91 rtf = best / dur 92 93 if rtf < 0.35: 94 verdict = "excellent - use this" 95 elif rtf < 0.6: 96 verdict = "fine for live" 97 elif rtf < 1.0: 98 verdict = "tight, captions will lag" 99 else: 100 verdict = "too slow" 101 print("%-20s %8.1fs %8.2fs %8.2f %s" % (name, load, best, rtf, verdict)) 102 del model 103 except Exception as e: 104 print("%-20s %s" % (name, repr(e)[:60])) 105 106 print("\nper-seg = delay between end of a sentence and its caption.") 107 print("Anything under ~0.6 RTF keeps up with continuous speech.\n") 108 109 110 if __name__ == "__main__": 111 main()