Recently Written · git

desktop-subtitle-replay

git clone https://github.com/equwal/desktop-subtitle-replay

Log | Files | Refs


bench.py (4296 bytes)

1 #!/usr/bin/env python3
2 """
3 bench.py - how fast can this machine transcribe? Picks your --model for you.
4 
5     python bench.py                      # synthetic audio, default model set
6     python bench.py --models small,medium,large-v3-turbo
7     python bench.py --wav selftest.wav   # far more accurate: use real speech
8 
9 What matters for live captions is the wall-clock time to transcribe one
10 segment, because that is the delay between someone finishing a sentence and
11 the caption appearing. Whisper's encoder always runs over a padded 30s window,
12 so that cost is roughly constant no matter how short the segment is.
13 """
14 
15 import argparse
16 import time
17 import wave
18 from pathlib import Path
19 
20 import numpy as np
21 
22 HERE = Path(__file__).resolve().parent
23 TARGET_SR = 16000
24 
25 
26 def load_wav(path):
27     with wave.open(str(path), "rb") as w:
28         if w.getframerate() != TARGET_SR or w.getnchannels() != 1:
29             raise SystemExit("%s must be 16 kHz mono (selftest.wav always is)." % path)
30         raw = w.readframes(w.getnframes())
31     return np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0
32 
33 
34 def synth(seconds):
35     """Speech-ish: a few formants, amplitude-modulated at a syllable rate."""
36     t = np.arange(int(TARGET_SR * seconds)) / TARGET_SR
37     rng = np.random.default_rng(0)
38     sig = np.zeros_like(t)
39     for f in (140, 420, 900, 1800, 2600):
40         sig += np.sin(2 * np.pi * f * t + rng.uniform(0, 6.28)) / f ** 0.35
41     syllables = 0.5 + 0.5 * np.sin(2 * np.pi * 4.5 * t)
42     sig = sig * syllables + 0.01 * rng.standard_normal(len(t))
43     return (0.3 * sig / np.max(np.abs(sig))).astype(np.float32)
44 
45 
46 def main():
47     p = argparse.ArgumentParser()
48     p.add_argument("--models", default="tiny,base,small,large-v3-turbo")
49     p.add_argument("--lang", default="fi")
50     p.add_argument("--compute", default="int8")
51     p.add_argument("--threads", type=int, default=None)
52     p.add_argument("--seconds", type=float, default=6.0)
53     p.add_argument("--runs", type=int, default=3)
54     p.add_argument("--wav", default=None, help="16 kHz mono wav of real speech")
55     args = p.parse_args()
56 
57     import os
58     from faster_whisper import WhisperModel
59 
60     threads = args.threads or max(2, (os.cpu_count() or 8) - 2)
61     if args.wav:
62         audio = load_wav(Path(args.wav) if Path(args.wav).is_absolute() else HERE / args.wav)
63         source = args.wav
64     else:
65         audio = synth(args.seconds)
66         source = "synthetic (install-free, but optimistic on decode time)"
67     dur = len(audio) / TARGET_SR
68 
69     print("\naudio: %s  (%.1fs)   threads: %d   compute: %s\n"
70           % (source, dur, threads, args.compute))
71     print("%-20s %9s %9s %8s   %s" % ("model", "load", "per-seg", "RTF", "verdict"))
72     print("-" * 72)
73 
74     for name in [m.strip() for m in args.models.split(",") if m.strip()]:
75         try:
76             t0 = time.time()
77             model = WhisperModel(name, device="cpu", compute_type=args.compute,
78                                  cpu_threads=threads, num_workers=1)
79             load = time.time() - t0
80 
81             times = []
82             for i in range(args.runs):
83                 t0 = time.time()
84                 segs, _ = model.transcribe(
85                     audio, language=None if args.lang == "auto" else args.lang,
86                     beam_size=5, temperature=0.0, condition_on_previous_text=False,
87                     vad_filter=False, without_timestamps=True)
88                 list(segs)                      # generator: force the work
89                 times.append(time.time() - t0)
90             best = min(times[1:] or times)      # first run includes warm-up
91             rtf = best / dur
92 
93             if rtf < 0.35:
94                 verdict = "excellent - use this"
95             elif rtf < 0.6:
96                 verdict = "fine for live"
97             elif rtf < 1.0:
98                 verdict = "tight, captions will lag"
99             else:
100                 verdict = "too slow"
101             print("%-20s %8.1fs %8.2fs %8.2f   %s" % (name, load, best, rtf, verdict))
102             del model
103         except Exception as e:
104             print("%-20s %s" % (name, repr(e)[:60]))
105 
106     print("\nper-seg = delay between end of a sentence and its caption.")
107     print("Anything under ~0.6 RTF keeps up with continuous speech.\n")
108 
109 
110 if __name__ == "__main__":
111     main()