commit 902287ac6f2c3bf6000c113600ae03de91a518c9 equwal <truex@equwal.com> 2026-08-26 03:15:11 -0700 Live Whisper captions for OBS Local speech-to-caption pipeline for OBS Studio on Windows: WASAPI capture -> energy VAD segmentation -> faster-whisper -> WebSocket -> transparent browser overlay, with a plain-text mirror for GDI+ text sources. Runs fully offline. Defaults to the small model, which is the largest that holds real time on CPU (RTF 0.57 on real speech, 16 cores, int8); the README documents measured numbers for tiny through large-v3-turbo so users can pick. Ships a converter for a Finnish-fine-tuned small, same speed but better Finnish. Includes partial (in-progress) captions, hallucination filtering for Whisper's stock silence phrases, adaptive noise-floor gating, level meter and selftest diagnostics, and optional English translation lines. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
.gitattributes | 3 + .github/workflows/ci.yml | 37 +++ .gitignore | 24 ++ LICENSE | 21 ++ README.md | 228 +++++++++++++++++++ bench.py | 111 +++++++++ get-finnish-model.ps1 | 91 ++++++++ livecap.py | 571 +++++++++++++++++++++++++++++++++++++++++++++++ overlay.html | 154 +++++++++++++ requirements.txt | 5 + run.ps1 | 9 + setup.ps1 | 30 +++ tests/test_smoke.py | 94 ++++++++ 13 files changed, 1378 insertions(+)
diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..d42edcd --- /dev/null +++ b/.gitattributes @@ -0,0 +1,3 @@ +* text=auto eol=lf +*.ps1 text eol=crlf +*.bat text eol=crlf diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..f032299 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,37 @@ +name: ci + +on: + push: + branches: [main] + pull_request: + workflow_dispatch: + +jobs: + check: + runs-on: windows-latest + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.11", "3.12"] + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -r requirements.txt + + - name: Byte-compile + run: python -m compileall -q livecap.py bench.py + + - name: CLI smoke check + run: | + python livecap.py --help + python bench.py --help + + - name: Tests + run: python tests/test_smoke.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..c975da6 --- /dev/null +++ b/.gitignore @@ -0,0 +1,24 @@ +# virtualenvs +.venv/ +.venv-convert/ +build/ + +# python +__pycache__/ +*.py[cod] + +# runtime output +captions.txt +captions.log +selftest.wav +models/ +*.pid +.livecap.pid +_live.log +_live.err + +# editors / os +.vscode/ +.idea/ +Thumbs.db +desktop.ini diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..160e1db --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 equwal + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..77c0775 --- /dev/null +++ b/README.md @@ -0,0 +1,228 @@ +# livecap — live captions for OBS + +Real-time speech captions for OBS Studio on Windows. Captures audio, detects +speech, transcribes it with Whisper, and renders it into a transparent Browser +Source overlay. + +Runs entirely on your machine. No API keys, no cloud, no internet after the +model is downloaded. Built for Finnish, works with any language Whisper +supports. + +``` +audio device ──► VAD segmenter ──► faster-whisper ──► WebSocket ──► overlay.html + └─► captions.txt (GDI+ fallback) +``` + +## Requirements + +- Windows 10/11 +- Python 3.9–3.12 (3.11 recommended) +- OBS Studio 28+ +- ~2 GB disk for the model cache +- A GPU is *not* required — this is tuned to run on CPU + +## Install + +```bash +.\setup.ps1 +``` + +Creates a `.venv` and installs dependencies. One-time. + +## Run + +```bash +.\run.ps1 +``` + +Prints a Browser Source URL. In OBS: **+ → Browser**, paste it, set +**1920 × 1080**, and untick **Shutdown source when not visible**. + +``` +http://127.0.0.1:8777/overlay.html?ws=8765&lines=2&size=42&hide=8 +``` + +The overlay is transparent and places captions in the lower third. Keep the +console window open while streaming; Ctrl+C stops it. + +Check your devices first if needed: + +```bash +.\run.ps1 --list-devices +``` + +## Choosing what gets captioned + +**Default (`--loopback`)** captures everything your speakers play — guests, +video, game audio, music. Zero setup. The catch: your own microphone is *not* +included unless OBS monitors it, and music gets fed to Whisper too. + +**Recommended: route a dedicated mix through a virtual cable.** With +[VB-Audio Virtual Cable](https://vb-audio.com/Cable/) installed: + +1. OBS → **Settings → Audio → Advanced → Monitoring Device** = + `CABLE Input (VB-Audio Virtual Cable)` +2. Audio Mixer → gear icon on each source you want captioned → + **Advanced Audio Properties** → **Audio Monitoring** = **Monitor and Output** + (this keeps the source audible to viewers; *Monitor Only* would mute it) +3. Leave music, game and alert sources on **Monitor Off** +4. Run: + +```bash +.\run.ps1 --mic --audio-device CABLE +``` + +Accuracy improves noticeably once Whisper stops trying to transcribe your +background music. + +## Model selection + +Measured on a 16-core CPU with no CUDA GPU, `int8` quantisation: + +| model | real speech RTF | latency per segment | Finnish quality | +|---|---|---|---| +| `tiny` | 0.07 | ~0.4 s | poor | +| `base` | 0.10 | ~0.6 s | weak | +| **`small`** (default) | **0.57** | **~2.5 s** | good | +| `large-v3-turbo` | 1.77 | ~10.6 s | unusable — falls behind | + +RTF (real-time factor) below ~0.6 keeps up with continuous speech. Without a +CUDA GPU, `small` is the largest model that stays real-time, which is why it is +the default. With an NVIDIA GPU, add `--compute-device cuda --compute float16` +and `large-v3` becomes viable. + +Benchmark your own machine: + +```bash +.\.venv\Scripts\python.exe .\bench.py --models small,medium +``` + +Synthetic audio makes models look faster than they are, because there are fewer +tokens to decode. For a realistic number, point it at a recording: + +```bash +.\.venv\Scripts\python.exe .\bench.py --wav selftest.wav +``` + +### Better Finnish at the same speed + +Stock `small` is a generalist. A Finnish-fine-tuned `small` is the same size — +so the same speed — but markedly better at Finnish. +`get-finnish-model.ps1` fetches one and converts it to CTranslate2: + +```bash +.\get-finnish-model.ps1 +``` + +```bash +.\run.ps1 --model .\build\models\fi-small-ct2 +``` + +The conversion needs ~8 GB free and pulls in torch, which is used only for that +one-off step. Use `-BuildRoot D:\somewhere` to build on another drive, and +delete `.venv-convert` under the build root afterwards to reclaim the space. + +## Caption latency and stream delay + +The overlay is composited into the program feed *before* any stream delay or +replay buffer, so captions ride along with the video automatically — a buffer +needs no special handling. + +A caption appears roughly **2.5 s after the sentence ends**: inference time plus +the 0.65 s of silence used to detect the end of the sentence. To lock captions +to lips, delay the picture to match: + +- **Render Delay** filter of `2500` ms on your video sources +- **Sync Offset** of `2500` ms on the audio sources + (Advanced Audio Properties) + +If you already run a replay buffer or stream delay, the extra 2.5 s costs you +nothing. + +## Options + +| flag | default | effect | +|---|---|---| +| `--lang` | `fi` | `fi`, `en`, `sv`, … or `auto` | +| `--model` | `small` | model name or a local CTranslate2 directory | +| `--mic` | off | capture an input device instead of desktop output | +| `--audio-device` | auto | index from `--list-devices`, or part of the name | +| `--pause` | `0.65` | silence (s) that ends a caption; lower = snappier, more fragments | +| `--min-speech` | `0.45` | ignore bursts shorter than this | +| `--max-seg` | `11.0` | force a cut during non-stop speech | +| `--vad-floor` | `0.004` | absolute level gate; raise if noise triggers captions | +| `--vad-ratio` | `3.0` | gate relative to the running noise floor | +| `--no-partials` | off | only show finished sentences; roughly halves CPU | +| `--translate` | off | add an English line beneath the original | +| `--lines` / `--size` | `2` / `42` | overlay line count and font size | +| `--hide` | `8` | fade the overlay out after N idle seconds | +| `--compute-device` | `cpu` | `cuda` if you have an NVIDIA GPU | + +## Diagnostics + +```bash +.\run.ps1 --selftest 12 +``` + +Records 12 s, writes `selftest.wav`, transcribes it once and reports the +real-time factor. Run this first — speak or play audio during those 12 seconds +and check what comes back. + +```bash +.\run.ps1 --meter +``` + +Live level meter showing the VAD gate, for tuning `--vad-floor`. Add +`--verbose` to log every segment the VAD decides to send. + +| symptom | cause | +|---|---| +| `no speech recognised in N.Ns segment` | music/noise reaching Whisper, or wrong `--lang` | +| nothing at all in the log | VAD never fires — check `--meter`, wrong device | +| `backlog full` | model too slow for real time; use a smaller one | +| red dot in the overlay | overlay lost the WebSocket; it retries automatically | + +Whisper likes to hallucinate stock phrases over silence (`Tekstitys: YLE`, +`Kiitos kun katsoit!`, `Thanks for watching`). Short results matching those are +filtered out, alongside a no-speech-probability threshold. + +## Overlay styling + +Append query parameters to the Browser Source URL: + +| param | example | effect | +|---|---|---| +| `size` | `size=52` | font size in px | +| `lines` | `lines=3` | how many lines stay on screen | +| `align` | `align=top` | `top`, `center`, default bottom | +| `fg` | `fg=ffe066` | text colour (hex, no `#`) | +| `accent` | `accent=7dd3fc` | translation line colour | +| `box` | `box=0` | remove the dark background pill | +| `hide` | `hide=0` | never auto-hide | +| `tr` | `tr=1` | show translation lines (with `--translate`) | + +## Text output + +`captions.txt` holds the last few lines for an OBS **Text (GDI+)** source with +*Read from file*, if you would rather avoid browser sources. `captions.log` +keeps the full timestamped transcript of the session, which doubles as stream +notes. + +## How it works + +`livecap.py` pulls 32 ms blocks from a WASAPI device via `soundcard`, tracks a +running noise floor, and marks blocks as speech when they exceed +`max(noise × ratio, floor)`. Speech accumulates into a segment; a segment closes +on `--pause` of trailing silence or at `--max-seg`. Closed segments go to +faster-whisper on a worker thread and are published as `final`. While a segment +is still open, idle worker time is spent transcribing the partial buffer and +publishing lower-confidence `partial` text, so captions appear before the +speaker finishes. Results are broadcast over WebSocket to `overlay.html` and +mirrored to disk. + +## License + +MIT — see [LICENSE](LICENSE). + +Uses [faster-whisper](https://github.com/SYSTRAN/faster-whisper) (MIT) and +OpenAI's Whisper models. diff --git a/bench.py b/bench.py new file mode 100644 index 0000000..13722c5 --- /dev/null +++ b/bench.py @@ -0,0 +1,111 @@ +#!/usr/bin/env python3 +""" +bench.py - how fast can this machine transcribe? Picks your --model for you. + + python bench.py # synthetic audio, default model set + python bench.py --models small,medium,large-v3-turbo + python bench.py --wav selftest.wav # far more accurate: use real speech + +What matters for live captions is the wall-clock time to transcribe one +segment, because that is the delay between someone finishing a sentence and +the caption appearing. Whisper's encoder always runs over a padded 30s window, +so that cost is roughly constant no matter how short the segment is. +""" + +import argparse +import time +import wave +from pathlib import Path + +import numpy as np + +HERE = Path(__file__).resolve().parent +TARGET_SR = 16000 + + +def load_wav(path): + with wave.open(str(path), "rb") as w: + if w.getframerate() != TARGET_SR or w.getnchannels() != 1: + raise SystemExit("%s must be 16 kHz mono (selftest.wav always is)." % path) + raw = w.readframes(w.getnframes()) + return np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0 + + +def synth(seconds): + """Speech-ish: a few formants, amplitude-modulated at a syllable rate.""" + t = np.arange(int(TARGET_SR * seconds)) / TARGET_SR + rng = np.random.default_rng(0) + sig = np.zeros_like(t) + for f in (140, 420, 900, 1800, 2600): + sig += np.sin(2 * np.pi * f * t + rng.uniform(0, 6.28)) / f ** 0.35 + syllables = 0.5 + 0.5 * np.sin(2 * np.pi * 4.5 * t) + sig = sig * syllables + 0.01 * rng.standard_normal(len(t)) + return (0.3 * sig / np.max(np.abs(sig))).astype(np.float32) + + +def main(): + p = argparse.ArgumentParser() + p.add_argument("--models", default="tiny,base,small,large-v3-turbo") + p.add_argument("--lang", default="fi") + p.add_argument("--compute", default="int8") + p.add_argument("--threads", type=int, default=None) + p.add_argument("--seconds", type=float, default=6.0) + p.add_argument("--runs", type=int, default=3) + p.add_argument("--wav", default=None, help="16 kHz mono wav of real speech") + args = p.parse_args() + + import os + from faster_whisper import WhisperModel + + threads = args.threads or max(2, (os.cpu_count() or 8) - 2) + if args.wav: + audio = load_wav(Path(args.wav) if Path(args.wav).is_absolute() else HERE / args.wav) + source = args.wav + else: + audio = synth(args.seconds) + source = "synthetic (install-free, but optimistic on decode time)" + dur = len(audio) / TARGET_SR + + print("\naudio: %s (%.1fs) threads: %d compute: %s\n" + % (source, dur, threads, args.compute)) + print("%-20s %9s %9s %8s %s" % ("model", "load", "per-seg", "RTF", "verdict")) + print("-" * 72) + + for name in [m.strip() for m in args.models.split(",") if m.strip()]: + try: + t0 = time.time() + model = WhisperModel(name, device="cpu", compute_type=args.compute, + cpu_threads=threads, num_workers=1) + load = time.time() - t0 + + times = [] + for i in range(args.runs): + t0 = time.time() + segs, _ = model.transcribe( + audio, language=None if args.lang == "auto" else args.lang, + beam_size=5, temperature=0.0, condition_on_previous_text=False, + vad_filter=False, without_timestamps=True) + list(segs) # generator: force the work + times.append(time.time() - t0) + best = min(times[1:] or times) # first run includes warm-up + rtf = best / dur + + if rtf < 0.35: + verdict = "excellent - use this" + elif rtf < 0.6: + verdict = "fine for live" + elif rtf < 1.0: + verdict = "tight, captions will lag" + else: + verdict = "too slow" + print("%-20s %8.1fs %8.2fs %8.2f %s" % (name, load, best, rtf, verdict)) + del model + except Exception as e: + print("%-20s %s" % (name, repr(e)[:60])) + + print("\nper-seg = delay between end of a sentence and its caption.") + print("Anything under ~0.6 RTF keeps up with continuous speech.\n") + + +if __name__ == "__main__": + main() diff --git a/get-finnish-model.ps1 b/get-finnish-model.ps1 new file mode 100644 index 0000000..8501791 --- /dev/null +++ b/get-finnish-model.ps1 @@ -0,0 +1,91 @@ +# Optional quality upgrade: a Finnish-fine-tuned Whisper 'small', converted to +# CTranslate2 int8 so faster-whisper can run it. +# +# Plain 'small' is a generalist and merely OK at Finnish. This fine-tune is the +# same size (= same speed on your CPU) but much better at Finnish specifically. +# +# torch + transformers are needed only for the one-off conversion, so they go +# into a throwaway venv under -BuildRoot, which you can delete afterwards. +# Needs roughly 8 GB free; use -BuildRoot to build on a roomier drive. +# +# .\get-finnish-model.ps1 +# .\get-finnish-model.ps1 -BuildRoot D:\livecap-build +# .\run.ps1 --model .\build\models\fi-small-ct2 + +param( + [string]$BuildRoot = "$PSScriptRoot\build" +) + +$ErrorActionPreference = "Stop" +Set-Location $PSScriptRoot + +$src = "RASMUS/Whisper_Finnish_finetuned_small_200k_samples" +$out = Join-Path $BuildRoot "models\fi-small-ct2" +$venv = Join-Path $BuildRoot ".venv-convert" + +try { + New-Item -ItemType Directory -Force -Path $BuildRoot -ErrorAction Stop | Out-Null +} catch { + throw "Cannot create $BuildRoot ($($_.Exception.Message)).`n" + + "Pick a writable location: .\get-finnish-model.ps1 -BuildRoot D:\some\writable\dir" +} + +# Keep the multi-GB HF download off C: as well. +$env:HF_HOME = Join-Path $BuildRoot "hf-cache" +$env:PIP_CACHE_DIR = Join-Path $BuildRoot "pip-cache" + +if (Test-Path $out) { + Write-Host "$out already exists. Delete it to rebuild." -ForegroundColor Yellow + Write-Host "Use it with: .\run.ps1 --model `"$out`"" + exit 0 +} + +$freeGB = $null +try { + $root = [System.IO.Path]::GetPathRoot((Resolve-Path $BuildRoot).Path) + $freeGB = [math]::Round((Get-CimInstance Win32_LogicalDisk ` + -Filter "DeviceID='$($root.TrimEnd('\'))'").FreeSpace / 1GB, 1) +} catch { } + +if ($freeGB) { + Write-Host "Build root: $BuildRoot ($freeGB GB free)" -ForegroundColor Cyan + if ($freeGB -lt 8) { + throw "Need ~8 GB free for the conversion; only $freeGB GB available on $root.`n" + + "Free up space, or use -BuildRoot on a roomier drive." + } +} else { + Write-Host "Build root: $BuildRoot (free space unknown; need ~8 GB)" -ForegroundColor Cyan +} + +if (-not (Test-Path "$venv\Scripts\python.exe")) { + Write-Host "Creating conversion venv (one-time, ~2.5 GB of torch)..." -ForegroundColor Cyan + & py -3.11 -m venv $venv + & "$venv\Scripts\python.exe" -m pip install --upgrade pip --quiet + & "$venv\Scripts\python.exe" -m pip install --index-url https://download.pytorch.org/whl/cpu torch + if ($LASTEXITCODE -ne 0) { throw "torch install failed" } + & "$venv\Scripts\python.exe" -m pip install transformers ctranslate2 "numpy<3" + if ($LASTEXITCODE -ne 0) { throw "transformers/ctranslate2 install failed" } +} + +Write-Host "Downloading + converting $src ..." -ForegroundColor Cyan +& "$venv\Scripts\ct2-transformers-converter.exe" ` + --model $src ` + --output_dir $out ` + --quantization int8 ` + --copy_files preprocessor_config.json tokenizer_config.json special_tokens_map.json ` + vocab.json merges.txt normalizer.json added_tokens.json +if ($LASTEXITCODE -ne 0) { throw "conversion failed" } + +# faster-whisper wants a real tokenizer.json, otherwise it quietly falls back to +# the stock openai/whisper-tiny tokenizer. +$py = @" +from transformers import WhisperTokenizerFast +WhisperTokenizerFast.from_pretrained(r'$src').save_pretrained(r'$out') +print('tokenizer.json written') +"@ +$py | & "$venv\Scripts\python.exe" - + +Write-Host "" +Write-Host "Done -> $out" -ForegroundColor Green +Write-Host "Use it with: .\run.ps1 --model `"$out`"" +Write-Host "Reclaim space afterwards with: Remove-Item -Recurse -Force `"$venv`"" diff --git a/livecap.py b/livecap.py new file mode 100644 index 0000000..930bb9f --- /dev/null +++ b/livecap.py @@ -0,0 +1,571 @@ +#!/usr/bin/env python3 +""" +livecap.py - live speech-to-captions for OBS. + +Captures audio (WASAPI loopback = "whatever your speakers play", or any mic), +segments it with an energy VAD, transcribes with faster-whisper, and publishes +captions over WebSocket to an OBS Browser Source overlay. + + python livecap.py --list-devices + python livecap.py --selftest 12 + python livecap.py --lang fi --model small +""" + +import argparse +import asyncio +import json +import os +import queue +import re +import sys +import threading +import time +import wave +from collections import deque +from functools import partial +from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +import numpy as np +import soundcard as sc +from scipy.signal import resample_poly + +HERE = Path(__file__).resolve().parent +TARGET_SR = 16000 +CAPTURE_SR = 48000 # WASAPI shared-mode mix rate on virtually every Windows box +BLOCK_MS = 32 + +# Whisper invents these when fed silence or noise. Only applied to short results. +HALLUCINATIONS = [ + r"^tekstitys", + r"^tekstityksen tuotti", + r"^k[aa]a?nn[oo]s", + r"^kiitos( kun katsoit| paljon| katsomisesta)?[.!]?$", + r"^suomennos", + r"^subtitles? by", + r"^amara\.org", + r"^thanks? for watching", + r"^\W*$", +] +HALLUCINATION_RE = [re.compile(p, re.I) for p in HALLUCINATIONS] + + +def log(*a): + print("[" + time.strftime("%H:%M:%S") + "]", *a, flush=True) + + +# ---------------------------------------------------------------- devices --- + +def candidates(loopback): + """Devices usable in the current mode, in a stable order.""" + mics = sc.all_microphones(include_loopback=True) + return [m for m in mics if bool(m.isloopback) == bool(loopback)] + + +def list_devices(): + print("\n=== LOOPBACK sources (default mode: captures what this device plays) ===") + for i, m in enumerate(candidates(True)): + print(" %2d %s (%dch)" % (i, m.name, m.channels)) + try: + print("\n default (used when you pass no --audio-device): %s" + % sc.default_speaker().name) + except Exception: + pass + + print("\n=== INPUT devices (--mic mode: microphones, virtual cables) ===") + for i, m in enumerate(candidates(False)): + print(" %2d %s (%dch)" % (i, m.name, m.channels)) + try: + print("\n default (--mic with no --audio-device): %s" % sc.default_microphone().name) + except Exception: + pass + print("\nSelect with --audio-device followed by the number above, or any\n" + "case-insensitive part of the name (e.g. --audio-device CABLE).\n") + + +def resolve_device(spec, loopback): + """spec: None, an index into the listing, or a substring of the device name.""" + pool = candidates(loopback) + if not pool: + raise RuntimeError("No %s devices found." % ("loopback" if loopback else "input")) + + if spec is None: + try: + if loopback: + return sc.get_microphone(sc.default_speaker().id, include_loopback=True) + return sc.default_microphone() + except Exception: + return pool[0] + + try: + idx = int(spec) + except ValueError: + pass + else: + if 0 <= idx < len(pool): + return pool[idx] + raise RuntimeError("Device index %d out of range (0-%d)." % (idx, len(pool) - 1)) + + want = spec.lower() + for m in pool: + if want in m.name.lower(): + return m + raise RuntimeError("No %s device matching %r. Run --list-devices." + % ("loopback" if loopback else "input", spec)) + + +# ---------------------------------------------------------------- capture --- + +class Capture: + """Pumps mono float32 blocks off a soundcard device onto a queue.""" + + def __init__(self, mic, loopback=True): + self.mic = mic + self.name = mic.name + self.loopback = loopback + self.sr = CAPTURE_SR + self.channels = max(1, int(mic.channels)) + self.blocksize = int(self.sr * BLOCK_MS / 1000) + self.q = queue.Queue(maxsize=256) + self.dropped = 0 + self.error = None + self._stop = threading.Event() + self._ready = threading.Event() + self._thread = None + + def _pump(self): + try: + with self.mic.recorder(samplerate=self.sr, channels=self.channels, + blocksize=self.blocksize) as rec: + self._ready.set() + while not self._stop.is_set(): + data = rec.record(numframes=self.blocksize) + if data.ndim > 1 and data.shape[1] > 1: + mono = data.mean(axis=1) + else: + mono = data.reshape(-1) + try: + self.q.put_nowait(np.ascontiguousarray(mono, dtype=np.float32)) + except queue.Full: + self.dropped += 1 + except Exception as e: + self.error = e + log("capture failed:", repr(e)) + finally: + self._ready.set() + + def __enter__(self): + self._thread = threading.Thread(target=self._pump, daemon=True) + self._thread.start() + self._ready.wait(timeout=10) + if self.error is not None: + raise self.error + log("capturing: %s (%dch @ %dHz, %s)" + % (self.name, self.channels, self.sr, "loopback" if self.loopback else "input")) + return self + + def __exit__(self, *a): + self._stop.set() + if self._thread is not None: + self._thread.join(timeout=2) + + +def to_whisper(audio, sr): + """native-rate mono float32 -> 16 kHz float32, gently gain-staged.""" + if sr != TARGET_SR: + g = int(np.gcd(sr, TARGET_SR)) + audio = resample_poly(audio, TARGET_SR // g, sr // g) + audio = np.asarray(audio, dtype=np.float32) + peak = float(np.max(np.abs(audio))) if audio.size else 0.0 + if 0.0 < peak < 0.35: + audio = audio * (0.35 / peak) + return np.clip(audio, -1.0, 1.0).astype(np.float32) + + +# -------------------------------------------------------------- publishing -- + +class Bus: + def __init__(self, args): + self.args = args + self.loop = None + self.clients = set() + self.history = deque(maxlen=12) + self.finals = deque(maxlen=max(1, args.txt_lines)) + self.txt = HERE / args.txt + self.logf = HERE / "captions.log" + + def attach(self, loop): + self.loop = loop + + def publish(self, msg): + if msg["type"] == "final": + self.history.append(msg) + self.finals.append(msg["text"]) + self._write_files(msg) + if self.loop is not None: + payload = json.dumps(msg, ensure_ascii=False) + self.loop.call_soon_threadsafe(self._fanout, payload) + + def _fanout(self, payload): + for ws in list(self.clients): + try: + asyncio.get_running_loop().create_task(ws.send(payload)) + except Exception: + self.clients.discard(ws) + + def _write_files(self, msg): + try: + self.txt.write_text("\n".join(self.finals) + "\n", encoding="utf-8") + line = time.strftime("%Y-%m-%d %H:%M:%S") + "\t" + msg["text"] + if msg.get("tr"): + line += "\t|| " + msg["tr"] + with self.logf.open("a", encoding="utf-8") as fh: + fh.write(line + "\n") + except OSError as e: + log("file write failed:", e) + + +async def ws_server(bus, args, stop): + import websockets + + async def handler(ws): + bus.clients.add(ws) + log("overlay connected (%d client(s))" % len(bus.clients)) + try: + for m in list(bus.history)[-4:]: + await ws.send(json.dumps(m, ensure_ascii=False)) + await ws.wait_closed() + finally: + bus.clients.discard(ws) + log("overlay disconnected (%d client(s))" % len(bus.clients)) + + async with websockets.serve(handler, "127.0.0.1", args.ws_port, ping_interval=20): + log("websocket ws://127.0.0.1:%d" % args.ws_port) + await stop.wait() + + +class QuietHandler(SimpleHTTPRequestHandler): + def log_message(self, *a): + pass + + +def http_server(args): + handler = partial(QuietHandler, directory=str(HERE)) + srv = ThreadingHTTPServer(("127.0.0.1", args.http_port), handler) + threading.Thread(target=srv.serve_forever, daemon=True).start() + url = ("http://127.0.0.1:%d/overlay.html?ws=%d&lines=%d&size=%d&hide=%s" + % (args.http_port, args.ws_port, args.lines, args.size, args.hide)) + if args.translate: + url += "&tr=1" + log("OBS Browser Source URL (copy this):") + log(" " + url) + return srv + + +# ----------------------------------------------------------- transcription -- + +def looks_hallucinated(text): + t = text.strip().lower() + if len(t) > 40: + return False + return any(r.search(t) for r in HALLUCINATION_RE) + + +class Transcriber: + def __init__(self, args, bus): + from faster_whisper import WhisperModel + + self.args = args + self.bus = bus + log("loading model %r on %s/%s (first run downloads it)..." + % (args.model, args.compute_device, args.compute)) + t0 = time.time() + self.model = WhisperModel( + args.model, device=args.compute_device, compute_type=args.compute, + cpu_threads=args.threads, num_workers=1) + log("model ready in %.1fs" % (time.time() - t0)) + self.context = "" + self.busy = threading.Event() + + def run(self, audio, final): + a = self.args + segs, info = self.model.transcribe( + audio, + language=None if a.lang == "auto" else a.lang, + task="transcribe", + beam_size=a.beam if final else 1, + temperature=0.0, + condition_on_previous_text=False, + initial_prompt=(self.context or None) if (final and not a.no_context) else None, + vad_filter=final, + no_speech_threshold=0.6, + log_prob_threshold=-1.0, + without_timestamps=True, + ) + parts, nsp = [], [] + for s in segs: + parts.append(s.text.strip()) + nsp.append(getattr(s, "no_speech_prob", 0.0)) + text = re.sub(r"\s+", " ", " ".join(parts)).strip() + return text, (max(nsp) if nsp else 1.0), info + + def translate(self, audio): + segs, _ = self.model.transcribe( + audio, language=None if self.args.lang == "auto" else self.args.lang, + task="translate", beam_size=1, temperature=0.0, + condition_on_previous_text=False, vad_filter=True, without_timestamps=True) + return re.sub(r"\s+", " ", " ".join(s.text.strip() for s in segs)).strip() + + def worker(self, jobs, stop): + while not stop.is_set(): + try: + kind, audio, sr = jobs.get(timeout=0.25) + except queue.Empty: + continue + self.busy.set() + try: + t0 = time.time() + a16 = to_whisper(audio, sr) + dur = len(a16) / TARGET_SR + text, nsp, _ = self.run(a16, final=(kind == "final")) + if not text: + if kind == "final": + log("no speech recognised in %.1fs segment " + "(music/noise, or wrong --lang)" % dur) + continue + if kind == "final": + if nsp > 0.75 or looks_hallucinated(text): + log("dropped (no_speech=%.2f): %r" % (nsp, text)) + continue + tr = self.translate(a16) if self.args.translate else "" + if not self.args.no_context: + self.context = (self.context + " " + text)[-220:] + took = time.time() - t0 + log("FINAL %4.1fs audio in %4.1fs (rtf %.2f) %s" + % (dur, took, took / max(dur, 0.01), text)) + self.bus.publish({"type": "final", "text": text, + "tr": tr, "ts": time.time()}) + else: + self.bus.publish({"type": "partial", "text": text, "ts": time.time()}) + except Exception as e: + log("transcribe error:", repr(e)) + finally: + self.busy.clear() + jobs.task_done() + + +# ---------------------------------------------------------- segmentation ---- + +def segmenter(cap, tr, jobs, args, stop): + dur = cap.blocksize / cap.sr + noise = 1e-4 + preroll = deque(maxlen=max(1, int(0.35 / dur))) + seg = [] + speech_blocks = 0 + silence = 0.0 + last_partial = 0.0 + last_meter = 0.0 + + while not stop.is_set(): + try: + blk = cap.q.get(timeout=0.3) + except queue.Empty: + continue + + rms = float(np.sqrt(np.mean(blk * blk)) + 1e-12) + if rms < noise: + noise = 0.90 * noise + 0.10 * rms + else: + noise = 0.995 * noise + 0.005 * rms + gate = max(noise * args.vad_ratio, args.vad_floor) + voiced = rms > gate + + if args.meter and time.time() - last_meter > 0.25: + last_meter = time.time() + db = 20 * np.log10(max(rms, 1e-9)) + bars = int(np.clip((db + 60) / 60 * 40, 0, 40)) + sys.stdout.write("\r%-40s %6.1f dBFS gate %6.1f %s" + % ("#" * bars, db, 20 * np.log10(gate), + "VOICE" if voiced else " ")) + sys.stdout.flush() + + if not seg: + if not voiced: + preroll.append(blk) + continue + seg = list(preroll) + preroll.clear() + + seg.append(blk) + if voiced: + speech_blocks += 1 + silence = 0.0 + else: + silence += dur + + seg_dur = len(seg) * dur + speech_dur = speech_blocks * dur + + if speech_dur >= args.min_speech and (silence >= args.pause or seg_dur >= args.max_seg): + if args.verbose: + log("segment queued: %.1fs (%.1fs of it speech)" % (seg_dur, speech_dur)) + try: + jobs.put_nowait(("final", np.concatenate(seg), cap.sr)) + except queue.Full: + log("backlog full - dropping a segment (model too slow for live)") + seg, speech_blocks, silence = [], 0, 0.0 + elif silence >= args.pause: + seg, speech_blocks, silence = [], 0, 0.0 # noise blip + elif (args.partials and speech_dur >= 0.7 + and time.time() - last_partial >= args.partial_every + and not tr.busy.is_set() and jobs.empty()): + last_partial = time.time() + try: + jobs.put_nowait(("partial", np.concatenate(seg), cap.sr)) + except queue.Full: + pass + + +# -------------------------------------------------------------- selftest ---- + +def selftest(args, seconds): + dev = resolve_device(args.audio_device, args.loopback) + with Capture(dev, args.loopback) as cap: + log("recording %ds -- play / speak Finnish audio NOW..." % seconds) + blocks, t0, peak = [], time.time(), 0.0 + while time.time() - t0 < seconds: + try: + b = cap.q.get(timeout=0.5) + except queue.Empty: + continue + blocks.append(b) + peak = max(peak, float(np.max(np.abs(b)))) + sr = cap.sr + + if not blocks: + log("NO AUDIO CAPTURED -- wrong device? Run --list-devices.") + return 1 + audio = np.concatenate(blocks) + log("captured %.1fs, peak %.1f dBFS" % (len(audio) / sr, 20 * np.log10(max(peak, 1e-9)))) + if peak < 0.001: + log("WARNING: that is silence. Pick another device with --audio-device.") + + a16 = to_whisper(audio, sr) + with wave.open(str(HERE / "selftest.wav"), "wb") as w: + w.setnchannels(1) + w.setsampwidth(2) + w.setframerate(TARGET_SR) + w.writeframes((a16 * 32767).astype(np.int16).tobytes()) + log("wrote selftest.wav -- play it to confirm you grabbed the right audio") + + t = Transcriber(args, Bus(args)) + t0 = time.time() + text, nsp, info = t.run(a16, final=True) + took = time.time() - t0 + rtf = took / max(len(a16) / TARGET_SR, 0.01) + log("detected language=%s no_speech=%.2f" % (getattr(info, "language", "?"), nsp)) + verdict = "FAST ENOUGH for live" if rtf < 0.6 else "TOO SLOW -- use a smaller --model" + log("transcribed in %.1fs -> RTF %.2f (%s)" % (took, rtf, verdict)) + print("\n " + (text or "(nothing recognised)") + "\n") + return 0 + + +# ------------------------------------------------------------------ main ---- + +def build_parser(): + p = argparse.ArgumentParser( + description="Live captions for OBS", + formatter_class=argparse.ArgumentDefaultsHelpFormatter) + + g = p.add_argument_group("audio") + g.add_argument("--list-devices", action="store_true", help="show devices and exit") + g.add_argument("--audio-device", default=None, help="index or name substring") + g.add_argument("--loopback", dest="loopback", action="store_true", default=True, + help="capture desktop output (default)") + g.add_argument("--mic", dest="loopback", action="store_false", + help="capture an input device instead of desktop output") + g.add_argument("--meter", action="store_true", help="print a live level meter") + g.add_argument("--verbose", action="store_true", help="log VAD segment decisions") + + g = p.add_argument_group("model") + g.add_argument("--model", default="small", + help="tiny|base|small|medium|large-v3|large-v3-turbo, or a local CT2 dir") + g.add_argument("--lang", default="fi", help="fi, en, sv, ... or 'auto'") + g.add_argument("--compute-device", default="cpu", choices=["cpu", "cuda"]) + g.add_argument("--compute", default="int8", help="int8|int8_float32|float32|float16") + g.add_argument("--threads", type=int, default=max(2, (os.cpu_count() or 8) - 2)) + g.add_argument("--beam", type=int, default=5) + g.add_argument("--translate", action="store_true", + help="also emit an English translation line") + g.add_argument("--no-context", action="store_true", + help="do not feed previous text back as a prompt") + + g = p.add_argument_group("segmentation") + g.add_argument("--pause", type=float, default=0.65, help="silence (s) that closes a caption") + g.add_argument("--min-speech", type=float, default=0.45, help="min speech (s) worth sending") + g.add_argument("--max-seg", type=float, default=11.0, help="force a cut after this many s") + g.add_argument("--vad-floor", type=float, default=0.004, help="absolute RMS gate") + g.add_argument("--vad-ratio", type=float, default=3.0, help="gate = noise floor * this") + g.add_argument("--partials", dest="partials", action="store_true", default=True) + g.add_argument("--no-partials", dest="partials", action="store_false", + help="only show finished sentences (lower CPU)") + g.add_argument("--partial-every", type=float, default=0.9) + + g = p.add_argument_group("output") + g.add_argument("--ws-port", type=int, default=8765) + g.add_argument("--http-port", type=int, default=8777) + g.add_argument("--txt", default="captions.txt", help="plain-text file for a GDI+ text source") + g.add_argument("--txt-lines", type=int, default=2) + g.add_argument("--lines", type=int, default=2, help="lines shown in the overlay") + g.add_argument("--size", type=int, default=42, help="overlay font size in px") + g.add_argument("--hide", type=float, default=8, help="auto-hide overlay after N idle seconds") + + p.add_argument("--selftest", nargs="?", type=int, const=12, default=None, metavar="SECONDS", + help="record N seconds, transcribe once, report speed, then exit") + return p + + +def main(): + args = build_parser().parse_args() + + if args.list_devices: + list_devices() + return 0 + if args.selftest is not None: + return selftest(args, args.selftest) + + bus = Bus(args) + stop_ev = threading.Event() + jobs = queue.Queue(maxsize=3) + + tr = Transcriber(args, bus) + dev = resolve_device(args.audio_device, args.loopback) + + srv = http_server(args) + loop = asyncio.new_event_loop() + bus.attach(loop) + ws_stop = asyncio.Event() + + def run_loop(): + asyncio.set_event_loop(loop) + loop.run_until_complete(ws_server(bus, args, ws_stop)) + + threading.Thread(target=run_loop, daemon=True).start() + threading.Thread(target=tr.worker, args=(jobs, stop_ev), daemon=True).start() + + try: + with Capture(dev, args.loopback) as cap: + log("lang=%s model=%s -- Ctrl+C to stop" % (args.lang, args.model)) + segmenter(cap, tr, jobs, args, stop_ev) + except KeyboardInterrupt: + print() + log("stopping") + finally: + stop_ev.set() + loop.call_soon_threadsafe(ws_stop.set) + srv.shutdown() + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/overlay.html b/overlay.html new file mode 100644 index 0000000..f890e9b --- /dev/null +++ b/overlay.html @@ -0,0 +1,154 @@ +<!doctype html> +<html lang="fi"> +<head> +<meta charset="utf-8"> +<title>Live Captions</title> +<style> + :root { + --size: 42px; + --fg: #ffffff; + --box: rgba(0, 0, 0, .62); + --accent: #7dd3fc; + } + html, body { + margin: 0; padding: 0; height: 100%; + background: transparent; + overflow: hidden; + font-family: "Inter", "Segoe UI Variable Display", "Segoe UI", system-ui, sans-serif; + } + #stage { + position: absolute; inset: 0; + display: flex; flex-direction: column; + justify-content: flex-end; align-items: center; + padding: 0 4vw 3vh; + box-sizing: border-box; + gap: .28em; + transition: opacity .45s ease; + } + #stage.idle { opacity: 0; } + .line { + max-width: 100%; + font-size: var(--size); + font-weight: 650; + line-height: 1.25; + letter-spacing: .005em; + color: var(--fg); + text-align: center; + text-wrap: balance; + background: var(--box); + padding: .16em .5em; + border-radius: .22em; + box-decoration-break: clone; + -webkit-box-decoration-break: clone; + text-shadow: 0 2px 6px rgba(0,0,0,.85); + animation: rise .22s ease-out; + } + .line.partial { opacity: .74; } + .line.tr { + font-size: calc(var(--size) * .62); + font-weight: 550; + color: var(--accent); + opacity: .95; + } + @keyframes rise { + from { opacity: 0; transform: translateY(.35em); } + to { opacity: 1; transform: none; } + } + #dot { + position: absolute; left: 8px; bottom: 8px; + width: 10px; height: 10px; border-radius: 50%; + background: #ef4444; box-shadow: 0 0 8px #ef4444; + opacity: 0; transition: opacity .3s; + } + #dot.show { opacity: .9; } +</style> +</head> +<body> + <div id="stage"><div id="dot" title="disconnected"></div></div> + +<script> +(function () { + const q = new URLSearchParams(location.search); + const num = (k, d) => { const v = parseFloat(q.get(k)); return isNaN(v) ? d : v; }; + + const MAXLINES = Math.max(1, num("lines", 2)); + const HIDE_MS = num("hide", 8) * 1000; + const WS_PORT = num("ws", 8765); + const SHOW_TR = q.get("tr") === "1"; + + document.documentElement.style.setProperty("--size", num("size", 42) + "px"); + if (q.get("fg")) document.documentElement.style.setProperty("--fg", "#" + q.get("fg")); + if (q.get("accent")) document.documentElement.style.setProperty("--accent", "#" + q.get("accent")); + if (q.get("box") === "0") document.documentElement.style.setProperty("--box", "transparent"); + if (q.get("align") === "top") document.getElementById("stage").style.justifyContent = "flex-start"; + if (q.get("align") === "center") document.getElementById("stage").style.justifyContent = "center"; + + const stage = document.getElementById("stage"); + const dot = document.getElementById("dot"); + + let finals = []; // [{text, tr}] + let partial = ""; + let idleTimer = null; + + function render() { + const rows = []; + const shown = finals.slice(-MAXLINES + (partial ? 1 : 0)); + for (const f of shown) { + rows.push({ cls: "line", text: f.text }); + if (SHOW_TR && f.tr) rows.push({ cls: "line tr", text: f.tr }); + } + if (partial) rows.push({ cls: "line partial", text: partial }); + + // Rebuild only when content actually changed, so animations do not restart. + const key = rows.map(r => r.cls + "|" + r.text).join("\n"); + if (key === stage.dataset.key) return; + stage.dataset.key = key; + + stage.querySelectorAll(".line").forEach(n => n.remove()); + for (const r of rows) { + const el = document.createElement("div"); + el.className = r.cls; + el.textContent = r.text; + stage.appendChild(el); + } + } + + function poke() { + stage.classList.remove("idle"); + clearTimeout(idleTimer); + if (HIDE_MS > 0) { + idleTimer = setTimeout(() => { + stage.classList.add("idle"); + setTimeout(() => { finals = []; partial = ""; render(); }, 500); + }, HIDE_MS); + } + } + + function connect() { + const ws = new WebSocket("ws://127.0.0.1:" + WS_PORT); + ws.onopen = () => dot.classList.remove("show"); + ws.onclose = () => { dot.classList.add("show"); setTimeout(connect, 1500); }; + ws.onerror = () => ws.close(); + ws.onmessage = (ev) => { + let m; + try { m = JSON.parse(ev.data); } catch (e) { return; } + if (m.type === "partial") { + partial = m.text; + } else if (m.type === "final") { + partial = ""; + finals.push({ text: m.text, tr: m.tr || "" }); + if (finals.length > 8) finals.shift(); + } else if (m.type === "clear") { + finals = []; partial = ""; + } + render(); + poke(); + }; + } + + render(); + connect(); +})(); +</script> +</body> +</html> diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..9965d4b --- /dev/null +++ b/requirements.txt @@ -0,0 +1,5 @@ +faster-whisper>=1.1.0 +soundcard>=0.4.3 +numpy>=1.26 +scipy>=1.11 +websockets>=12.0 diff --git a/run.ps1 b/run.ps1 new file mode 100644 index 0000000..a6bc5a4 --- /dev/null +++ b/run.ps1 @@ -0,0 +1,9 @@ +# Runs livecap.py inside the venv. All arguments are passed straight through. +# .\run.ps1 -> Finnish captions, desktop audio +# .\run.ps1 --selftest 12 -> 12s recording + speed check +# .\run.ps1 --model medium --translate +$ErrorActionPreference = "Stop" +Set-Location $PSScriptRoot +if (-not (Test-Path ".\.venv\Scripts\python.exe")) { throw "Run .\setup.ps1 first." } +$env:KMP_DUPLICATE_LIB_OK = "TRUE" +& ".\.venv\Scripts\python.exe" ".\livecap.py" @args diff --git a/setup.ps1 b/setup.ps1 new file mode 100644 index 0000000..5fd1d02 --- /dev/null +++ b/setup.ps1 @@ -0,0 +1,30 @@ +# One-time setup: creates .venv and installs dependencies. +$ErrorActionPreference = "Stop" +Set-Location $PSScriptRoot + +$py = $null +foreach ($cand in @("py -3.11", "py -3.12", "py -3", "python")) { + $parts = $cand.Split(" ") + $exe = $parts[0] + $rest = $parts[1..($parts.Length - 1)] + try { + $v = & $exe @rest -c "import sys;print('%d.%d'%sys.version_info[:2])" 2>$null + if ($LASTEXITCODE -eq 0 -and $v) { + $minor = [int]($v.Split(".")[1]) + if ($minor -ge 9 -and $minor -le 12) { $py = $cand; Write-Host "Using Python $v ($cand)"; break } + } + } catch { } +} +if (-not $py) { throw "Need Python 3.9-3.12. Install 3.11 from python.org and re-run." } + +$parts = $py.Split(" ") +& $parts[0] @($parts[1..($parts.Length - 1)]) -m venv .venv + +& ".\.venv\Scripts\python.exe" -m pip install --upgrade pip --quiet +& ".\.venv\Scripts\python.exe" -m pip install -r requirements.txt + +Write-Host "" +Write-Host "Done. Next:" -ForegroundColor Green +Write-Host " .\run.ps1 --list-devices" +Write-Host " .\run.ps1 --selftest 12" +Write-Host " .\run.ps1" diff --git a/tests/test_smoke.py b/tests/test_smoke.py new file mode 100644 index 0000000..e09f76a --- /dev/null +++ b/tests/test_smoke.py @@ -0,0 +1,94 @@ +"""Dependency-free smoke tests. Run with: python tests\\test_smoke.py""" + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +import numpy as np + +import livecap + + +def test_hallucination_filter(): + """Stock Whisper-on-silence phrases are dropped, real speech is kept.""" + for bad in ("Tekstitys: YLE 2021", "Kiitos kun katsoit!", "Kiitos.", + "Thanks for watching", "Subtitles by someone", " "): + assert livecap.looks_hallucinated(bad), bad + for good in ("Moi, mita kuuluu tanaan?", + "Nyt ollaan taas sen verran syrjaisilla seuduilla", + "Kiitos kun tulit mukaan, puhutaan seuraavaksi saasta ja " + "siita mita ensi viikolla tapahtuu"): + assert not livecap.looks_hallucinated(good), good + + +def test_resample_to_whisper_rate(): + """48 kHz in, 16 kHz out, still in range, tone preserved.""" + sr = 48000 + t = np.arange(sr) / sr + sig = (0.2 * np.sin(2 * np.pi * 440 * t)).astype(np.float32) + + out = livecap.to_whisper(sig, sr) + + assert out.dtype == np.float32, out.dtype + assert abs(len(out) - livecap.TARGET_SR) <= 2, len(out) + assert np.max(np.abs(out)) <= 1.0 + + spec = np.abs(np.fft.rfft(out)) + freq = np.fft.rfftfreq(len(out), 1 / livecap.TARGET_SR)[int(np.argmax(spec))] + assert 430 < freq < 450, freq + + +def test_quiet_audio_is_gained_up(): + """Very quiet input is normalised toward a level Whisper can use.""" + quiet = (0.01 * np.sin(np.linspace(0, 100, 16000))).astype(np.float32) + out = livecap.to_whisper(quiet, livecap.TARGET_SR) + assert np.max(np.abs(out)) > 0.3, np.max(np.abs(out)) + + +def test_silence_does_not_divide_by_zero(): + out = livecap.to_whisper(np.zeros(16000, dtype=np.float32), livecap.TARGET_SR) + assert np.all(out == 0) + assert len(out) == 16000 + + +def test_captions_file_keeps_last_n_lines(tmp_path=None): + """captions.txt is a rolling window; captions.log keeps everything.""" + import argparse + import os + import tempfile + + cwd = os.getcwd() + with tempfile.TemporaryDirectory() as d: + os.chdir(d) + try: + livecap.HERE = Path(d) + bus = livecap.Bus(argparse.Namespace( + txt="captions.txt", txt_lines=2, translate=False)) + for line in ("eka", "toka", "kolmas"): + bus.publish({"type": "final", "text": line, "tr": "", "ts": 0}) + + assert (Path(d) / "captions.txt").read_text(encoding="utf-8").split() \ + == ["toka", "kolmas"] + log = (Path(d) / "captions.log").read_text(encoding="utf-8") + assert log.count("\n") == 3 and "eka" in log + finally: + os.chdir(cwd) + + +def main(): + tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")] + failed = 0 + for fn in tests: + try: + fn() + print(" PASS %s" % fn.__name__) + except Exception as e: + failed += 1 + print(" FAIL %s: %r" % (fn.__name__, e)) + print("\n%d passed, %d failed" % (len(tests) - failed, failed)) + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(main())