commit 503b41a38f1df97f78c4182243aaf85ec347300e equwal <truex@equwal.com> 2026-08-26 12:20:34 -0700 Use the Finnish fine-tune for live captions when it is built Built and measured it on the 17.6s clip: stock small transcribes "harvokseltaan", the fine-tune and large-v3-turbo both get "harvakseltaan", but the fine-tune runs at RTF 0.21 against turbo's 0.57. Turbo-level accuracy at small's speed, so livecap now prefers it automatically for --lang fi when present. --model therefore defaults to None and resolves in main(), because the old check compared against the literal "small" and so could not tell an explicit --model small from no flag at all - the override silently did nothing. Logic is extracted to resolve_model() and covered by tests. Turbo stays the offline default: the fine-tune punctuates less reliably and runs sentences together, which matters where cues are split on sentence boundaries for subtitles and Anki cards, but not live where the VAD segments. get-finnish-model.ps1 could not actually complete before this. Windows PowerShell turns native stderr into a terminating error under ErrorActionPreference = Stop, and the HF Hub always warns about unauthenticated requests, so the script aborted after conversion and never wrote tokenizer.json - leaving faster-whisper to fall back to the stock whisper-tiny tokenizer. Native calls are now judged by exit code, and the output is verified. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
README.md | 28 ++++++++++++++++++++++------ get-finnish-model.ps1 | 34 +++++++++++++++++++++++++--------- livecap.py | 33 +++++++++++++++++++++++++++++++-- tests/test_smoke.py | 28 ++++++++++++++++++++++++++++ 4 files changed, 106 insertions(+), 17 deletions(-)
diff --git a/README.md b/README.md index 01cb8a5..94205bd 100644 --- a/README.md +++ b/README.md @@ -232,12 +232,28 @@ the same speed, but markedly better at Finnish: .\get-finnish-model.ps1 ``` -```bash -.\run.ps1 --model .\build\models\fi-small-ct2 -``` +Once built, `run.ps1` picks it up automatically whenever `--lang fi` and no +`--model` is given. `--model small` overrides it; other languages ignore it. + +Measured on the same 17.6 s Finnish clip: + +| model | transcription | RTF | +|---|---|---| +| `small` stock | "harv**o**kseltaan kuljetuilla" ✗ | 0.20 | +| **`fi-small-ct2`** | "harv**a**kseltaan kuljetuilla" ✓ | **0.21** | +| `large-v3-turbo` | "harvakseltaan kuljetuilla" ✓ | 0.57 | + +Turbo-level accuracy at stock-`small` speed — 2.7× faster than turbo for the +same correct word. That is why it is the live default when present. + +**Keep turbo for replay clips, though.** The fine-tune punctuates less +reliably, running sentences together, and the offline path splits cues on +sentence boundaries — so turbo yields cleaner subtitle cues and better Anki +card boundaries. Live, the VAD does the segmenting, so this costs nothing. -Needs ~8 GB free and pulls torch for the one-off conversion; use -`-BuildRoot D:\somewhere` to build elsewhere, then delete `.venv-convert`. +The conversion needs ~8 GB free and pulls torch, used only for that one step. +`-BuildRoot D:\somewhere` builds elsewhere; delete `.venv-convert` under the +build root afterwards to reclaim about 5 GB. ## Syncing captions to the picture @@ -258,7 +274,7 @@ If you already run a replay buffer, this costs you nothing. | flag | default | effect | |---|---|---| | `--lang` / `--langs` | `fi` / `fi,ru,ja,es,pt,en,auto` | active language, and the panel's buttons | -| `--model` | `small` | model name or local CTranslate2 directory | +| `--model` | `small`, or the Finnish fine-tune if built | model name or local CTranslate2 directory | | `--mic` / `--audio-device` | loopback / auto | capture an input device instead | | `--pause` | `0.65` | silence that closes a caption; lower is snappier | | `--min-speech` | `0.45` | ignore bursts shorter than this | diff --git a/get-finnish-model.ps1 b/get-finnish-model.ps1 index 8501791..cee7140 100644 --- a/get-finnish-model.ps1 +++ b/get-finnish-model.ps1 @@ -59,31 +59,47 @@ if ($freeGB) { if (-not (Test-Path "$venv\Scripts\python.exe")) { Write-Host "Creating conversion venv (one-time, ~2.5 GB of torch)..." -ForegroundColor Cyan + # pip and venv both write progress to stderr, which would otherwise abort + # this script under ErrorActionPreference = Stop. Check exit codes instead. + $ErrorActionPreference = "Continue" & py -3.11 -m venv $venv + if ($LASTEXITCODE -ne 0) { $ErrorActionPreference = "Stop"; throw "could not create venv at $venv" } & "$venv\Scripts\python.exe" -m pip install --upgrade pip --quiet & "$venv\Scripts\python.exe" -m pip install --index-url https://download.pytorch.org/whl/cpu torch - if ($LASTEXITCODE -ne 0) { throw "torch install failed" } + if ($LASTEXITCODE -ne 0) { $ErrorActionPreference = "Stop"; throw "torch install failed" } & "$venv\Scripts\python.exe" -m pip install transformers ctranslate2 "numpy<3" - if ($LASTEXITCODE -ne 0) { throw "transformers/ctranslate2 install failed" } + if ($LASTEXITCODE -ne 0) { $ErrorActionPreference = "Stop"; throw "transformers/ctranslate2 install failed" } + $ErrorActionPreference = "Stop" } Write-Host "Downloading + converting $src ..." -ForegroundColor Cyan + +# Windows PowerShell turns any native stderr output into a terminating error +# while ErrorActionPreference is Stop, and the HF Hub always warns about +# unauthenticated requests. Judge these calls by their exit code instead. +$ErrorActionPreference = "Continue" + & "$venv\Scripts\ct2-transformers-converter.exe" ` --model $src ` --output_dir $out ` --quantization int8 ` --copy_files preprocessor_config.json tokenizer_config.json special_tokens_map.json ` vocab.json merges.txt normalizer.json added_tokens.json -if ($LASTEXITCODE -ne 0) { throw "conversion failed" } +if ($LASTEXITCODE -ne 0) { $ErrorActionPreference = "Stop"; throw "conversion failed" } # faster-whisper wants a real tokenizer.json, otherwise it quietly falls back to -# the stock openai/whisper-tiny tokenizer. -$py = @" -from transformers import WhisperTokenizerFast -WhisperTokenizerFast.from_pretrained(r'$src').save_pretrained(r'$out') -print('tokenizer.json written') -"@ +# the stock openai/whisper-tiny tokenizer and mis-decodes. +$py = "from transformers import WhisperTokenizerFast" + "`n" + + "WhisperTokenizerFast.from_pretrained(r'$src').save_pretrained(r'$out')" + "`n" + + "print('tokenizer.json written')" $py | & "$venv\Scripts\python.exe" - +if ($LASTEXITCODE -ne 0) { $ErrorActionPreference = "Stop"; throw "tokenizer export failed" } + +$ErrorActionPreference = "Stop" +if (-not (Test-Path (Join-Path $out "model.bin"))) { throw "no model.bin in $out" } +if (-not (Test-Path (Join-Path $out "tokenizer.json"))) { + Write-Host "WARNING: tokenizer.json missing; faster-whisper will fall back " -ForegroundColor Yellow +} Write-Host "" Write-Host "Done -> $out" -ForegroundColor Green diff --git a/livecap.py b/livecap.py index 96303c7..8106225 100644 --- a/livecap.py +++ b/livecap.py @@ -821,6 +821,26 @@ def selftest(args, seconds): # ------------------------------------------------------------------ main ---- +DEFAULT_MODEL = "small" +FI_MODEL_DIR = Path("build") / "models" / "fi-small-ct2" + + +def resolve_model(requested, lang, here): + """Pick the model, returning (name, message-to-log-or-None). + + With nothing requested, prefer the Finnish fine-tune when it has been built + and Finnish is what we are transcribing: same size and speed as small, but + better Finnish. An explicit --model always wins, which is why the flag + defaults to None rather than to "small". + """ + if requested is not None: + return requested, None + local = Path(here) / FI_MODEL_DIR + if lang == "fi" and (local / "model.bin").is_file(): + return str(local), "using the Finnish fine-tune (override with --model small)" + return DEFAULT_MODEL, None + + def build_parser(): p = argparse.ArgumentParser( description="Live captions for OBS", @@ -837,8 +857,11 @@ def build_parser(): g.add_argument("--verbose", action="store_true", help="log VAD segment decisions") g = p.add_argument_group("model") - g.add_argument("--model", default="small", - help="tiny|base|small|medium|large-v3|large-v3-turbo, or a local CT2 dir") + # Default is resolved in main() so an explicit --model small is + # distinguishable from not passing --model at all. + g.add_argument("--model", default=None, + help="tiny|base|small|medium|large-v3|large-v3-turbo, or a local " + "CT2 dir (default: small, or the Finnish fine-tune if built)") g.add_argument("--lang", default="fi", help="fi, ru, ja, es, pt, ... or 'auto'") g.add_argument("--langs", default="fi,ru,ja,es,pt,en,auto", help="languages offered as one-click buttons in the control panel") @@ -890,6 +913,12 @@ def build_parser(): def main(): args = build_parser().parse_args() + + chosen, why = resolve_model(args.model, args.lang, HERE) + args.model = chosen + if why: + log(why) + args.langs = [s.strip() for s in args.langs.split(",") if s.strip()] args.model_choices = [s.strip() for s in args.model_choices.split(",") if s.strip()] if args.model not in args.model_choices: diff --git a/tests/test_smoke.py b/tests/test_smoke.py index d7ed8fe..7b2ccb0 100644 --- a/tests/test_smoke.py +++ b/tests/test_smoke.py @@ -76,6 +76,34 @@ def test_captions_file_keeps_last_n_lines(tmp_path=None): os.chdir(cwd) +def test_model_resolution(): + """An explicit --model must win, even when it equals the default.""" + import tempfile + + with tempfile.TemporaryDirectory() as d: + here = Path(d) + + # nothing built yet + assert livecap.resolve_model(None, "fi", here)[0] == "small" + assert livecap.resolve_model(None, "ru", here)[0] == "small" + + built = here / livecap.FI_MODEL_DIR + built.mkdir(parents=True) + (built / "model.bin").write_bytes(b"x") + + name, why = livecap.resolve_model(None, "fi", here) + assert name == str(built), name + assert why and "fine-tune" in why + + # the fine-tune is Finnish-only + assert livecap.resolve_model(None, "ru", here)[0] == "small" + assert livecap.resolve_model(None, "ja", here)[0] == "small" + + # explicit wins, including the string that happens to be the default + assert livecap.resolve_model("small", "fi", here) == ("small", None) + assert livecap.resolve_model("large-v3", "fi", here)[0] == "large-v3" + + def _controller(): import argparse