commit e4c47a25354bda546c6b1facb8a1dc26c129143d
equwal <truex@equwal.com>
2026-08-26 03:51:25 -0700
Fix translation silently returning the source language
--anki-translate produced Finnish where English was expected. Two causes,
both silent.
large-v3-turbo, the offline default, is a transcription-only fine-tune: asked
to translate it returns the source language rather than failing. Translation
now routes to a model that supports the task, with --translate-model to
override, and says so when it substitutes. Verified on the same clip: turbo
returned "Aaaah, kyllä se tästä", small returned "Ah, that's it."
Translating one short sentence at a time also starved Whisper of context and
encouraged that echo. The clip is now translated once and aligned to sentences
by timestamp, which is both better and faster than N calls.
Whisper's Finnish-to-English remains weak regardless ("sen verran syrjäisillä
seuduilla" became "the lake of Sennvera"), so the README points at Yomitan
lookups instead of machine-translated card backs.
Also:
- --watch with no folder now reads OBS's own config for the recording path,
honouring Simple vs Advanced output mode. OBS writes its ini files with a
UTF-8 BOM, which makes configparser raise MissingSectionHeaderError; that
was being swallowed, so detection would have failed for everyone.
- start.ps1 runs both halves together.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
README.md | 24 ++++++++-
start.ps1 | 51 +++++++++++++++++++
subtitle.py | 131 ++++++++++++++++++++++++++++++++++++++++++-------
tests/test_subtitle.py | 65 ++++++++++++++++++++++++
4 files changed, 250 insertions(+), 21 deletions(-)
diff --git a/README.md b/README.md
index 13d5e15..6f61f83 100644
--- a/README.md
+++ b/README.md
@@ -34,6 +34,16 @@ mistakes. Replay needs accuracy and does not care about time.
.\setup.ps1
```
+## Start everything
+
+```bash
+.\start.ps1
+```
+
+Runs both halves: live captions, plus a watcher that subtitles every replay clip
+OBS saves and serves it as a mining page. `-Lang ru`, `-LiveOnly`, `-ReplayOnly`
+and `-Model` adjust it. The two halves are also usable separately, below.
+
## Live captions
```bash
@@ -102,10 +112,12 @@ Writes `clip.srt` and `clip.html` next to the clip. The HTML page is the mining
surface: video on the left, every sentence listed as selectable text, click to
seek, `Loop cue` to repeat a line while you work it out.
-Watch your replay folder and subtitle clips automatically as OBS saves them:
+Watch your replay folder and subtitle clips automatically as OBS saves them.
+With no folder given, it reads OBS's own config to find where recordings go
+(honouring Simple vs Advanced output mode):
```bash
-.\.venv\Scripts\python.exe subtitle.py --watch "C:\Users\you\Videos" --serve
+.\.venv\Scripts\python.exe subtitle.py --watch --serve
```
`--serve` matters more than it looks. Opening `clip.html` from `file://`
@@ -132,6 +144,14 @@ its own, which is usually what you want.
Screenshots need Pillow (`pip install pillow`); without it the image column is
left empty and everything else still works.
+**Two things about `--anki-translate`.** First, `large-v3-turbo` is a
+transcription-only fine-tune and *cannot translate* — asked to, it silently
+returns the source language. Translation is therefore routed to `small` unless
+you set `--translate-model`. Second, Whisper's Finnish→English is genuinely
+weak: it rendered *"sen verran syrjäisillä seuduilla"* as "the lake of Sennvera".
+For learning, Yomitan's dictionary lookups are far more trustworthy than a
+machine-translated card back.
+
## Speed, honestly
Measured on a 16-core CPU, no CUDA, `int8`, on real Finnish speech:
diff --git a/start.ps1 b/start.ps1
new file mode 100644
index 0000000..2866a1d
--- /dev/null
+++ b/start.ps1
@@ -0,0 +1,51 @@
+# Starts both halves: live captions now, and automatic subtitling of every
+# replay clip OBS saves.
+#
+# .\start.ps1 # Finnish
+# .\start.ps1 -Lang ru # Russian
+# .\start.ps1 -Lang ja -Model base
+# .\start.ps1 -LiveOnly
+#
+# Two windows open. Close either to stop that half.
+
+param(
+ [string]$Lang = "fi",
+ [string]$Model = "small",
+ [string]$ReplayModel = "large-v3-turbo",
+ [string]$WatchFolder = "",
+ [int]$ServePort = 8778,
+ [switch]$LiveOnly,
+ [switch]$ReplayOnly
+)
+
+$ErrorActionPreference = "Stop"
+Set-Location $PSScriptRoot
+$py = ".\.venv\Scripts\python.exe"
+if (-not (Test-Path $py)) { throw "Run .\setup.ps1 first." }
+
+if (-not $ReplayOnly) {
+ Write-Host "Starting live captions ($Lang, $Model)..." -ForegroundColor Cyan
+ Start-Process -FilePath $py `
+ -ArgumentList ".\livecap.py", "--lang", $Lang, "--model", $Model `
+ -WorkingDirectory $PSScriptRoot
+ Start-Sleep -Seconds 2
+ Write-Host " OBS overlay http://127.0.0.1:8777/overlay.html?ws=8765"
+ Write-Host " Reader http://127.0.0.1:8777/reader.html?ws=8765" -ForegroundColor Green
+ Write-Host " Control http://127.0.0.1:8777/control.html?ws=8765"
+}
+
+if (-not $LiveOnly) {
+ $watchArgs = @(".\subtitle.py", "--lang", $Lang, "--model", $ReplayModel,
+ "--serve", "$ServePort", "--watch")
+ if ($WatchFolder) { $watchArgs += $WatchFolder }
+
+ Write-Host ""
+ Write-Host "Starting replay watcher ($ReplayModel)..." -ForegroundColor Cyan
+ Start-Process -FilePath $py -ArgumentList $watchArgs -WorkingDirectory $PSScriptRoot
+ Start-Sleep -Seconds 2
+ Write-Host " Clips http://127.0.0.1:$ServePort/" -ForegroundColor Green
+ Write-Host " Save a replay in OBS and its mining page appears there."
+}
+
+Write-Host ""
+Write-Host "Open the Reader and the Clips pages in the browser where Yomitan lives." -ForegroundColor Yellow
diff --git a/subtitle.py b/subtitle.py
index 91aedb7..298f614 100644
--- a/subtitle.py
+++ b/subtitle.py
@@ -503,6 +503,32 @@ class Subtitler:
compute_type=args.compute,
cpu_threads=args.threads, num_workers=1)
log("model ready in %.1fs" % (time.time() - t0))
+ self._tmodel = None
+ self._tname = None
+
+ def translator(self):
+ """A model that can actually translate.
+
+ large-v3-turbo is a transcription-only fine-tune: asked to translate it
+ returns the source language instead, silently. Fall back to a model
+ that supports the task rather than emitting untranslated text.
+ """
+ from faster_whisper import WhisperModel
+
+ a = self.args
+ name = a.translate_model or a.model
+ if not a.translate_model and "turbo" in a.model.lower():
+ name = "small"
+ log("note: %s cannot translate (transcription-only fine-tune); "
+ "using %r instead - override with --translate-model" % (a.model, name))
+ if name == a.model:
+ return self.model
+ if self._tmodel is None or self._tname != name:
+ log("loading translation model %r ..." % name)
+ self._tmodel = WhisperModel(name, device=a.device, compute_type=a.compute,
+ cpu_threads=a.threads, num_workers=1)
+ self._tname = name
+ return self._tmodel
def run(self, path):
from faster_whisper.audio import decode_audio
@@ -524,7 +550,8 @@ class Subtitler:
return None
t0 = time.time()
- segments, info = self.model.transcribe(
+ engine = self.translator() if a.translate else self.model
+ segments, info = engine.transcribe(
audio,
language=None if a.lang == "auto" else a.lang,
task="translate" if a.translate else "transcribe",
@@ -566,27 +593,82 @@ class Subtitler:
return out
def translate_cues(self, audio, cues):
- """English for the back of each card, translated per sentence."""
+ """English for the back of each card.
+
+ Translating one short sentence at a time starves Whisper of context and
+ it often just echoes the source language back. Translating the whole
+ clip once and aligning by timestamp is both better and faster.
+ """
+ log("translating for card backs...")
+ try:
+ segs, _ = self.translator().transcribe(
+ audio, language=None if self.args.lang == "auto" else self.args.lang,
+ task="translate", beam_size=self.args.beam,
+ temperature=[0.0, 0.2, 0.4],
+ condition_on_previous_text=True, vad_filter=True)
+ segs = list(segs)
+ except Exception as e:
+ log("translation failed: %s" % e)
+ return [""] * len(cues)
+
out = []
- log("translating %d sentences for card backs..." % len(cues))
for start, end, _ in cues:
- lo, hi = max(0, int(start * 16000)), min(len(audio), int(end * 16000))
- clip = audio[lo:hi]
- if len(clip) < 1600:
- out.append("")
- continue
- try:
- segs, _ = self.model.transcribe(
- clip, language=None if self.args.lang == "auto" else self.args.lang,
- task="translate", beam_size=self.args.beam, temperature=0.0,
- condition_on_previous_text=False, vad_filter=False,
- without_timestamps=True)
- out.append(" ".join(s.text.strip() for s in segs).strip())
- except Exception:
- out.append("")
+ parts = [s.text.strip() for s in segs
+ if s.start < end - 0.15 and s.end > start + 0.15]
+ out.append(re.sub(r"\s+", " ", " ".join(parts)).strip())
return out
+def obs_recording_folder():
+ """Where OBS writes recordings and replay-buffer clips, from its own config.
+
+ Which key holds it depends on the output mode, so read Mode first rather
+ than guessing.
+ """
+ import configparser
+
+ root = Path(os.environ.get("APPDATA", "")) / "obs-studio"
+ if not root.is_dir():
+ return None
+
+ profile_dir = None
+ for name in ("user.ini", "global.ini"):
+ cfg = configparser.RawConfigParser(strict=False)
+ try:
+ cfg.read(root / name, encoding="utf-8-sig")
+ profile_dir = cfg.get("Basic", "ProfileDir", fallback=None) or profile_dir
+ except (configparser.Error, OSError):
+ pass
+
+ profiles = root / "basic" / "profiles"
+ candidates = []
+ if profile_dir and (profiles / profile_dir).is_dir():
+ candidates.append(profiles / profile_dir)
+ if profiles.is_dir():
+ candidates.extend(p for p in profiles.iterdir() if p.is_dir())
+
+ for prof in candidates:
+ cfg = configparser.RawConfigParser(strict=False)
+ try:
+ cfg.read(prof / "basic.ini", encoding="utf-8-sig")
+ except (configparser.Error, OSError):
+ continue
+ mode = (cfg.get("Output", "Mode", fallback="") or "").lower()
+ if mode.startswith("adv"):
+ keys = [("AdvOut", "RecFilePath"), ("AdvOut", "FFFilePath"),
+ ("SimpleOutput", "FilePath")]
+ else:
+ keys = [("SimpleOutput", "FilePath"), ("AdvOut", "RecFilePath")]
+ for section, key in keys:
+ raw = cfg.get(section, key, fallback=None)
+ if not raw:
+ continue
+ path = Path(raw.replace("\\\\", "\\").strip())
+ if path.is_dir():
+ return path
+ return None
+
+
class _Limited:
"""Feeds copyfile only the bytes belonging to the requested range."""
@@ -732,8 +814,9 @@ def main():
description="Subtitle recordings and replay clips",
formatter_class=argparse.ArgumentDefaultsHelpFormatter)
p.add_argument("files", nargs="*", help="video/audio files to subtitle")
- p.add_argument("--watch", metavar="FOLDER",
- help="watch a folder and subtitle clips as they appear")
+ p.add_argument("--watch", nargs="?", const="auto", default=None, metavar="FOLDER",
+ help="watch a folder and subtitle clips as they appear; "
+ "with no folder, read OBS's own recording path")
p.add_argument("--poll", type=float, default=3.0, help="watch interval (s)")
p.add_argument("--model", default="large-v3-turbo",
@@ -745,6 +828,9 @@ def main():
p.add_argument("--beam", type=int, default=5)
p.add_argument("--translate", action="store_true",
help="write English subtitles instead of the original language")
+ p.add_argument("--translate-model", default=None,
+ help="model used for translation; defaults to --model, except "
+ "for turbo builds which cannot translate at all")
p.add_argument("--format", default="srt", choices=sorted(WRITERS))
p.add_argument("--width", type=int, default=42, help="max characters per line")
@@ -782,6 +868,13 @@ def main():
if not args.files and not args.watch:
p.error("give some files, or --watch a folder")
+ if args.watch == "auto":
+ found = obs_recording_folder()
+ if not found:
+ p.error("could not find OBS's recording folder; pass --watch FOLDER")
+ args.watch = str(found)
+ log("OBS records to: %s" % args.watch)
+
sub = Subtitler(args)
done = []
diff --git a/tests/test_subtitle.py b/tests/test_subtitle.py
index 9e5d1e3..65f23f4 100644
--- a/tests/test_subtitle.py
+++ b/tests/test_subtitle.py
@@ -140,6 +140,71 @@ def test_esc_strips_tabs_and_newlines():
assert subtitle.esc("a\tb\nc ") == "a b c"
+def test_obs_folder_detection_handles_bom_and_output_mode():
+ """OBS writes its ini files with a UTF-8 BOM, which trips configparser."""
+ import os
+ import tempfile
+
+ with tempfile.TemporaryDirectory() as d:
+ root = Path(d) / "obs-studio"
+ prof = root / "basic" / "profiles" / "Untitled"
+ prof.mkdir(parents=True)
+ recdir = Path(d) / "recordings"
+ recdir.mkdir()
+ simpledir = Path(d) / "simple"
+ simpledir.mkdir()
+
+ (root / "user.ini").write_text(
+ "[Basic]\nProfile=Untitled\nProfileDir=Untitled\n", encoding="utf-8")
+ (prof / "basic.ini").write_text(
+ "[Output]\nMode=Advanced\n\n"
+ "[SimpleOutput]\nFilePath=%s\n\n"
+ "[AdvOut]\nRecFilePath=%s\n"
+ % (str(simpledir).replace("\\", "\\\\"),
+ str(recdir).replace("\\", "\\\\")),
+ encoding="utf-8")
+
+ old = os.environ.get("APPDATA")
+ os.environ["APPDATA"] = d
+ try:
+ found = subtitle.obs_recording_folder()
+ finally:
+ if old is None:
+ os.environ.pop("APPDATA", None)
+ else:
+ os.environ["APPDATA"] = old
+
+ assert found is not None, "BOM or mode parsing failed"
+ assert Path(found) == recdir, (found, recdir)
+
+
+def test_obs_folder_detection_simple_mode():
+ import os
+ import tempfile
+
+ with tempfile.TemporaryDirectory() as d:
+ root = Path(d) / "obs-studio"
+ prof = root / "basic" / "profiles" / "P"
+ prof.mkdir(parents=True)
+ simpledir = Path(d) / "simple"
+ simpledir.mkdir()
+ (root / "user.ini").write_text("[Basic]\nProfileDir=P\n", encoding="utf-8")
+ (prof / "basic.ini").write_text(
+ "[Output]\nMode=Simple\n\n[SimpleOutput]\nFilePath=%s\n"
+ % str(simpledir).replace("\\", "\\\\"), encoding="utf-8")
+
+ old = os.environ.get("APPDATA")
+ os.environ["APPDATA"] = d
+ try:
+ found = subtitle.obs_recording_folder()
+ finally:
+ if old is None:
+ os.environ.pop("APPDATA", None)
+ else:
+ os.environ["APPDATA"] = old
+ assert Path(found) == simpledir, found
+
+
def test_range_requests_are_served():
"""Without 206 support a browser will not seek in a <video> at all."""
import http.client