commit 542f27364d5ad0bc458514d1c965f43ae214915f equwal <truex@equwal.com> 2026-08-26 18:47:41 -0700 Constrain automatic language detection to the languages in use Whisper answers with any of 99 languages, decides afresh every call, and has little evidence in a short utterance, so --lang auto flipped mid-conversation and picked languages that are never watched. Detection is now restricted to --langs, with close relatives folded in: Estonian counts toward Finnish, Ukrainian and Bulgarian toward Russian, Galician toward Portuguese, Catalan toward Spanish. Without folding a clip can lose because fi and et split the vote and a third language wins outright. Evidence then accumulates across segments weighted by duration, and a challenger must lead by a margin for several consecutive detections before it takes over. Restriction alone made things worse, which the real-audio test caught: deleting the competitors inflates confidence, so "en 0.38, ko 0.25, nn 0.10" becomes a commanding en 0.85 and the session locks onto English from one ambiguous leading window. How much mass lands inside the allowed set is the honest signal, so a reading that mostly falls outside it is downweighted and never adopted - the detector returns None and lets Whisper handle that segment. On a Finnish clip sampled in short windows, plain Whisper committed to en on the ambiguous window; this declines to guess there and holds fi across all eleven remaining windows. Detection runs on a schedule rather than per segment. It costs an encoder pass, but passing an explicit language to transcribe() skips Whisper's internal detection, so steady-state cost is about neutral. The resolved language is published so the control panel can show "auto -> ja". Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
README.md | 42 +++++++++++++- control.html | 9 ++- livecap.py | 158 +++++++++++++++++++++++++++++++++++++++++++++++++++- tests/test_smoke.py | 117 ++++++++++++++++++++++++++++++++++++++ 4 files changed, 318 insertions(+), 8 deletions(-)
diff --git a/README.md b/README.md index 94205bd..acc9508 100644 --- a/README.md +++ b/README.md @@ -86,9 +86,45 @@ for a few seconds. Defaults cover `fi,ru,ja,es,pt,en,auto`: toward standard orthography and will not reliably reproduce regional slang. Brazilian and European Portuguese are both `pt`. -**`auto` re-detects per segment**, which flips on short or noisy audio and can -mislabel mid-sentence. For deliberate language changes the `1`–`9` keys are far -more reliable. +### How `auto` decides + +Whisper's own detection answers with any of 99 languages, decides afresh on +every call, and has little to go on in a two-word utterance. `auto` here is +built on top of it and constrained to the languages you actually configured: + +1. **Restricted to `--langs`.** A Finnish clip cannot come back as Estonian. +2. **Close relatives are folded in.** Whisper splits mass between neighbours, + so Estonian counts toward Finnish, Ukrainian and Bulgarian toward Russian, + Galician toward Portuguese, Catalan toward Spanish. Otherwise a clip can + lose because `fi` and `et` split the vote and a third language wins. +3. **Reliability gate.** Restricting the set deletes the competitors and so + inflates confidence — `en 0.38, ko 0.25, nn 0.10` looks like a commanding + `en` once `ko` and `nn` are dropped. If too little mass lands inside your + set, it declines to guess and lets Whisper handle that one segment. +4. **Evidence accumulates over time**, weighted by segment length, instead of + each segment deciding alone. +5. **Hysteresis.** A new language must lead by a margin for several + consecutive detections before it takes over, so one odd segment cannot + flip the caption language mid-conversation. + +Detection runs on a schedule (`--detect-every`, default 6 s) rather than every +segment. It costs an encoder pass, but passing an explicit language into +Whisper skips its *internal* detection, so the steady-state cost is roughly +neutral. + +Measured on a Finnish clip in short windows, plain Whisper committed to `en` +on an ambiguous leading window; the detector declined to guess there and then +held `fi` across every remaining window. + +| flag | default | effect | +|---|---|---| +| `--detect-every` | `6.0` | seconds between detection passes | +| `--detect-min-audio` | `1.6` | skip detection on shorter segments | +| `--detect-margin` | `1.3` | how far a challenger must lead to switch | +| `--detect-hold` | `2` | consecutive detections before switching | + +The control panel shows `auto → ja` once it settles. Pinning with `1`–`9` is +still the most reliable option when you already know the language. ### Choosing what gets captioned diff --git a/control.html b/control.html index 4e391e4..29a8be0 100644 --- a/control.html +++ b/control.html @@ -154,8 +154,13 @@ const dot = document.createElement("span"); dot.className = "dot " + (state.loading ? "busy" : "up"); el("conn").appendChild(dot); - el("conn").appendChild(document.createTextNode( - (NAMES[state.lang] || state.lang) + " · " + state.model)); + let label = NAMES[state.lang] || state.lang; + if (state.lang === "auto") { + label += state.detected + ? " → " + (NAMES[state.detected] || state.detected) + : " (listening…)"; + } + el("conn").appendChild(document.createTextNode(label + " · " + state.model)); } function renderFeed() { diff --git a/livecap.py b/livecap.py index 692043b..36796fb 100644 --- a/livecap.py +++ b/livecap.py @@ -323,11 +323,13 @@ class Controller: self.tr = None # set once the Transcriber exists self.cmds = queue.Queue() # model swaps run on the worker thread self.loading = False + self.detected = None # what 'auto' has settled on def status(self): return { "type": "status", "lang": self.args.lang, + "detected": self.detected, "model": self.args.model, "loading": self.loading, "translate": bool(self.args.translate), @@ -352,8 +354,12 @@ class Controller: return if val != self.args.lang: self.args.lang = val + self.detected = None if self.tr is not None: self.tr.context = "" # old-language prompt would poison it + # Evidence gathered for a different language is worthless. + self.tr.detector.reset() + self.tr._last_detect = 0.0 log("language -> %s" % val) self.bus.publish({"type": "clear"}) self.push_status() @@ -388,6 +394,105 @@ class Controller: self.push_status() +# Whisper spreads probability across close relatives. When the relative is not +# one of the languages in use, its mass belongs to the neighbour that is - +# otherwise a Finnish clip can lose to Finnish-plus-Estonian splitting the vote. +CONFUSABLE = { + "et": "fi", # Estonian + "uk": "ru", "be": "ru", "bg": "ru", "mk": "ru", # Cyrillic Slavic + "sr": "ru", "kk": "ru", + "gl": "pt", # Galician leans Portuguese + "ca": "es", "oc": "es", "an": "es", # Iberian Romance + "cy": "en", "gd": "en", # frequent English mis-picks +} + + +class LanguageDetector: + """Language identification restricted to the languages actually in use. + + Three things make Whisper's own per-window detection unreliable here: + it may answer with any of 99 languages, it decides afresh every call so the + answer flaps mid-conversation, and short utterances carry little evidence. + + So: fold the distribution onto the allowed set, accumulate it over time with + a decay, and only switch when a challenger leads by a margin for several + observations in a row. + """ + + def __init__(self, allowed, margin=1.3, hold=2, decay=0.55, min_audio=1.6, + min_mass=0.5, min_confidence=0.55): + self.allowed = [c for c in allowed if c != "auto"] or ["en"] + self.margin = margin + self.hold = hold + self.decay = decay + self.min_audio = min_audio + # Restricting the distribution deletes the competitors, which inflates + # confidence: "en 0.38, ko 0.25, nn 0.10" becomes a commanding en 0.85. + # How much mass landed inside the allowed set is the honest signal of + # whether this audio is any of these languages at all. + self.min_mass = min_mass + self.min_confidence = min_confidence + self.scores = {c: 0.0 for c in self.allowed} + self.current = None + self._pending = None + self._pending_n = 0 + + def reset(self): + self.scores = {c: 0.0 for c in self.allowed} + self.current = None + self._pending, self._pending_n = None, 0 + + def fold(self, probs): + """Restrict the distribution to the allowed set, returning (dict, mass).""" + out = {c: 0.0 for c in self.allowed} + for lang, p in probs: + target = lang if lang in out else CONFUSABLE.get(lang) + if target in out: + out[target] += p + return out, sum(out.values()) + + def confidence(self): + total = sum(self.scores.values()) + if total <= 0 or self.current is None: + return 0.0 + return self.scores[self.current] / total + + def observe(self, probs, duration=3.0): + """Feed one detection result; returns the language to actually use.""" + folded, mass = self.fold(probs) + if mass <= 0: + return self.current + + # Longer audio is better evidence than a two-word utterance, and mass + # outside the allowed set means this probably is not one of them. + reliability = min(1.0, mass / self.min_mass) if self.min_mass > 0 else 1.0 + weight = max(0.05, min(1.0, duration / 3.0)) * reliability + for code in self.scores: + self.scores[code] = (self.scores[code] * self.decay + + (folded[code] / mass) * weight) + + best = max(self.scores, key=lambda c: self.scores[c]) + if self.current is None: + # Adopting a language from one ambiguous reading is how a whole + # session ends up locked to the wrong one. + total = sum(self.scores.values()) + share = self.scores[best] / total if total > 0 else 0.0 + if mass >= self.min_mass and share >= self.min_confidence: + self.current = best + return self.current + + leader, held = self.scores[best], self.scores[self.current] + if best != self.current and leader > held * self.margin: + self._pending_n = self._pending_n + 1 if self._pending == best else 1 + self._pending = best + if self._pending_n >= self.hold: + self.current = best + self._pending, self._pending_n = None, 0 + else: + self._pending, self._pending_n = None, 0 + return self.current + + def valid_language(code): if code == "auto": return True @@ -469,6 +574,42 @@ class Transcriber: log("model ready in %.1fs" % (time.time() - t0)) self.context = "" self.busy = threading.Event() + self.detector = LanguageDetector( + args.langs, margin=args.detect_margin, hold=args.detect_hold, + min_audio=args.detect_min_audio) + self._last_detect = 0.0 + + def pick_language(self, audio, duration, ctl=None): + """Resolve 'auto' to one of the configured languages. + + Detection runs on a schedule rather than every segment: it costs an + encoder pass, and once the answer is settled re-deciding constantly is + what makes it flap. Passing an explicit language into transcribe() also + skips Whisper's own internal detection, so this is close to free. + """ + a = self.args + if a.lang != "auto": + return a.lang + + due = (self.detector.current is None + or time.time() - self._last_detect >= a.detect_every) + if due and duration >= self.detector.min_audio: + self._last_detect = time.time() + try: + _, _, probs = self.model.detect_language(audio) + except Exception as e: + log("language detection failed:", repr(e)) + return self.detector.current + before = self.detector.current + now = self.detector.observe(probs, duration) + if now != before: + log("detected language: %s (confidence %.2f)" + % (now, self.detector.confidence())) + self.context = "" # prompt from another language misleads + if ctl is not None: + ctl.detected = now + ctl.push_status() + return self.detector.current def reload(self, name, ctl): """Swap the model in place. Runs on the worker thread.""" @@ -495,11 +636,12 @@ class Transcriber: ctl.push_status() log("model -> %s (%.1fs)" % (name, time.time() - t0)) - def run(self, audio, final): + def run(self, audio, final, language=None): a = self.args + lang = language or (None if a.lang == "auto" else a.lang) segs, info = self.model.transcribe( audio, - language=None if a.lang == "auto" else a.lang, + language=lang, task="transcribe", beam_size=a.beam if final else 1, temperature=0.0, @@ -556,7 +698,9 @@ class Transcriber: t0 = time.time() a16 = to_whisper(audio, sr) dur = len(a16) / TARGET_SR - text, nsp, _ = self.run(a16, final=(kind == "final")) + lang = (self.pick_language(a16, dur, ctl) if kind == "final" + else self.detector.current) + text, nsp, _ = self.run(a16, final=(kind == "final"), language=lang) if not text: if kind == "final": log("no speech recognised in %.1fs segment " @@ -942,6 +1086,14 @@ def build_parser(): g.add_argument("--compute", default="int8", help="int8|int8_float32|float32|float16") g.add_argument("--threads", type=int, default=max(2, (os.cpu_count() or 8) - 2)) g.add_argument("--beam", type=int, default=5) + g.add_argument("--detect-every", type=float, default=6.0, + help="seconds between language-detection passes when --lang auto") + g.add_argument("--detect-min-audio", type=float, default=1.6, + help="skip detection on segments shorter than this") + g.add_argument("--detect-margin", type=float, default=1.3, + help="how far a new language must lead before switching") + g.add_argument("--detect-hold", type=int, default=2, + help="consecutive detections needed before switching") g.add_argument("--repetition-penalty", type=float, default=1.15, help="discourage Whisper from looping on a phrase; " "Japanese and Chinese need this more than European " diff --git a/tests/test_smoke.py b/tests/test_smoke.py index 6b9d9b2..41686ba 100644 --- a/tests/test_smoke.py +++ b/tests/test_smoke.py @@ -112,6 +112,123 @@ def test_is_repetitive(): assert not livecap.is_repetitive("hi") +LANGS = ["fi", "ru", "ja", "es", "pt", "en"] + + +def _det(**kw): + return livecap.LanguageDetector(LANGS + ["auto"], **kw) + + +def test_detector_ignores_languages_not_in_use(): + """An outsider may win outright and still not be the answer. + + But only when enough evidence lands inside the set - if the audio is + overwhelmingly a language not in use, the honest answer is "no idea", + not the best of the leftovers. + """ + d = _det() + # de leads, yet fi holds the majority of the mass that is actually usable + assert d.observe([("de", 0.3), ("fi", 0.55), ("ja", 0.05)], duration=4) == "fi" + + d2 = _det() + # here de dominates and almost nothing is in the set: refuse to guess + assert d2.observe([("de", 0.7), ("fi", 0.2), ("ja", 0.05)], duration=4) is None + + +def test_detector_folds_close_relatives(): + """Estonian mass belongs to Finnish; Ukrainian to Russian; Galician to Portuguese.""" + d = _det() + folded, mass = d.fold([("et", 0.5), ("fi", 0.3), ("ru", 0.1)]) + assert folded["fi"] == 0.8, folded + assert 0.89 < mass < 0.91, mass + + folded, _ = d.fold([("uk", 0.4), ("bg", 0.2), ("ru", 0.1)]) + assert abs(folded["ru"] - 0.7) < 1e-9, folded + + folded, _ = d.fold([("gl", 0.6), ("es", 0.2)]) + assert abs(folded["pt"] - 0.6) < 1e-9 and abs(folded["es"] - 0.2) < 1e-9, folded + + +def test_detector_splits_do_not_lose_to_an_outsider(): + """fi+et together beat ja, even though ja beats each individually.""" + d = _det() + assert d.observe([("fi", 0.3), ("et", 0.3), ("ja", 0.35)], duration=4) == "fi" + + +def test_detector_does_not_flap_on_one_odd_segment(): + d = _det(hold=2) + for _ in range(4): + d.observe([("ja", 0.95)], duration=4) + assert d.current == "ja" + # a single confident Spanish reading must not switch it + assert d.observe([("es", 0.95)], duration=4) == "ja" + + +def test_detector_switches_when_change_is_sustained(): + d = _det(hold=2) + for _ in range(4): + d.observe([("ja", 0.95)], duration=4) + assert d.current == "ja" + d.observe([("ru", 0.95)], duration=4) + d.observe([("ru", 0.95)], duration=4) + for _ in range(3): + d.observe([("ru", 0.95)], duration=4) + assert d.current == "ru", d.scores + + +def test_detector_weights_short_audio_less(): + d = _det() + for _ in range(3): + d.observe([("en", 0.9)], duration=5) + strong = dict(d.scores) + d2 = _det() + for _ in range(3): + d2.observe([("en", 0.9)], duration=0.5) + assert strong["en"] > d2.scores["en"] + + +def test_detector_reset_and_confidence(): + d = _det() + d.observe([("ja", 0.99)], duration=4) + assert d.current == "ja" and d.confidence() > 0.9 + d.reset() + assert d.current is None and d.confidence() == 0.0 + assert all(v == 0.0 for v in d.scores.values()) + + +def test_detector_defers_on_an_ambiguous_first_reading(): + """Restricting the set inflates confidence; mass outside it is the tell. + + "en 0.38, ko 0.25, nn 0.10" looks like a commanding en once ko and nn are + dropped, but only 0.38 of the mass was ever inside the allowed set. + """ + d = _det() + assert d.observe([("en", 0.38), ("ko", 0.25), ("nn", 0.10)], duration=2) is None + assert d.current is None + # a genuinely confident reading is adopted immediately afterwards + assert d.observe([("fi", 0.96), ("nn", 0.02), ("en", 0.01)], duration=4) == "fi" + + +def test_detector_adopts_a_confident_first_reading(): + d = _det() + assert d.observe([("ja", 0.93), ("zh", 0.04)], duration=4) == "ja" + + +def test_detector_low_mass_never_locks_in(): + d = _det() + for _ in range(5): + d.observe([("de", 0.6), ("nl", 0.3), ("en", 0.05)], duration=4) + assert d.current is None, d.scores + + +def test_detector_handles_empty_and_unusable_input(): + d = _det() + assert d.observe([], duration=4) is None + assert d.observe([("de", 0.9), ("zh", 0.1)], duration=4) is None + d.observe([("fi", 0.9)], duration=4) + assert d.observe([("de", 1.0)], duration=4) == "fi" # keeps the last good one + + def test_model_resolution(): """An explicit --model must win, even when it equals the default.""" import tempfile