Recently Written · git

desktop-subtitle-replay

git clone https://github.com/equwal/desktop-subtitle-replay

Log | Files | Refs


commit ca1a576e27c7e45fe5ff5a5c4b14552923c1fa6c
equwal <truex@equwal.com>
2026-08-26 17:32:02 -0700

Stop Whisper looping on a phrase, which Japanese triggers constantly

Japanese captions degenerated into the same sentence repeated - one segment
came back as "私はそれを見つけたことがありました" four times - and decoding
slowed to RTF 1.2-2.2 because the loop generates tokens indefinitely.

Part of this was self-inflicted: the previous caption is fed back as
initial_prompt for continuity, so once a looped caption is emitted it primes
the model to loop again on the next segment. A caption detected as repetitive
now clears that context instead of seeding it.

Also added a decoder-level repetition_penalty (default 1.15, --repetition-penalty
to tune) and a post-filter. The filter handles both observed shapes: identical
chunks joined across segments, and a unit looping to the end of a chunk. The
latter needed anchoring at the end rather than the start, because the loop does
not always begin at the beginning - "この日の日の日の日" is "こ" then "の日"
four times. Whole phrases collapse on a single repeat; short words need three
in a row, since doubling is normal speech.

After this, RTF on the same machine returned to 0.45-0.88 with no repeats.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

 livecap.py          | 60 ++++++++++++++++++++++++++++++++++++++++++++++++++++-
 tests/test_smoke.py | 36 ++++++++++++++++++++++++++++++++
 2 files changed, 95 insertions(+), 1 deletion(-)
diff --git a/livecap.py b/livecap.py
index 3e59794..692043b 100644
--- a/livecap.py
+++ b/livecap.py
@@ -400,6 +400,53 @@ def valid_language(code):
         return bool(re.fullmatch(r"[a-z]{2,3}", code))
 
 
+def collapse_looping_tail(s, min_reps=3):
+    """Trim a unit that repeats to the end of the string.
+
+    The loop does not necessarily start at the beginning - "この日の日の日の日"
+    is "こ" followed by "の日" four times - so anchor the search at the end.
+    """
+    n = len(s)
+    for p in range(1, n // min_reps + 1):
+        unit = s[n - p:]
+        k = 0
+        while (k + 1) * p <= n and s[n - (k + 1) * p: n - k * p] == unit:
+            k += 1
+        if k >= min_reps:
+            return s[: n - (k - 1) * p]
+    return s
+
+
+def collapse_repeats(text):
+    """Whisper degenerates into repeating a phrase; keep one copy.
+
+    Two shapes show up: identical chunks joined from consecutive segments
+    ("A A A"), and a unit looped inside one chunk ("この日の日の日").
+    """
+    s = " ".join(text.split())
+    if not s:
+        return s
+
+    out = []
+    for c in s.split(" "):
+        # Only whole phrases collapse on a single repeat; short words legitimately
+        # repeat ("very very good"), so those need three in a row.
+        run = 1 if len(c) >= 8 else 2
+        if len(out) >= run and all(x == c for x in out[-run:]):
+            continue
+        out.append(c)
+
+    return " ".join(collapse_looping_tail(c) for c in out)
+
+
+def is_repetitive(text):
+    """True when a caption is mostly one phrase repeated."""
+    s = " ".join(text.split())
+    if len(s) < 8:
+        return False
+    return len(collapse_repeats(s)) * 2 < len(s)
+
+
 def looks_hallucinated(text):
     t = text.strip().lower()
     if len(t) > 40:
@@ -461,6 +508,7 @@ class Transcriber:
             vad_filter=final,
             no_speech_threshold=0.6,
             log_prob_threshold=-1.0,
+            repetition_penalty=a.repetition_penalty,
             without_timestamps=True,
         )
         parts, nsp = [], []
@@ -518,8 +566,14 @@ class Transcriber:
                     if nsp > 0.75 or looks_hallucinated(text):
                         log("dropped (no_speech=%.2f): %r" % (nsp, text))
                         continue
+                    looped = is_repetitive(text)
+                    text = collapse_repeats(text)
+                    if looped:
+                        # Feeding a looped caption back as the prompt is how the
+                        # loop sustains itself across segments.
+                        self.context = ""
                     tr = self.translate(a16) if self.args.translate else ""
-                    if not self.args.no_context:
+                    if not self.args.no_context and not looped:
                         self.context = (self.context + " " + text)[-220:]
                     took = time.time() - t0
                     log("FINAL %4.1fs audio in %4.1fs (rtf %.2f)  %s"
@@ -888,6 +942,10 @@ def build_parser():
     g.add_argument("--compute", default="int8", help="int8|int8_float32|float32|float16")
     g.add_argument("--threads", type=int, default=max(2, (os.cpu_count() or 8) - 2))
     g.add_argument("--beam", type=int, default=5)
+    g.add_argument("--repetition-penalty", type=float, default=1.15,
+                   help="discourage Whisper from looping on a phrase; "
+                        "Japanese and Chinese need this more than European "
+                        "languages. 1.0 disables it")
     g.add_argument("--translate", action="store_true",
                    help="also emit an English translation line")
     g.add_argument("--no-context", action="store_true",
diff --git a/tests/test_smoke.py b/tests/test_smoke.py
index 7b2ccb0..6b9d9b2 100644
--- a/tests/test_smoke.py
+++ b/tests/test_smoke.py
@@ -76,6 +76,42 @@ def test_captions_file_keeps_last_n_lines(tmp_path=None):
             os.chdir(cwd)
 
 
+def test_collapse_repeated_chunks():
+    """Consecutive segments repeating the same sentence collapse to one."""
+    one = "私はそれを見つけたことがありました"
+    assert livecap.collapse_repeats(" ".join([one] * 4)) == one
+    assert livecap.collapse_repeats(one) == one
+
+
+def test_collapse_repeated_unit_inside_a_segment():
+    """The loop need not start at the beginning of the string."""
+    assert livecap.collapse_repeats("この日の日の日の日") == "この日"
+    assert livecap.collapse_repeats("abababab") == "ab"
+    assert livecap.collapse_repeats("xabababab") == "xab"
+
+
+def test_short_words_may_repeat_twice():
+    """Doubling is normal speech; a longer run is Whisper looping."""
+    assert livecap.collapse_repeats("very very good") == "very very good"
+    assert livecap.collapse_repeats("no no no no no") == "no no"
+
+
+def test_collapse_leaves_normal_text_alone():
+    for text in ["今日はとてもいい天気ですね。",
+                 "Kyllä se tästä, kyllä se tästä.",
+                 "Привет, как дела сегодня?",
+                 "the cat sat on the mat"]:
+        assert livecap.collapse_repeats(text) == " ".join(text.split()), text
+
+
+def test_is_repetitive():
+    one = "私はそれを見つけたことがありました"
+    assert livecap.is_repetitive(" ".join([one] * 4))
+    assert not livecap.is_repetitive(one)
+    assert not livecap.is_repetitive("Kyllä se tästä, kyllä se tästä.")
+    assert not livecap.is_repetitive("hi")
+
+
 def test_model_resolution():
     """An explicit --model must win, even when it equals the default."""
     import tempfile