commit 5d555b7bfd9c419d5eb11e99e8457dbd79c5f6ce equwal <truex@equwal.com> 2026-08-26 17:26:15 -0700 Handle Japanese and Russian properly Japanese subtitle lines never wrapped. wrap() splits on spaces and Japanese has none, so every cue came out as one long line that ran off the screen. Added character-level breaking with basic kinsoku shori: closing punctuation, small kana and the long-vowel mark cannot begin a line, opening brackets cannot end one, and breaks land after 。or 、 where possible. Whisper's silence hallucinations are language-specific, and the stock filter only knew Finnish and English ones. Japanese fills silence with "ご視聴ありがとうございました" and "チャンネル登録お願いします" constantly; Russian with "Субтитры сделал ..." and "Спасибо за просмотр". Added both, plus filler-only utterances. Verified the filter still passes real sentences in all three languages. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
livecap.py | 16 ++++++++++++++++ subtitle.py | 41 ++++++++++++++++++++++++++++++++++++++++- tests/test_subtitle.py | 31 +++++++++++++++++++++++++++++++ 3 files changed, 87 insertions(+), 1 deletion(-)
diff --git a/livecap.py b/livecap.py index 8106225..3e59794 100644 --- a/livecap.py +++ b/livecap.py @@ -46,14 +46,30 @@ BLOCK_MS = 32 # Whisper invents these when fed silence or noise. Only applied to short results. HALLUCINATIONS = [ + # Finnish r"^tekstitys", r"^tekstityksen tuotti", r"^k[aa]a?nn[oo]s", r"^kiitos( kun katsoit| paljon| katsomisesta)?[.!]?$", r"^suomennos", + # Japanese - Whisper emits these constantly over silence + r"ご(視聴|清聴)(いただき)?ありがとうございま", + r"チャンネル登録", + r"^字幕", + r"^おわり[。]?$", + r"^(お|ご)?しまい[。]?$", + r"^ありがとうございま(した|す)[。]?$", + r"^(えー|あー|ん|う)+[ー。]?$", + # Russian + r"^субтитры", + r"^спасибо за просмотр", + r"^продолжение следует", + r"^редактор субтитров", + # English / generic r"^subtitles? by", r"^amara\.org", r"^thanks? for watching", + r"^please subscribe", r"^\W*$", ] HALLUCINATION_RE = [re.compile(p, re.I) for p in HALLUCINATIONS] diff --git a/subtitle.py b/subtitle.py index 298f614..7a062fd 100644 --- a/subtitle.py +++ b/subtitle.py @@ -51,6 +51,42 @@ def ts(seconds, sep=","): return "%02d:%02d:%02d%s%03d" % (h, m, s, sep, ms) +# Characters that may not begin a line (kinsoku shori). Closing punctuation, +# small kana and the long-vowel mark all cling to what precedes them. +NO_LINE_START = "、。!?…」』)】〉》・ゃゅょっぁぃぅぇぉーヵヶャュョッァィゥェォ,.!?:;" +NO_LINE_END = "「『(【〈《" + + +def looks_cjk(text): + return any("" <= c <= "ヿ" or "一" <= c <= "鿿" + or "" <= c <= "" for c in text) + + +def wrap_cjk(text, width): + """Break Japanese/Chinese text, which has no spaces to split on.""" + if len(text) <= width: + return text + mid = len(text) / 2 + best = None + for i in range(1, len(text)): + if text[i] in NO_LINE_START or text[i - 1] in NO_LINE_END: + continue + if max(i, len(text) - i) > width: + continue + score = abs(i - mid) + # Prefer breaking straight after sentence punctuation. + if text[i - 1] in "、。!?": + score -= width / 2 + if best is None or score < best[0]: + best = (score, i) + if best is None: + i = max(1, min(len(text) - 1, int(mid))) + while i < len(text) - 1 and text[i] in NO_LINE_START: + i += 1 + return text[:i] + "\n" + text[i:] + return text[:best[1]] + "\n" + text[best[1]:] + + def wrap(text, width): """Split into at most two balanced lines, the way subtitles are normally set.""" text = " ".join(text.split()) @@ -58,7 +94,10 @@ def wrap(text, width): return text words = text.split() if len(words) < 2: - return text + return wrap_cjk(text, width) if looks_cjk(text) else text + if looks_cjk(text) and len(words) < len(text) / 8: + # Mostly CJK with a stray space or two: character breaking reads better. + return wrap_cjk(text, width) fits, over = None, None for i in range(1, len(words)): diff --git a/tests/test_subtitle.py b/tests/test_subtitle.py index 65f23f4..b13d3ab 100644 --- a/tests/test_subtitle.py +++ b/tests/test_subtitle.py @@ -61,6 +61,37 @@ def test_wrap_single_word_cannot_split(): assert "\n" not in subtitle.wrap("Rindfleischetikettierungsgesetz", 5) +def test_wrap_japanese_has_no_spaces_to_split_on(): + text = "今日はとてもいい天気ですね。散歩に行きましょうか。" + out = subtitle.wrap(text, 16) + lines = out.split("\n") + assert len(lines) == 2, out + assert max(len(x) for x in lines) <= 16, out + assert "".join(lines) == text + + +def test_wrap_japanese_prefers_breaking_after_punctuation(): + text = "今日はいい天気。散歩に行こう。" + out = subtitle.wrap(text, 10) + assert out.split("\n")[0].endswith("。"), out + + +def test_wrap_japanese_never_starts_a_line_with_closing_marks(): + for text in ["これはテストです、そしてこれも試験です。", + "彼は「そうだね」と言ったのでした。", + "ちょっとまってっていったよね。"]: + out = subtitle.wrap(text, 10) + for line in out.split("\n")[1:]: + assert line[0] not in subtitle.NO_LINE_START, (text, out) + + +def test_looks_cjk(): + assert subtitle.looks_cjk("今日は") + assert subtitle.looks_cjk("テスト") + assert not subtitle.looks_cjk("Kyllä se tästä") + assert not subtitle.looks_cjk("Привет") + + def test_sentences_split_on_terminators(): ws = words("Yksi kaksi.") + words("Kolme nelja.", t0=2.0) groups = subtitle.group_sentences(ws, max_dur=6.0, max_gap=0.8)