tests/test_subtitle.py (10185 bytes)
1 """Tests for the offline subtitler. Run with: python tests\\test_subtitle.py""" 2 3 import sys 4 from pathlib import Path 5 6 sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) 7 8 import subtitle 9 10 11 class W: 12 """Minimal stand-in for a faster-whisper word.""" 13 14 def __init__(self, start, end, word): 15 self.start, self.end, self.word = start, end, word 16 17 18 class Seg: 19 def __init__(self, words, text=None, start=0.0, end=0.0): 20 self.words = words 21 self.text = text if text is not None else "".join(w.word for w in words) 22 self.start, self.end = start, end 23 24 25 def words(spec, t0=0.0, step=0.4): 26 out, t = [], t0 27 for token in spec.split(" "): 28 out.append(W(t, t + step, token + " ")) 29 t += step 30 return out 31 32 33 def test_timestamp_format(): 34 assert subtitle.ts(0) == "00:00:00,000" 35 assert subtitle.ts(1.5) == "00:00:01,500" 36 assert subtitle.ts(3661.25) == "01:01:01,250" 37 assert subtitle.ts(-5) == "00:00:00,000" 38 assert subtitle.ts(1.5, ".") == "00:00:01.500" 39 40 41 def test_wrap_balances_two_lines(): 42 out = subtitle.wrap("aaa bbb ccc ddd", 8) 43 assert out == "aaa bbb\nccc ddd", repr(out) 44 45 46 def test_wrap_leaves_short_text_alone(): 47 assert subtitle.wrap("short", 42) == "short" 48 49 50 def test_wrap_falls_back_when_nothing_fits(): 51 """A long sentence with no split under the limit still gets two lines.""" 52 text = ("Ollaan taas sen verran syrjaisilla seuduilla ja " 53 "harvakseltaan kuljetuilla seuduilla.") 54 out = subtitle.wrap(text, 42) 55 lines = out.split("\n") 56 assert len(lines) == 2, out 57 assert max(len(x) for x in lines) < len(text), out 58 59 60 def test_wrap_single_word_cannot_split(): 61 assert "\n" not in subtitle.wrap("Rindfleischetikettierungsgesetz", 5) 62 63 64 def test_wrap_japanese_has_no_spaces_to_split_on(): 65 text = "今日はとてもいい天気ですね。散歩に行きましょうか。" 66 out = subtitle.wrap(text, 16) 67 lines = out.split("\n") 68 assert len(lines) == 2, out 69 assert max(len(x) for x in lines) <= 16, out 70 assert "".join(lines) == text 71 72 73 def test_wrap_japanese_prefers_breaking_after_punctuation(): 74 text = "今日はいい天気。散歩に行こう。" 75 out = subtitle.wrap(text, 10) 76 assert out.split("\n")[0].endswith("。"), out 77 78 79 def test_wrap_japanese_never_starts_a_line_with_closing_marks(): 80 for text in ["これはテストです、そしてこれも試験です。", 81 "彼は「そうだね」と言ったのでした。", 82 "ちょっとまってっていったよね。"]: 83 out = subtitle.wrap(text, 10) 84 for line in out.split("\n")[1:]: 85 assert line[0] not in subtitle.NO_LINE_START, (text, out) 86 87 88 def test_looks_cjk(): 89 assert subtitle.looks_cjk("今日は") 90 assert subtitle.looks_cjk("テスト") 91 assert not subtitle.looks_cjk("Kyllä se tästä") 92 assert not subtitle.looks_cjk("Привет") 93 94 95 def test_sentences_split_on_terminators(): 96 ws = words("Yksi kaksi.") + words("Kolme nelja.", t0=2.0) 97 groups = subtitle.group_sentences(ws, max_dur=6.0, max_gap=0.8) 98 assert len(groups) == 2, [subtitle.text_of(g) for g in groups] 99 assert subtitle.text_of(groups[0]) == "Yksi kaksi." 100 101 102 def test_sentences_split_on_long_gap(): 103 ws = words("yksi kaksi") + words("kolme nelja", t0=9.0) 104 groups = subtitle.group_sentences(ws, max_dur=6.0, max_gap=0.8) 105 assert len(groups) == 2 106 107 108 def test_cue_never_starts_mid_sentence_orphan(): 109 """A trailing runt is folded back rather than shown alone.""" 110 ws = words("aaaa bbbb cccc dddd eeee ffff gggg hhhh iiii jjjj kkkk") 111 cues = subtitle.build_cues([Seg(ws)], max_chars=30, max_dur=99, max_gap=9) 112 assert cues 113 for _, _, text in cues: 114 assert len(text) >= 16 or len(cues) == 1, cues 115 116 117 def test_segment_without_word_timings_still_produces_a_cue(): 118 seg = Seg([], text="Ei sanatason aikaleimoja.", start=1.0, end=3.0) 119 cues = subtitle.build_cues([seg], 84, 6.0, 0.8) 120 assert len(cues) == 1 121 assert cues[0][2] == "Ei sanatason aikaleimoja." 122 assert cues[0][0] == 1.0 and cues[0][1] == 3.0 123 124 125 def test_srt_roundtrip(tmp=None): 126 import tempfile 127 128 cues = [(0.0, 1.5, "Ensimmainen."), (2.0, 3.25, "Toinen rivi tassa.")] 129 with tempfile.TemporaryDirectory() as d: 130 out = Path(d) / "x.srt" 131 subtitle.write_srt(cues, out, 42) 132 text = out.read_text(encoding="utf-8") 133 assert "1\n00:00:00,000 --> 00:00:01,500\nEnsimmainen." in text, text 134 assert "2\n00:00:02,000 --> 00:00:03,250" in text, text 135 assert text.endswith("\n\n") 136 137 138 def test_vtt_has_header_and_dot_timestamps(): 139 import tempfile 140 141 with tempfile.TemporaryDirectory() as d: 142 out = Path(d) / "x.vtt" 143 subtitle.write_vtt([(0.0, 1.0, "Moi")], out, 42) 144 text = out.read_text(encoding="utf-8") 145 assert text.startswith("WEBVTT") 146 assert "00:00:00.000 --> 00:00:01.000" in text 147 148 149 def test_mine_page_embeds_cues_and_escapes_nothing_odd(): 150 import json 151 import tempfile 152 153 cues = [(0.0, 1.0, "今日はいい天気ですね。"), (1.0, 2.0, "Kyllä se tästä.")] 154 with tempfile.TemporaryDirectory() as d: 155 video = Path(d) / "clip.mp4" 156 video.write_bytes(b"") 157 page = subtitle.write_mine_page(video, cues, "ja") 158 html = page.read_text(encoding="utf-8") 159 assert page.name == "clip.html" 160 assert 'src="clip.mp4"' in html 161 # Non-Latin text must survive verbatim for Yomitan to scan it. 162 assert "今日はいい天気ですね。" in html 163 assert "Kyllä se tästä." in html 164 assert "__CUES__" not in html and "__VIDEO__" not in html 165 start = html.index("const CUES = ") + len("const CUES = ") 166 data = json.loads(html[start:html.index("\n", start)].rstrip(";")) 167 assert len(data) == 2 and data[0]["text"] == "今日はいい天気ですね。" 168 169 170 def test_esc_strips_tabs_and_newlines(): 171 assert subtitle.esc("a\tb\nc ") == "a b c" 172 173 174 def test_obs_folder_detection_handles_bom_and_output_mode(): 175 """OBS writes its ini files with a UTF-8 BOM, which trips configparser.""" 176 import os 177 import tempfile 178 179 with tempfile.TemporaryDirectory() as d: 180 root = Path(d) / "obs-studio" 181 prof = root / "basic" / "profiles" / "Untitled" 182 prof.mkdir(parents=True) 183 recdir = Path(d) / "recordings" 184 recdir.mkdir() 185 simpledir = Path(d) / "simple" 186 simpledir.mkdir() 187 188 (root / "user.ini").write_text( 189 "[Basic]\nProfile=Untitled\nProfileDir=Untitled\n", encoding="utf-8") 190 (prof / "basic.ini").write_text( 191 "[Output]\nMode=Advanced\n\n" 192 "[SimpleOutput]\nFilePath=%s\n\n" 193 "[AdvOut]\nRecFilePath=%s\n" 194 % (str(simpledir).replace("\\", "\\\\"), 195 str(recdir).replace("\\", "\\\\")), 196 encoding="utf-8") 197 198 old = os.environ.get("APPDATA") 199 os.environ["APPDATA"] = d 200 try: 201 found = subtitle.obs_recording_folder() 202 finally: 203 if old is None: 204 os.environ.pop("APPDATA", None) 205 else: 206 os.environ["APPDATA"] = old 207 208 assert found is not None, "BOM or mode parsing failed" 209 assert Path(found) == recdir, (found, recdir) 210 211 212 def test_obs_folder_detection_simple_mode(): 213 import os 214 import tempfile 215 216 with tempfile.TemporaryDirectory() as d: 217 root = Path(d) / "obs-studio" 218 prof = root / "basic" / "profiles" / "P" 219 prof.mkdir(parents=True) 220 simpledir = Path(d) / "simple" 221 simpledir.mkdir() 222 (root / "user.ini").write_text("[Basic]\nProfileDir=P\n", encoding="utf-8") 223 (prof / "basic.ini").write_text( 224 "[Output]\nMode=Simple\n\n[SimpleOutput]\nFilePath=%s\n" 225 % str(simpledir).replace("\\", "\\\\"), encoding="utf-8") 226 227 old = os.environ.get("APPDATA") 228 os.environ["APPDATA"] = d 229 try: 230 found = subtitle.obs_recording_folder() 231 finally: 232 if old is None: 233 os.environ.pop("APPDATA", None) 234 else: 235 os.environ["APPDATA"] = old 236 assert Path(found) == simpledir, found 237 238 239 def test_range_requests_are_served(): 240 """Without 206 support a browser will not seek in a <video> at all.""" 241 import http.client 242 import tempfile 243 244 payload = bytes(range(256)) * 8 # 2048 bytes 245 with tempfile.TemporaryDirectory() as d: 246 (Path(d) / "clip.mp4").write_bytes(payload) 247 srv = subtitle.serve(Path(d), 0) # port 0 = pick a free one 248 port = srv.server_address[1] 249 try: 250 def req(headers): 251 c = http.client.HTTPConnection("127.0.0.1", port, timeout=5) 252 c.request("GET", "/clip.mp4", headers=headers) 253 r = c.getresponse() 254 body = r.read() 255 c.close() 256 return r, body 257 258 r, body = req({}) 259 assert r.status == 200, r.status 260 assert r.getheader("Accept-Ranges") == "bytes" 261 assert body == payload 262 263 r, body = req({"Range": "bytes=10-19"}) 264 assert r.status == 206, r.status 265 assert r.getheader("Content-Range") == "bytes 10-19/2048" 266 assert body == payload[10:20], body 267 268 r, body = req({"Range": "bytes=2040-"}) 269 assert r.status == 206 270 assert body == payload[2040:] 271 272 r, body = req({"Range": "bytes=-8"}) 273 assert r.status == 206 274 assert body == payload[-8:] 275 276 r, _ = req({"Range": "bytes=99999-"}) 277 assert r.status == 416, r.status 278 finally: 279 srv.shutdown() 280 srv.server_close() 281 282 283 def main(): 284 tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")] 285 failed = 0 286 for fn in tests: 287 try: 288 fn() 289 print(" PASS %s" % fn.__name__) 290 except Exception as e: 291 failed += 1 292 print(" FAIL %s: %r" % (fn.__name__, e)) 293 print("\n%d passed, %d failed" % (len(tests) - failed, failed)) 294 return 1 if failed else 0 295 296 297 if __name__ == "__main__": 298 sys.exit(main())