subtitle.py (34626 bytes)
1 #!/usr/bin/env python3 2 """ 3 subtitle.py - subtitle recordings and replay-buffer clips. 4 5 Latency does not matter here, so this uses a much larger model than the live 6 captioner can afford and produces far better text, especially for Finnish, 7 Russian and Japanese. 8 9 python subtitle.py clip.mp4 10 python subtitle.py *.mkv --lang ru 11 python subtitle.py --watch "C:\\Users\\me\\Videos" # auto-subtitle new clips 12 python subtitle.py clip.mp4 --format vtt --translate 13 14 Writes clip.srt next to clip.mp4. OBS, VLC, YouTube and Premiere all read it. 15 No ffmpeg needed - audio is decoded through PyAV, which ships with 16 faster-whisper. 17 """ 18 19 import argparse 20 import os 21 import re 22 import sys 23 import threading 24 import time 25 from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer 26 from pathlib import Path 27 28 VIDEO_EXT = {".mp4", ".mkv", ".mov", ".flv", ".webm", ".avi", ".ts", 29 ".m4a", ".mp3", ".wav", ".opus", ".flac", ".ogg"} 30 31 for _s in (sys.stdout, sys.stderr): 32 try: 33 _s.reconfigure(encoding="utf-8", errors="replace") 34 except (AttributeError, ValueError): 35 pass 36 37 38 def log(*a): 39 print("[" + time.strftime("%H:%M:%S") + "]", *a, flush=True) 40 41 42 # ------------------------------------------------------------------ cues ---- 43 44 def ts(seconds, sep=","): 45 if seconds < 0: 46 seconds = 0.0 47 ms = int(round(seconds * 1000)) 48 h, ms = divmod(ms, 3600000) 49 m, ms = divmod(ms, 60000) 50 s, ms = divmod(ms, 1000) 51 return "%02d:%02d:%02d%s%03d" % (h, m, s, sep, ms) 52 53 54 # Characters that may not begin a line (kinsoku shori). Closing punctuation, 55 # small kana and the long-vowel mark all cling to what precedes them. 56 NO_LINE_START = "、。!?…」』)】〉》・ゃゅょっぁぃぅぇぉーヵヶャュョッァィゥェォ,.!?:;" 57 NO_LINE_END = "「『(【〈《" 58 59 60 def looks_cjk(text): 61 return any("" <= c <= "ヿ" or "一" <= c <= "鿿" 62 or "" <= c <= "" for c in text) 63 64 65 def wrap_cjk(text, width): 66 """Break Japanese/Chinese text, which has no spaces to split on.""" 67 if len(text) <= width: 68 return text 69 mid = len(text) / 2 70 best = None 71 for i in range(1, len(text)): 72 if text[i] in NO_LINE_START or text[i - 1] in NO_LINE_END: 73 continue 74 if max(i, len(text) - i) > width: 75 continue 76 score = abs(i - mid) 77 # Prefer breaking straight after sentence punctuation. 78 if text[i - 1] in "、。!?": 79 score -= width / 2 80 if best is None or score < best[0]: 81 best = (score, i) 82 if best is None: 83 i = max(1, min(len(text) - 1, int(mid))) 84 while i < len(text) - 1 and text[i] in NO_LINE_START: 85 i += 1 86 return text[:i] + "\n" + text[i:] 87 return text[:best[1]] + "\n" + text[best[1]:] 88 89 90 def wrap(text, width): 91 """Split into at most two balanced lines, the way subtitles are normally set.""" 92 text = " ".join(text.split()) 93 if len(text) <= width: 94 return text 95 words = text.split() 96 if len(words) < 2: 97 return wrap_cjk(text, width) if looks_cjk(text) else text 98 if looks_cjk(text) and len(words) < len(text) / 8: 99 # Mostly CJK with a stray space or two: character breaking reads better. 100 return wrap_cjk(text, width) 101 102 fits, over = None, None 103 for i in range(1, len(words)): 104 a, b = " ".join(words[:i]), " ".join(words[i:]) 105 longest, balance = max(len(a), len(b)), abs(len(a) - len(b)) 106 if longest <= width: 107 if fits is None or balance < fits[0]: 108 fits = (balance, a, b) 109 # Fallback for text with no split that fits: overflow as little as 110 # possible rather than emitting one very long line. 111 if over is None or (longest, balance) < (over[0], over[1]): 112 over = (longest, balance, a, b) 113 114 if fits: 115 return fits[1] + "\n" + fits[2] 116 return over[2] + "\n" + over[3] 117 118 119 SENT_END = (".", "!", "?", "…", "。", "!", "?", "؟") 120 121 122 class _Word: 123 """Stand-in when Whisper returns a segment without word timings.""" 124 125 def __init__(self, start, end, word): 126 self.start, self.end, self.word = start, end, word 127 128 129 def collect_words(segments): 130 words = [] 131 for seg in segments: 132 ws = list(getattr(seg, "words", None) or []) 133 if ws: 134 words.extend(ws) 135 elif seg.text.strip(): 136 words.append(_Word(seg.start, seg.end, seg.text.strip() + " ")) 137 return words 138 139 140 def group_sentences(words, max_dur, max_gap): 141 """Split on sentence endings first - this is the boundary that matters. 142 143 Long pauses and a hard duration cap act only as fallbacks, so a cue never 144 starts mid-sentence just because a character budget ran out. 145 """ 146 out, cur = [], [] 147 for w in words: 148 if cur: 149 gap = w.start - cur[-1].end 150 dur = w.end - cur[0].start 151 if gap > max_gap or dur > max_dur * 2.5: 152 out.append(cur) 153 cur = [] 154 cur.append(w) 155 if w.word.strip().endswith(SENT_END): 156 out.append(cur) 157 cur = [] 158 if cur: 159 out.append(cur) 160 return [g for g in out if "".join(x.word for x in g).strip()] 161 162 163 def text_of(group): 164 return " ".join("".join(w.word for w in group).split()) 165 166 167 def build_cues(segments, max_chars, max_dur, max_gap): 168 """Sentences first, then subdivide any that are too long to display.""" 169 cues = [] 170 for group in group_sentences(collect_words(segments), max_dur, max_gap): 171 chunks, chunk = [], [] 172 for w in group: 173 if chunk: 174 chars = sum(len(x.word) for x in chunk) + len(w.word) 175 dur = w.end - chunk[0].start 176 if chars > max_chars or dur > max_dur: 177 chunks.append(chunk) 178 chunk = [] 179 chunk.append(w) 180 if chunk: 181 chunks.append(chunk) 182 183 # Greedy filling can strand a word or two on the last line; fold a 184 # runt back into its predecessor rather than flashing it alone. 185 if len(chunks) > 1 and len("".join(w.word for w in chunks[-1]).strip()) < 16: 186 chunks[-2].extend(chunks.pop()) 187 188 for c in chunks: 189 cues.append((c[0].start, c[-1].end, text_of(c))) 190 return cues 191 192 193 def write_srt(cues, path, width): 194 with open(path, "w", encoding="utf-8") as fh: 195 for i, (start, end, text) in enumerate(cues, 1): 196 fh.write("%d\n%s --> %s\n%s\n\n" 197 % (i, ts(start), ts(end), wrap(text, width))) 198 199 200 def write_vtt(cues, path, width): 201 with open(path, "w", encoding="utf-8") as fh: 202 fh.write("WEBVTT\n\n") 203 for start, end, text in cues: 204 fh.write("%s --> %s\n%s\n\n" 205 % (ts(start, "."), ts(end, "."), wrap(text, width))) 206 207 208 def write_txt(cues, path, width): 209 with open(path, "w", encoding="utf-8") as fh: 210 for _, _, text in cues: 211 fh.write(text + "\n") 212 213 214 WRITERS = {"srt": write_srt, "vtt": write_vtt, "txt": write_txt} 215 216 217 # ------------------------------------------------------------------ anki ---- 218 219 def anki_media_dir(): 220 """Anki's collection.media for the default profile, if we can find it.""" 221 base = Path(os.environ.get("APPDATA", "")) / "Anki2" 222 if not base.is_dir(): 223 return None 224 for profile in sorted(base.iterdir()): 225 media = profile / "collection.media" 226 if media.is_dir(): 227 return media 228 return None 229 230 231 def slice_audio(audio, sr, start, end, pad): 232 lo = max(0, int((start - pad) * sr)) 233 hi = min(len(audio), int((end + pad) * sr)) 234 return audio[lo:hi] 235 236 237 def write_audio_clip(samples, sr, path): 238 """mp3 when PyAV has an encoder for it, otherwise wav.""" 239 import numpy as np 240 241 pcm = np.clip(samples, -1.0, 1.0) 242 if path.suffix == ".mp3": 243 try: 244 import av 245 246 with av.open(str(path), "w") as container: 247 stream = container.add_stream("mp3", rate=sr) 248 stream.layout = "mono" 249 frame = av.AudioFrame.from_ndarray( 250 (pcm * 32767).astype(np.int16).reshape(1, -1), 251 format="s16", layout="mono") 252 frame.rate = sr 253 for packet in stream.encode(frame): 254 container.mux(packet) 255 for packet in stream.encode(None): 256 container.mux(packet) 257 return path 258 except Exception: 259 path = path.with_suffix(".wav") 260 261 import wave 262 263 with wave.open(str(path), "wb") as w: 264 w.setnchannels(1) 265 w.setsampwidth(2) 266 w.setframerate(sr) 267 w.writeframes((pcm * 32767).astype(np.int16).tobytes()) 268 return path 269 270 271 def grab_frames(video, times, out_paths, width): 272 """Best-effort screenshots. Returns the paths actually written.""" 273 try: 274 import av 275 except ImportError: 276 return {} 277 written = {} 278 try: 279 with av.open(str(video)) as container: 280 if not container.streams.video: 281 return {} 282 vs = container.streams.video[0] 283 vs.thread_type = "AUTO" 284 for i, (t, dest) in enumerate(zip(times, out_paths)): 285 try: 286 container.seek(int(t / vs.time_base), stream=vs) 287 for frame in container.decode(vs): 288 if frame.time is None or frame.time + 0.001 < t: 289 continue 290 img = frame.to_image() 291 if width and img.width > width: 292 img = img.resize((width, max(1, round(img.height * width / img.width)))) 293 img.save(str(dest), quality=82) 294 written[i] = dest 295 break 296 except Exception: 297 continue 298 except Exception: 299 return written 300 return written 301 302 303 def esc(text): 304 return text.replace("\t", " ").replace("\n", " ").strip() 305 306 307 # ----------------------------------------------------------- mining page ---- 308 309 MINE_PAGE = """<!doctype html> 310 <html lang="__LANG__"> 311 <head> 312 <meta charset="utf-8"> 313 <title>__TITLE__</title> 314 <style> 315 :root { --bg:#0e1116; --panel:#161a21; --line:#252b36; --fg:#edf1f7; 316 --dim:#8b95a7; --accent:#7dd3fc; --size:26px; } 317 * { box-sizing:border-box; } 318 html,body { height:100%; margin:0; } 319 body { background:var(--bg); color:var(--fg); display:flex; flex-direction:column; 320 font-family:"Inter","Segoe UI","Yu Gothic UI","Meiryo","Noto Sans CJK JP", 321 "Noto Sans",system-ui,sans-serif; } 322 header { display:flex; gap:10px; align-items:center; flex-wrap:wrap; padding:10px 16px; 323 background:var(--panel); border-bottom:1px solid var(--line); font-size:13px; } 324 header .grow { flex:1; } 325 button { font:inherit; font-size:13px; color:var(--fg); cursor:pointer; background:#1f2531; 326 border:1px solid var(--line); border-radius:6px; padding:6px 11px; } 327 button:hover { background:#29313f; } 328 button.on { background:var(--accent); border-color:var(--accent); color:#06202c; } 329 main { flex:1; display:flex; min-height:0; flex-wrap:wrap; } 330 #left { flex:1 1 460px; min-width:320px; display:flex; flex-direction:column; 331 border-right:1px solid var(--line); } 332 video { width:100%; background:#000; max-height:52vh; } 333 #cur { padding:16px 20px; font-size:var(--size); line-height:1.7; min-height:3em; 334 user-select:text; border-top:1px solid var(--line); } 335 #list { flex:1 1 380px; min-width:300px; overflow-y:auto; padding:14px 18px 30vh; } 336 .cue { padding:8px 11px; border-radius:6px; border-left:3px solid transparent; 337 font-size:calc(var(--size)*.82); line-height:1.7; user-select:text; margin-bottom:8px; } 338 .cue::before { content:attr(data-time); display:block; font-size:11px; color:var(--dim); 339 user-select:none; margin-bottom:2px; font-variant-numeric:tabular-nums; } 340 .cue:hover { background:#151a22; } 341 .cue.active { background:#1a222e; border-left-color:var(--accent); } 342 .jump { float:right; font-size:11px; color:var(--dim); cursor:pointer; user-select:none; } 343 </style> 344 </head> 345 <body> 346 <header> 347 <b>__TITLE__</b> 348 <span class="grow"></span> 349 <button id="smaller">A-</button> 350 <button id="bigger">A+</button> 351 <button id="follow" class="on">Follow</button> 352 <button id="loop">Loop cue</button> 353 <button id="copy">Copy all</button> 354 </header> 355 <main> 356 <div id="left"> 357 <video id="v" src="__VIDEO__" controls preload="metadata"></video> 358 <div id="cur"></div> 359 </div> 360 <div id="list"></div> 361 </main> 362 <script> 363 const CUES = __CUES__; 364 const v = document.getElementById("v"); 365 const list = document.getElementById("list"); 366 const cur = document.getElementById("cur"); 367 let follow = true, looping = false, active = -1; 368 369 function fmt(t) { 370 const m = Math.floor(t / 60), s = Math.floor(t % 60); 371 return (m < 10 ? "0" : "") + m + ":" + (s < 10 ? "0" : "") + s; 372 } 373 374 CUES.forEach((c, i) => { 375 const d = document.createElement("div"); 376 d.className = "cue"; 377 d.dataset.time = fmt(c.start); 378 const j = document.createElement("span"); 379 j.className = "jump"; 380 j.textContent = "play"; 381 j.onclick = (e) => { e.stopPropagation(); v.currentTime = c.start; v.play(); }; 382 d.appendChild(j); 383 d.appendChild(document.createTextNode(c.text)); 384 d.onclick = () => { v.currentTime = c.start; }; 385 list.appendChild(d); 386 }); 387 388 function setActive(i) { 389 if (i === active) return; 390 const prev = list.children[active]; 391 if (prev) prev.classList.remove("active"); 392 active = i; 393 const el = list.children[i]; 394 cur.textContent = ""; 395 if (!el) return; 396 el.classList.add("active"); 397 cur.appendChild(document.createTextNode(CUES[i].text)); 398 if (follow) el.scrollIntoView({ block: "center", behavior: "smooth" }); 399 } 400 401 function syncNow() { 402 const t = v.currentTime; 403 if (looping && active >= 0 && !v.paused) { 404 const c = CUES[active]; 405 if (t > c.end + 0.05) { v.currentTime = c.start; return; } 406 } 407 let i = -1; 408 for (let k = 0; k < CUES.length; k++) { 409 if (t >= CUES[k].start - 0.05 && t <= CUES[k].end + 0.35) { i = k; break; } 410 } 411 // While scrubbing between cues, keep showing the one just passed. 412 if (i < 0) { 413 for (let k = CUES.length - 1; k >= 0; k--) { 414 if (t >= CUES[k].start) { i = k; break; } 415 } 416 } 417 if (i >= 0) setActive(i); 418 } 419 420 // timeupdate only fires during playback; seeked covers scrubbing while paused, 421 // which is most of what mining actually involves. 422 ["timeupdate", "seeked", "loadedmetadata", "play"].forEach( 423 (ev) => v.addEventListener(ev, syncNow)); 424 425 document.getElementById("bigger").onclick = () => bump(3); 426 document.getElementById("smaller").onclick = () => bump(-3); 427 function bump(d) { 428 const s = parseInt(getComputedStyle(document.documentElement) 429 .getPropertyValue("--size"), 10) + d; 430 document.documentElement.style.setProperty("--size", 431 Math.max(14, Math.min(60, s)) + "px"); 432 } 433 document.getElementById("follow").onclick = (e) => { 434 follow = !follow; e.target.className = follow ? "on" : ""; 435 }; 436 document.getElementById("loop").onclick = (e) => { 437 looping = !looping; e.target.className = looping ? "on" : ""; 438 }; 439 document.getElementById("copy").onclick = async () => { 440 try { 441 await navigator.clipboard.writeText(CUES.map((c) => c.text).join("\\n")); 442 const b = document.getElementById("copy"); 443 b.textContent = "Copied"; setTimeout(() => b.textContent = "Copy all", 1200); 444 } catch (e) {} 445 }; 446 document.addEventListener("keydown", (e) => { 447 if (e.key === " ") { e.preventDefault(); v.paused ? v.play() : v.pause(); } 448 if (e.key === "ArrowLeft" && active > 0) v.currentTime = CUES[active - 1].start; 449 if (e.key === "ArrowRight" && active < CUES.length - 1) v.currentTime = CUES[active + 1].start; 450 }); 451 </script> 452 </body> 453 </html> 454 """ 455 456 457 def write_mine_page(path, cues, lang): 458 """A self-contained page: the clip plus selectable, synced subtitles. 459 460 Yomitan mines from real DOM text, so the value is in the subtitles being 461 hoverable HTML rather than pixels burned into the video. 462 """ 463 import json 464 465 data = [{"start": round(s, 3), "end": round(e, 3), "text": t} for s, e, t in cues] 466 html = (MINE_PAGE 467 .replace("__CUES__", json.dumps(data, ensure_ascii=False)) 468 .replace("__VIDEO__", path.name.replace('"', "%22")) 469 .replace("__TITLE__", path.stem.replace("<", "<")) 470 .replace("__LANG__", "ja" if lang == "ja" else (lang or "en"))) 471 out = path.with_suffix(".html") 472 out.write_text(html, encoding="utf-8") 473 return out 474 475 476 def export_anki(path, cues, translations, args, log=log): 477 """One row per sentence: text, translation, audio, screenshot.""" 478 from faster_whisper.audio import decode_audio 479 480 media = Path(args.anki_media) if args.anki_media else path.parent / (path.stem + "_media") 481 media.mkdir(parents=True, exist_ok=True) 482 483 sr = 24000 484 try: 485 audio = decode_audio(str(path), sampling_rate=sr) 486 except Exception as e: 487 log("cannot re-decode audio for cards: %s" % e) 488 return None 489 490 stem = "".join(c if (c.isalnum() or c in "-_") else "_" for c in path.stem)[:48] 491 audio_names, image_names = [], [] 492 493 for i, (start, end, _) in enumerate(cues, 1): 494 clip = slice_audio(audio, sr, start, end, args.anki_pad) 495 dest = write_audio_clip(clip, sr, media / ("%s_%04d.mp3" % (stem, i))) 496 audio_names.append(dest.name) 497 498 if args.anki_images: 499 mids = [(s + e) / 2 for s, e, _ in cues] 500 dests = [media / ("%s_%04d.jpg" % (stem, i)) for i in range(1, len(cues) + 1)] 501 got = grab_frames(path, mids, dests, args.anki_image_width) 502 image_names = [dests[i].name if i in got else "" for i in range(len(cues))] 503 if not got: 504 log("no screenshots (PyAV/Pillow could not decode video frames)") 505 else: 506 image_names = [""] * len(cues) 507 508 tsv = path.with_suffix(".anki.tsv") 509 with open(tsv, "w", encoding="utf-8", newline="") as fh: 510 for i, (start, end, text) in enumerate(cues): 511 row = [ 512 esc(text), 513 esc(translations[i]) if translations else "", 514 "[sound:%s]" % audio_names[i], 515 '<img src="%s">' % image_names[i] if image_names[i] else "", 516 esc(path.name), 517 ts(start), 518 ] 519 fh.write("\t".join(row) + "\n") 520 521 log("%d cards -> %s" % (len(cues), tsv.name)) 522 log(" media -> %s" % media) 523 if not args.anki_media: 524 target = anki_media_dir() 525 if target: 526 log(" copy the media files into: %s" % target) 527 else: 528 log(" copy the media files into your Anki collection.media folder") 529 return tsv 530 531 532 # ------------------------------------------------------------ processing ---- 533 534 class Subtitler: 535 def __init__(self, args): 536 from faster_whisper import WhisperModel 537 538 self.args = args 539 log("loading %r on %s/%s ..." % (args.model, args.device, args.compute)) 540 t0 = time.time() 541 self.model = WhisperModel(args.model, device=args.device, 542 compute_type=args.compute, 543 cpu_threads=args.threads, num_workers=1) 544 log("model ready in %.1fs" % (time.time() - t0)) 545 self._tmodel = None 546 self._tname = None 547 548 def translator(self): 549 """A model that can actually translate. 550 551 large-v3-turbo is a transcription-only fine-tune: asked to translate it 552 returns the source language instead, silently. Fall back to a model 553 that supports the task rather than emitting untranslated text. 554 """ 555 from faster_whisper import WhisperModel 556 557 a = self.args 558 name = a.translate_model or a.model 559 if not a.translate_model and "turbo" in a.model.lower(): 560 name = "small" 561 log("note: %s cannot translate (transcription-only fine-tune); " 562 "using %r instead - override with --translate-model" % (a.model, name)) 563 if name == a.model: 564 return self.model 565 if self._tmodel is None or self._tname != name: 566 log("loading translation model %r ..." % name) 567 self._tmodel = WhisperModel(name, device=a.device, compute_type=a.compute, 568 cpu_threads=a.threads, num_workers=1) 569 self._tname = name 570 return self._tmodel 571 572 def run(self, path): 573 from faster_whisper.audio import decode_audio 574 575 a = self.args 576 out = path.with_suffix("." + a.format) 577 if out.exists() and not a.overwrite: 578 log("skip (exists): %s" % out.name) 579 return out 580 581 try: 582 audio = decode_audio(str(path), sampling_rate=16000) 583 except Exception as e: 584 log("cannot decode %s: %s" % (path.name, e)) 585 return None 586 dur = len(audio) / 16000 587 if dur < 0.2: 588 log("skip (no audio): %s" % path.name) 589 return None 590 591 t0 = time.time() 592 engine = self.translator() if a.translate else self.model 593 segments, info = engine.transcribe( 594 audio, 595 language=None if a.lang == "auto" else a.lang, 596 task="translate" if a.translate else "transcribe", 597 beam_size=a.beam, 598 temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0], 599 condition_on_previous_text=True, 600 vad_filter=True, 601 word_timestamps=True, 602 no_speech_threshold=0.6, 603 log_prob_threshold=-1.0, 604 ) 605 segments = list(segments) 606 took = time.time() - t0 607 608 if a.anki: 609 # Cards want whole sentences; subtitles want display-sized chunks. 610 groups = group_sentences(collect_words(segments), a.max_dur, a.max_gap) 611 card_cues = [(g[0].start, g[-1].end, text_of(g)) for g in groups] 612 else: 613 card_cues = [] 614 615 cues = build_cues(segments, a.max_chars, a.max_dur, a.max_gap) 616 if not cues: 617 log("no speech found in %s" % path.name) 618 return None 619 620 WRITERS[a.format](cues, out, a.width) 621 log("%s -> %s (%d cues, %s, %.0fs audio in %.0fs)" 622 % (path.name, out.name, len(cues), 623 getattr(info, "language", a.lang), dur, took)) 624 625 if a.mine: 626 page = write_mine_page(path, cues, getattr(info, "language", a.lang)) 627 log("mining page -> %s" % page.name) 628 629 if a.anki: 630 translations = self.translate_cues(audio, card_cues) if a.anki_translate else None 631 export_anki(path, card_cues, translations, a) 632 return out 633 634 def translate_cues(self, audio, cues): 635 """English for the back of each card. 636 637 Translating one short sentence at a time starves Whisper of context and 638 it often just echoes the source language back. Translating the whole 639 clip once and aligning by timestamp is both better and faster. 640 """ 641 log("translating for card backs...") 642 try: 643 segs, _ = self.translator().transcribe( 644 audio, language=None if self.args.lang == "auto" else self.args.lang, 645 task="translate", beam_size=self.args.beam, 646 temperature=[0.0, 0.2, 0.4], 647 condition_on_previous_text=True, vad_filter=True) 648 segs = list(segs) 649 except Exception as e: 650 log("translation failed: %s" % e) 651 return [""] * len(cues) 652 653 out = [] 654 for start, end, _ in cues: 655 parts = [s.text.strip() for s in segs 656 if s.start < end - 0.15 and s.end > start + 0.15] 657 out.append(re.sub(r"\s+", " ", " ".join(parts)).strip()) 658 return out 659 660 661 def obs_recording_folder(): 662 """Where OBS writes recordings and replay-buffer clips, from its own config. 663 664 Which key holds it depends on the output mode, so read Mode first rather 665 than guessing. 666 """ 667 import configparser 668 669 root = Path(os.environ.get("APPDATA", "")) / "obs-studio" 670 if not root.is_dir(): 671 return None 672 673 profile_dir = None 674 for name in ("user.ini", "global.ini"): 675 cfg = configparser.RawConfigParser(strict=False) 676 try: 677 cfg.read(root / name, encoding="utf-8-sig") 678 profile_dir = cfg.get("Basic", "ProfileDir", fallback=None) or profile_dir 679 except (configparser.Error, OSError): 680 pass 681 682 profiles = root / "basic" / "profiles" 683 candidates = [] 684 if profile_dir and (profiles / profile_dir).is_dir(): 685 candidates.append(profiles / profile_dir) 686 if profiles.is_dir(): 687 candidates.extend(p for p in profiles.iterdir() if p.is_dir()) 688 689 for prof in candidates: 690 cfg = configparser.RawConfigParser(strict=False) 691 try: 692 cfg.read(prof / "basic.ini", encoding="utf-8-sig") 693 except (configparser.Error, OSError): 694 continue 695 mode = (cfg.get("Output", "Mode", fallback="") or "").lower() 696 if mode.startswith("adv"): 697 keys = [("AdvOut", "RecFilePath"), ("AdvOut", "FFFilePath"), 698 ("SimpleOutput", "FilePath")] 699 else: 700 keys = [("SimpleOutput", "FilePath"), ("AdvOut", "RecFilePath")] 701 for section, key in keys: 702 raw = cfg.get(section, key, fallback=None) 703 if not raw: 704 continue 705 path = Path(raw.replace("\\\\", "\\").strip()) 706 if path.is_dir(): 707 return path 708 return None 709 710 711 class _Limited: 712 """Feeds copyfile only the bytes belonging to the requested range.""" 713 714 def __init__(self, fh, remaining): 715 self.fh, self.remaining = fh, remaining 716 717 def read(self, size=-1): 718 if self.remaining <= 0: 719 return b"" 720 if size is None or size < 0: 721 size = self.remaining 722 data = self.fh.read(min(size, self.remaining)) 723 self.remaining -= len(data) 724 return data 725 726 def close(self): 727 self.fh.close() 728 729 730 class RangeHandler(SimpleHTTPRequestHandler): 731 """Static files with HTTP Range support. 732 733 Browsers refuse to seek in a <video> unless the server answers range 734 requests, and Python's stock handler does not - so scrubbing a clip, which 735 is most of what mining involves, would silently not work. 736 """ 737 738 def log_message(self, *a): 739 pass 740 741 def end_headers(self): 742 self.send_header("Accept-Ranges", "bytes") 743 SimpleHTTPRequestHandler.end_headers(self) 744 745 def send_head(self): 746 rng = self.headers.get("Range") 747 if not rng: 748 return SimpleHTTPRequestHandler.send_head(self) 749 750 path = self.translate_path(self.path) 751 if os.path.isdir(path): 752 return SimpleHTTPRequestHandler.send_head(self) 753 try: 754 fh = open(path, "rb") 755 except OSError: 756 self.send_error(404, "File not found") 757 return None 758 759 size = os.fstat(fh.fileno()).st_size 760 m = re.match(r"bytes=(\d*)-(\d*)\s*$", rng.strip()) 761 if not m or (not m.group(1) and not m.group(2)): 762 fh.close() 763 self.send_error(400, "Malformed Range") 764 return None 765 766 if not m.group(1): # bytes=-N (last N bytes) 767 start, end = max(0, size - int(m.group(2))), size - 1 768 else: 769 start = int(m.group(1)) 770 end = int(m.group(2)) if m.group(2) else size - 1 771 end = min(end, size - 1) 772 773 if start >= size or start > end: 774 fh.close() 775 self.send_response(416) 776 self.send_header("Content-Range", "bytes */%d" % size) 777 self.send_header("Content-Length", "0") 778 self.end_headers() 779 return None 780 781 self.send_response(206) 782 self.send_header("Content-Type", self.guess_type(path)) 783 self.send_header("Content-Range", "bytes %d-%d/%d" % (start, end, size)) 784 self.send_header("Content-Length", str(end - start + 1)) 785 self.end_headers() 786 fh.seek(start) 787 return _Limited(fh, end - start + 1) 788 789 790 def serve(folder, port): 791 from functools import partial 792 793 handler = partial(RangeHandler, directory=str(folder)) 794 srv = ThreadingHTTPServer(("127.0.0.1", port), handler) 795 threading.Thread(target=srv.serve_forever, daemon=True).start() 796 log("serving %s at http://127.0.0.1:%d/ (video seeking enabled)" % (folder, port)) 797 pages = sorted(p.name for p in Path(folder).glob("*.html")) 798 for name in pages[-10:]: 799 log(" http://127.0.0.1:%d/%s" % (port, name)) 800 return srv 801 802 803 def stable(path, checks=3, delay=1.0): 804 """Wait until a file stops growing, so we do not read a half-written clip.""" 805 last = -1 806 for _ in range(120): 807 try: 808 size = path.stat().st_size 809 except OSError: 810 return False 811 if size == last: 812 checks -= 1 813 if checks <= 0: 814 return size > 0 815 else: 816 checks = 3 817 last = size 818 time.sleep(delay) 819 return False 820 821 822 def watch(sub, folder, args): 823 folder = Path(folder) 824 if not folder.is_dir(): 825 raise SystemExit("not a folder: %s" % folder) 826 seen = {p.resolve() for p in folder.iterdir() 827 if p.suffix.lower() in VIDEO_EXT} 828 log("watching %s for new clips (%d already present) - Ctrl+C to stop" 829 % (folder, len(seen))) 830 while True: 831 try: 832 for p in sorted(folder.iterdir()): 833 if p.suffix.lower() not in VIDEO_EXT: 834 continue 835 r = p.resolve() 836 if r in seen: 837 continue 838 seen.add(r) 839 log("new clip: %s" % p.name) 840 if stable(p): 841 sub.run(p) 842 else: 843 log("gave up waiting for %s to finish writing" % p.name) 844 time.sleep(args.poll) 845 except KeyboardInterrupt: 846 print() 847 log("stopped") 848 return 849 850 851 def main(): 852 p = argparse.ArgumentParser( 853 description="Subtitle recordings and replay clips", 854 formatter_class=argparse.ArgumentDefaultsHelpFormatter) 855 p.add_argument("files", nargs="*", help="video/audio files to subtitle") 856 p.add_argument("--watch", nargs="?", const="auto", default=None, metavar="FOLDER", 857 help="watch a folder and subtitle clips as they appear; " 858 "with no folder, read OBS's own recording path") 859 p.add_argument("--poll", type=float, default=3.0, help="watch interval (s)") 860 861 p.add_argument("--model", default="large-v3-turbo", 862 help="quality matters more than speed here") 863 p.add_argument("--lang", default="fi", help="language code, or 'auto'") 864 p.add_argument("--device", default="cpu", choices=["cpu", "cuda"]) 865 p.add_argument("--compute", default="int8") 866 p.add_argument("--threads", type=int, default=max(2, (os.cpu_count() or 8) - 2)) 867 p.add_argument("--beam", type=int, default=5) 868 p.add_argument("--translate", action="store_true", 869 help="write English subtitles instead of the original language") 870 p.add_argument("--translate-model", default=None, 871 help="model used for translation; defaults to --model, except " 872 "for turbo builds which cannot translate at all") 873 874 p.add_argument("--format", default="srt", choices=sorted(WRITERS)) 875 p.add_argument("--width", type=int, default=42, help="max characters per line") 876 p.add_argument("--max-chars", type=int, default=84, help="max characters per cue") 877 p.add_argument("--max-dur", type=float, default=6.0, help="max seconds per cue") 878 p.add_argument("--max-gap", type=float, default=0.8, 879 help="silence (s) that forces a new cue") 880 p.add_argument("--overwrite", action="store_true", 881 help="re-subtitle files that already have output") 882 883 g = p.add_argument_group("mining") 884 g.add_argument("--mine", dest="mine", action="store_true", default=True, 885 help="write a browser page with the clip and hoverable subtitles") 886 g.add_argument("--no-mine", dest="mine", action="store_false") 887 g.add_argument("--serve", nargs="?", type=int, const=8778, default=None, 888 metavar="PORT", 889 help="serve the mining pages over http with video seeking, " 890 "so Yomitan works without file-URL permissions") 891 892 g = p.add_argument_group("anki cards") 893 g.add_argument("--anki", action="store_true", 894 help="also export one card per sentence, with clipped audio") 895 g.add_argument("--anki-translate", action="store_true", 896 help="add an English translation to each card") 897 g.add_argument("--anki-images", dest="anki_images", action="store_true", default=True, 898 help="grab a screenshot per card") 899 g.add_argument("--no-anki-images", dest="anki_images", action="store_false") 900 g.add_argument("--anki-image-width", type=int, default=640) 901 g.add_argument("--anki-pad", type=float, default=0.25, 902 help="seconds of padding around each audio clip") 903 g.add_argument("--anki-media", default=None, 904 help="write media straight into Anki's collection.media") 905 906 args = p.parse_args() 907 if not args.files and not args.watch: 908 p.error("give some files, or --watch a folder") 909 910 if args.watch == "auto": 911 found = obs_recording_folder() 912 if not found: 913 p.error("could not find OBS's recording folder; pass --watch FOLDER") 914 args.watch = str(found) 915 log("OBS records to: %s" % args.watch) 916 917 sub = Subtitler(args) 918 919 done = [] 920 for pattern in args.files: 921 path = Path(pattern) 922 matches = [path] if path.exists() else sorted(Path().glob(pattern)) 923 if not matches: 924 log("no such file: %s" % pattern) 925 for m in matches: 926 sub.run(m) 927 done.append(m) 928 929 srv = None 930 if args.serve: 931 folder = Path(args.watch) if args.watch else ( 932 done[0].parent if done else Path(".")) 933 srv = serve(folder.resolve(), args.serve) 934 935 if args.watch: 936 watch(sub, args.watch, args) 937 elif srv is not None: 938 log("Ctrl+C to stop serving") 939 try: 940 while True: 941 time.sleep(3600) 942 except KeyboardInterrupt: 943 print() 944 return 0 945 946 947 if __name__ == "__main__": 948 sys.exit(main())