Recently Written · git

desktop-subtitle-replay

git clone https://github.com/equwal/desktop-subtitle-replay

Log | Files | Refs


subtitle.py (34626 bytes)

1 #!/usr/bin/env python3
2 """
3 subtitle.py - subtitle recordings and replay-buffer clips.
4 
5 Latency does not matter here, so this uses a much larger model than the live
6 captioner can afford and produces far better text, especially for Finnish,
7 Russian and Japanese.
8 
9     python subtitle.py clip.mp4
10     python subtitle.py *.mkv --lang ru
11     python subtitle.py --watch "C:\\Users\\me\\Videos"      # auto-subtitle new clips
12     python subtitle.py clip.mp4 --format vtt --translate
13 
14 Writes clip.srt next to clip.mp4. OBS, VLC, YouTube and Premiere all read it.
15 No ffmpeg needed - audio is decoded through PyAV, which ships with
16 faster-whisper.
17 """
18 
19 import argparse
20 import os
21 import re
22 import sys
23 import threading
24 import time
25 from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
26 from pathlib import Path
27 
28 VIDEO_EXT = {".mp4", ".mkv", ".mov", ".flv", ".webm", ".avi", ".ts",
29              ".m4a", ".mp3", ".wav", ".opus", ".flac", ".ogg"}
30 
31 for _s in (sys.stdout, sys.stderr):
32     try:
33         _s.reconfigure(encoding="utf-8", errors="replace")
34     except (AttributeError, ValueError):
35         pass
36 
37 
38 def log(*a):
39     print("[" + time.strftime("%H:%M:%S") + "]", *a, flush=True)
40 
41 
42 # ------------------------------------------------------------------ cues ----
43 
44 def ts(seconds, sep=","):
45     if seconds < 0:
46         seconds = 0.0
47     ms = int(round(seconds * 1000))
48     h, ms = divmod(ms, 3600000)
49     m, ms = divmod(ms, 60000)
50     s, ms = divmod(ms, 1000)
51     return "%02d:%02d:%02d%s%03d" % (h, m, s, sep, ms)
52 
53 
54 # Characters that may not begin a line (kinsoku shori). Closing punctuation,
55 # small kana and the long-vowel mark all cling to what precedes them.
56 NO_LINE_START = "、。!?…」』)】〉》・ゃゅょっぁぃぅぇぉーヵヶャュョッァィゥェォ,.!?:;"
57 NO_LINE_END = "「『(【〈《"
58 
59 
60 def looks_cjk(text):
61     return any("぀" <= c <= "ヿ" or "一" <= c <= "鿿"
62                or "＀" <= c <= "￯" for c in text)
63 
64 
65 def wrap_cjk(text, width):
66     """Break Japanese/Chinese text, which has no spaces to split on."""
67     if len(text) <= width:
68         return text
69     mid = len(text) / 2
70     best = None
71     for i in range(1, len(text)):
72         if text[i] in NO_LINE_START or text[i - 1] in NO_LINE_END:
73             continue
74         if max(i, len(text) - i) > width:
75             continue
76         score = abs(i - mid)
77         # Prefer breaking straight after sentence punctuation.
78         if text[i - 1] in "、。!?":
79             score -= width / 2
80         if best is None or score < best[0]:
81             best = (score, i)
82     if best is None:
83         i = max(1, min(len(text) - 1, int(mid)))
84         while i < len(text) - 1 and text[i] in NO_LINE_START:
85             i += 1
86         return text[:i] + "\n" + text[i:]
87     return text[:best[1]] + "\n" + text[best[1]:]
88 
89 
90 def wrap(text, width):
91     """Split into at most two balanced lines, the way subtitles are normally set."""
92     text = " ".join(text.split())
93     if len(text) <= width:
94         return text
95     words = text.split()
96     if len(words) < 2:
97         return wrap_cjk(text, width) if looks_cjk(text) else text
98     if looks_cjk(text) and len(words) < len(text) / 8:
99         # Mostly CJK with a stray space or two: character breaking reads better.
100         return wrap_cjk(text, width)
101 
102     fits, over = None, None
103     for i in range(1, len(words)):
104         a, b = " ".join(words[:i]), " ".join(words[i:])
105         longest, balance = max(len(a), len(b)), abs(len(a) - len(b))
106         if longest <= width:
107             if fits is None or balance < fits[0]:
108                 fits = (balance, a, b)
109         # Fallback for text with no split that fits: overflow as little as
110         # possible rather than emitting one very long line.
111         if over is None or (longest, balance) < (over[0], over[1]):
112             over = (longest, balance, a, b)
113 
114     if fits:
115         return fits[1] + "\n" + fits[2]
116     return over[2] + "\n" + over[3]
117 
118 
119 SENT_END = (".", "!", "?", "…", "。", "!", "?", "؟")
120 
121 
122 class _Word:
123     """Stand-in when Whisper returns a segment without word timings."""
124 
125     def __init__(self, start, end, word):
126         self.start, self.end, self.word = start, end, word
127 
128 
129 def collect_words(segments):
130     words = []
131     for seg in segments:
132         ws = list(getattr(seg, "words", None) or [])
133         if ws:
134             words.extend(ws)
135         elif seg.text.strip():
136             words.append(_Word(seg.start, seg.end, seg.text.strip() + " "))
137     return words
138 
139 
140 def group_sentences(words, max_dur, max_gap):
141     """Split on sentence endings first - this is the boundary that matters.
142 
143     Long pauses and a hard duration cap act only as fallbacks, so a cue never
144     starts mid-sentence just because a character budget ran out.
145     """
146     out, cur = [], []
147     for w in words:
148         if cur:
149             gap = w.start - cur[-1].end
150             dur = w.end - cur[0].start
151             if gap > max_gap or dur > max_dur * 2.5:
152                 out.append(cur)
153                 cur = []
154         cur.append(w)
155         if w.word.strip().endswith(SENT_END):
156             out.append(cur)
157             cur = []
158     if cur:
159         out.append(cur)
160     return [g for g in out if "".join(x.word for x in g).strip()]
161 
162 
163 def text_of(group):
164     return " ".join("".join(w.word for w in group).split())
165 
166 
167 def build_cues(segments, max_chars, max_dur, max_gap):
168     """Sentences first, then subdivide any that are too long to display."""
169     cues = []
170     for group in group_sentences(collect_words(segments), max_dur, max_gap):
171         chunks, chunk = [], []
172         for w in group:
173             if chunk:
174                 chars = sum(len(x.word) for x in chunk) + len(w.word)
175                 dur = w.end - chunk[0].start
176                 if chars > max_chars or dur > max_dur:
177                     chunks.append(chunk)
178                     chunk = []
179             chunk.append(w)
180         if chunk:
181             chunks.append(chunk)
182 
183         # Greedy filling can strand a word or two on the last line; fold a
184         # runt back into its predecessor rather than flashing it alone.
185         if len(chunks) > 1 and len("".join(w.word for w in chunks[-1]).strip()) < 16:
186             chunks[-2].extend(chunks.pop())
187 
188         for c in chunks:
189             cues.append((c[0].start, c[-1].end, text_of(c)))
190     return cues
191 
192 
193 def write_srt(cues, path, width):
194     with open(path, "w", encoding="utf-8") as fh:
195         for i, (start, end, text) in enumerate(cues, 1):
196             fh.write("%d\n%s --> %s\n%s\n\n"
197                      % (i, ts(start), ts(end), wrap(text, width)))
198 
199 
200 def write_vtt(cues, path, width):
201     with open(path, "w", encoding="utf-8") as fh:
202         fh.write("WEBVTT\n\n")
203         for start, end, text in cues:
204             fh.write("%s --> %s\n%s\n\n"
205                      % (ts(start, "."), ts(end, "."), wrap(text, width)))
206 
207 
208 def write_txt(cues, path, width):
209     with open(path, "w", encoding="utf-8") as fh:
210         for _, _, text in cues:
211             fh.write(text + "\n")
212 
213 
214 WRITERS = {"srt": write_srt, "vtt": write_vtt, "txt": write_txt}
215 
216 
217 # ------------------------------------------------------------------ anki ----
218 
219 def anki_media_dir():
220     """Anki's collection.media for the default profile, if we can find it."""
221     base = Path(os.environ.get("APPDATA", "")) / "Anki2"
222     if not base.is_dir():
223         return None
224     for profile in sorted(base.iterdir()):
225         media = profile / "collection.media"
226         if media.is_dir():
227             return media
228     return None
229 
230 
231 def slice_audio(audio, sr, start, end, pad):
232     lo = max(0, int((start - pad) * sr))
233     hi = min(len(audio), int((end + pad) * sr))
234     return audio[lo:hi]
235 
236 
237 def write_audio_clip(samples, sr, path):
238     """mp3 when PyAV has an encoder for it, otherwise wav."""
239     import numpy as np
240 
241     pcm = np.clip(samples, -1.0, 1.0)
242     if path.suffix == ".mp3":
243         try:
244             import av
245 
246             with av.open(str(path), "w") as container:
247                 stream = container.add_stream("mp3", rate=sr)
248                 stream.layout = "mono"
249                 frame = av.AudioFrame.from_ndarray(
250                     (pcm * 32767).astype(np.int16).reshape(1, -1),
251                     format="s16", layout="mono")
252                 frame.rate = sr
253                 for packet in stream.encode(frame):
254                     container.mux(packet)
255                 for packet in stream.encode(None):
256                     container.mux(packet)
257             return path
258         except Exception:
259             path = path.with_suffix(".wav")
260 
261     import wave
262 
263     with wave.open(str(path), "wb") as w:
264         w.setnchannels(1)
265         w.setsampwidth(2)
266         w.setframerate(sr)
267         w.writeframes((pcm * 32767).astype(np.int16).tobytes())
268     return path
269 
270 
271 def grab_frames(video, times, out_paths, width):
272     """Best-effort screenshots. Returns the paths actually written."""
273     try:
274         import av
275     except ImportError:
276         return {}
277     written = {}
278     try:
279         with av.open(str(video)) as container:
280             if not container.streams.video:
281                 return {}
282             vs = container.streams.video[0]
283             vs.thread_type = "AUTO"
284             for i, (t, dest) in enumerate(zip(times, out_paths)):
285                 try:
286                     container.seek(int(t / vs.time_base), stream=vs)
287                     for frame in container.decode(vs):
288                         if frame.time is None or frame.time + 0.001 < t:
289                             continue
290                         img = frame.to_image()
291                         if width and img.width > width:
292                             img = img.resize((width, max(1, round(img.height * width / img.width))))
293                         img.save(str(dest), quality=82)
294                         written[i] = dest
295                         break
296                 except Exception:
297                     continue
298     except Exception:
299         return written
300     return written
301 
302 
303 def esc(text):
304     return text.replace("\t", " ").replace("\n", " ").strip()
305 
306 
307 # ----------------------------------------------------------- mining page ----
308 
309 MINE_PAGE = """<!doctype html>
310 <html lang="__LANG__">
311 <head>
312 <meta charset="utf-8">
313 <title>__TITLE__</title>
314 <style>
315   :root { --bg:#0e1116; --panel:#161a21; --line:#252b36; --fg:#edf1f7;
316           --dim:#8b95a7; --accent:#7dd3fc; --size:26px; }
317   * { box-sizing:border-box; }
318   html,body { height:100%; margin:0; }
319   body { background:var(--bg); color:var(--fg); display:flex; flex-direction:column;
320          font-family:"Inter","Segoe UI","Yu Gothic UI","Meiryo","Noto Sans CJK JP",
321                      "Noto Sans",system-ui,sans-serif; }
322   header { display:flex; gap:10px; align-items:center; flex-wrap:wrap; padding:10px 16px;
323            background:var(--panel); border-bottom:1px solid var(--line); font-size:13px; }
324   header .grow { flex:1; }
325   button { font:inherit; font-size:13px; color:var(--fg); cursor:pointer; background:#1f2531;
326            border:1px solid var(--line); border-radius:6px; padding:6px 11px; }
327   button:hover { background:#29313f; }
328   button.on { background:var(--accent); border-color:var(--accent); color:#06202c; }
329   main { flex:1; display:flex; min-height:0; flex-wrap:wrap; }
330   #left { flex:1 1 460px; min-width:320px; display:flex; flex-direction:column;
331           border-right:1px solid var(--line); }
332   video { width:100%; background:#000; max-height:52vh; }
333   #cur { padding:16px 20px; font-size:var(--size); line-height:1.7; min-height:3em;
334          user-select:text; border-top:1px solid var(--line); }
335   #list { flex:1 1 380px; min-width:300px; overflow-y:auto; padding:14px 18px 30vh; }
336   .cue { padding:8px 11px; border-radius:6px; border-left:3px solid transparent;
337          font-size:calc(var(--size)*.82); line-height:1.7; user-select:text; margin-bottom:8px; }
338   .cue::before { content:attr(data-time); display:block; font-size:11px; color:var(--dim);
339                  user-select:none; margin-bottom:2px; font-variant-numeric:tabular-nums; }
340   .cue:hover { background:#151a22; }
341   .cue.active { background:#1a222e; border-left-color:var(--accent); }
342   .jump { float:right; font-size:11px; color:var(--dim); cursor:pointer; user-select:none; }
343 </style>
344 </head>
345 <body>
346 <header>
347   <b>__TITLE__</b>
348   <span class="grow"></span>
349   <button id="smaller">A-</button>
350   <button id="bigger">A+</button>
351   <button id="follow" class="on">Follow</button>
352   <button id="loop">Loop cue</button>
353   <button id="copy">Copy all</button>
354 </header>
355 <main>
356   <div id="left">
357     <video id="v" src="__VIDEO__" controls preload="metadata"></video>
358     <div id="cur"></div>
359   </div>
360   <div id="list"></div>
361 </main>
362 <script>
363 const CUES = __CUES__;
364 const v = document.getElementById("v");
365 const list = document.getElementById("list");
366 const cur = document.getElementById("cur");
367 let follow = true, looping = false, active = -1;
368 
369 function fmt(t) {
370   const m = Math.floor(t / 60), s = Math.floor(t % 60);
371   return (m < 10 ? "0" : "") + m + ":" + (s < 10 ? "0" : "") + s;
372 }
373 
374 CUES.forEach((c, i) => {
375   const d = document.createElement("div");
376   d.className = "cue";
377   d.dataset.time = fmt(c.start);
378   const j = document.createElement("span");
379   j.className = "jump";
380   j.textContent = "play";
381   j.onclick = (e) => { e.stopPropagation(); v.currentTime = c.start; v.play(); };
382   d.appendChild(j);
383   d.appendChild(document.createTextNode(c.text));
384   d.onclick = () => { v.currentTime = c.start; };
385   list.appendChild(d);
386 });
387 
388 function setActive(i) {
389   if (i === active) return;
390   const prev = list.children[active];
391   if (prev) prev.classList.remove("active");
392   active = i;
393   const el = list.children[i];
394   cur.textContent = "";
395   if (!el) return;
396   el.classList.add("active");
397   cur.appendChild(document.createTextNode(CUES[i].text));
398   if (follow) el.scrollIntoView({ block: "center", behavior: "smooth" });
399 }
400 
401 function syncNow() {
402   const t = v.currentTime;
403   if (looping && active >= 0 && !v.paused) {
404     const c = CUES[active];
405     if (t > c.end + 0.05) { v.currentTime = c.start; return; }
406   }
407   let i = -1;
408   for (let k = 0; k < CUES.length; k++) {
409     if (t >= CUES[k].start - 0.05 && t <= CUES[k].end + 0.35) { i = k; break; }
410   }
411   // While scrubbing between cues, keep showing the one just passed.
412   if (i < 0) {
413     for (let k = CUES.length - 1; k >= 0; k--) {
414       if (t >= CUES[k].start) { i = k; break; }
415     }
416   }
417   if (i >= 0) setActive(i);
418 }
419 
420 // timeupdate only fires during playback; seeked covers scrubbing while paused,
421 // which is most of what mining actually involves.
422 ["timeupdate", "seeked", "loadedmetadata", "play"].forEach(
423   (ev) => v.addEventListener(ev, syncNow));
424 
425 document.getElementById("bigger").onclick = () => bump(3);
426 document.getElementById("smaller").onclick = () => bump(-3);
427 function bump(d) {
428   const s = parseInt(getComputedStyle(document.documentElement)
429         .getPropertyValue("--size"), 10) + d;
430   document.documentElement.style.setProperty("--size",
431         Math.max(14, Math.min(60, s)) + "px");
432 }
433 document.getElementById("follow").onclick = (e) => {
434   follow = !follow; e.target.className = follow ? "on" : "";
435 };
436 document.getElementById("loop").onclick = (e) => {
437   looping = !looping; e.target.className = looping ? "on" : "";
438 };
439 document.getElementById("copy").onclick = async () => {
440   try {
441     await navigator.clipboard.writeText(CUES.map((c) => c.text).join("\\n"));
442     const b = document.getElementById("copy");
443     b.textContent = "Copied"; setTimeout(() => b.textContent = "Copy all", 1200);
444   } catch (e) {}
445 };
446 document.addEventListener("keydown", (e) => {
447   if (e.key === " ") { e.preventDefault(); v.paused ? v.play() : v.pause(); }
448   if (e.key === "ArrowLeft" && active > 0) v.currentTime = CUES[active - 1].start;
449   if (e.key === "ArrowRight" && active < CUES.length - 1) v.currentTime = CUES[active + 1].start;
450 });
451 </script>
452 </body>
453 </html>
454 """
455 
456 
457 def write_mine_page(path, cues, lang):
458     """A self-contained page: the clip plus selectable, synced subtitles.
459 
460     Yomitan mines from real DOM text, so the value is in the subtitles being
461     hoverable HTML rather than pixels burned into the video.
462     """
463     import json
464 
465     data = [{"start": round(s, 3), "end": round(e, 3), "text": t} for s, e, t in cues]
466     html = (MINE_PAGE
467             .replace("__CUES__", json.dumps(data, ensure_ascii=False))
468             .replace("__VIDEO__", path.name.replace('"', "%22"))
469             .replace("__TITLE__", path.stem.replace("<", "&lt;"))
470             .replace("__LANG__", "ja" if lang == "ja" else (lang or "en")))
471     out = path.with_suffix(".html")
472     out.write_text(html, encoding="utf-8")
473     return out
474 
475 
476 def export_anki(path, cues, translations, args, log=log):
477     """One row per sentence: text, translation, audio, screenshot."""
478     from faster_whisper.audio import decode_audio
479 
480     media = Path(args.anki_media) if args.anki_media else path.parent / (path.stem + "_media")
481     media.mkdir(parents=True, exist_ok=True)
482 
483     sr = 24000
484     try:
485         audio = decode_audio(str(path), sampling_rate=sr)
486     except Exception as e:
487         log("cannot re-decode audio for cards: %s" % e)
488         return None
489 
490     stem = "".join(c if (c.isalnum() or c in "-_") else "_" for c in path.stem)[:48]
491     audio_names, image_names = [], []
492 
493     for i, (start, end, _) in enumerate(cues, 1):
494         clip = slice_audio(audio, sr, start, end, args.anki_pad)
495         dest = write_audio_clip(clip, sr, media / ("%s_%04d.mp3" % (stem, i)))
496         audio_names.append(dest.name)
497 
498     if args.anki_images:
499         mids = [(s + e) / 2 for s, e, _ in cues]
500         dests = [media / ("%s_%04d.jpg" % (stem, i)) for i in range(1, len(cues) + 1)]
501         got = grab_frames(path, mids, dests, args.anki_image_width)
502         image_names = [dests[i].name if i in got else "" for i in range(len(cues))]
503         if not got:
504             log("no screenshots (PyAV/Pillow could not decode video frames)")
505     else:
506         image_names = [""] * len(cues)
507 
508     tsv = path.with_suffix(".anki.tsv")
509     with open(tsv, "w", encoding="utf-8", newline="") as fh:
510         for i, (start, end, text) in enumerate(cues):
511             row = [
512                 esc(text),
513                 esc(translations[i]) if translations else "",
514                 "[sound:%s]" % audio_names[i],
515                 '<img src="%s">' % image_names[i] if image_names[i] else "",
516                 esc(path.name),
517                 ts(start),
518             ]
519             fh.write("\t".join(row) + "\n")
520 
521     log("%d cards -> %s" % (len(cues), tsv.name))
522     log("   media -> %s" % media)
523     if not args.anki_media:
524         target = anki_media_dir()
525         if target:
526             log("   copy the media files into: %s" % target)
527         else:
528             log("   copy the media files into your Anki collection.media folder")
529     return tsv
530 
531 
532 # ------------------------------------------------------------ processing ----
533 
534 class Subtitler:
535     def __init__(self, args):
536         from faster_whisper import WhisperModel
537 
538         self.args = args
539         log("loading %r on %s/%s ..." % (args.model, args.device, args.compute))
540         t0 = time.time()
541         self.model = WhisperModel(args.model, device=args.device,
542                                   compute_type=args.compute,
543                                   cpu_threads=args.threads, num_workers=1)
544         log("model ready in %.1fs" % (time.time() - t0))
545         self._tmodel = None
546         self._tname = None
547 
548     def translator(self):
549         """A model that can actually translate.
550 
551         large-v3-turbo is a transcription-only fine-tune: asked to translate it
552         returns the source language instead, silently. Fall back to a model
553         that supports the task rather than emitting untranslated text.
554         """
555         from faster_whisper import WhisperModel
556 
557         a = self.args
558         name = a.translate_model or a.model
559         if not a.translate_model and "turbo" in a.model.lower():
560             name = "small"
561             log("note: %s cannot translate (transcription-only fine-tune); "
562                 "using %r instead - override with --translate-model" % (a.model, name))
563         if name == a.model:
564             return self.model
565         if self._tmodel is None or self._tname != name:
566             log("loading translation model %r ..." % name)
567             self._tmodel = WhisperModel(name, device=a.device, compute_type=a.compute,
568                                         cpu_threads=a.threads, num_workers=1)
569             self._tname = name
570         return self._tmodel
571 
572     def run(self, path):
573         from faster_whisper.audio import decode_audio
574 
575         a = self.args
576         out = path.with_suffix("." + a.format)
577         if out.exists() and not a.overwrite:
578             log("skip (exists): %s" % out.name)
579             return out
580 
581         try:
582             audio = decode_audio(str(path), sampling_rate=16000)
583         except Exception as e:
584             log("cannot decode %s: %s" % (path.name, e))
585             return None
586         dur = len(audio) / 16000
587         if dur < 0.2:
588             log("skip (no audio): %s" % path.name)
589             return None
590 
591         t0 = time.time()
592         engine = self.translator() if a.translate else self.model
593         segments, info = engine.transcribe(
594             audio,
595             language=None if a.lang == "auto" else a.lang,
596             task="translate" if a.translate else "transcribe",
597             beam_size=a.beam,
598             temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
599             condition_on_previous_text=True,
600             vad_filter=True,
601             word_timestamps=True,
602             no_speech_threshold=0.6,
603             log_prob_threshold=-1.0,
604         )
605         segments = list(segments)
606         took = time.time() - t0
607 
608         if a.anki:
609             # Cards want whole sentences; subtitles want display-sized chunks.
610             groups = group_sentences(collect_words(segments), a.max_dur, a.max_gap)
611             card_cues = [(g[0].start, g[-1].end, text_of(g)) for g in groups]
612         else:
613             card_cues = []
614 
615         cues = build_cues(segments, a.max_chars, a.max_dur, a.max_gap)
616         if not cues:
617             log("no speech found in %s" % path.name)
618             return None
619 
620         WRITERS[a.format](cues, out, a.width)
621         log("%s  ->  %s  (%d cues, %s, %.0fs audio in %.0fs)"
622             % (path.name, out.name, len(cues),
623                getattr(info, "language", a.lang), dur, took))
624 
625         if a.mine:
626             page = write_mine_page(path, cues, getattr(info, "language", a.lang))
627             log("mining page  ->  %s" % page.name)
628 
629         if a.anki:
630             translations = self.translate_cues(audio, card_cues) if a.anki_translate else None
631             export_anki(path, card_cues, translations, a)
632         return out
633 
634     def translate_cues(self, audio, cues):
635         """English for the back of each card.
636 
637         Translating one short sentence at a time starves Whisper of context and
638         it often just echoes the source language back. Translating the whole
639         clip once and aligning by timestamp is both better and faster.
640         """
641         log("translating for card backs...")
642         try:
643             segs, _ = self.translator().transcribe(
644                 audio, language=None if self.args.lang == "auto" else self.args.lang,
645                 task="translate", beam_size=self.args.beam,
646                 temperature=[0.0, 0.2, 0.4],
647                 condition_on_previous_text=True, vad_filter=True)
648             segs = list(segs)
649         except Exception as e:
650             log("translation failed: %s" % e)
651             return [""] * len(cues)
652 
653         out = []
654         for start, end, _ in cues:
655             parts = [s.text.strip() for s in segs
656                      if s.start < end - 0.15 and s.end > start + 0.15]
657             out.append(re.sub(r"\s+", " ", " ".join(parts)).strip())
658         return out
659 
660 
661 def obs_recording_folder():
662     """Where OBS writes recordings and replay-buffer clips, from its own config.
663 
664     Which key holds it depends on the output mode, so read Mode first rather
665     than guessing.
666     """
667     import configparser
668 
669     root = Path(os.environ.get("APPDATA", "")) / "obs-studio"
670     if not root.is_dir():
671         return None
672 
673     profile_dir = None
674     for name in ("user.ini", "global.ini"):
675         cfg = configparser.RawConfigParser(strict=False)
676         try:
677             cfg.read(root / name, encoding="utf-8-sig")
678             profile_dir = cfg.get("Basic", "ProfileDir", fallback=None) or profile_dir
679         except (configparser.Error, OSError):
680             pass
681 
682     profiles = root / "basic" / "profiles"
683     candidates = []
684     if profile_dir and (profiles / profile_dir).is_dir():
685         candidates.append(profiles / profile_dir)
686     if profiles.is_dir():
687         candidates.extend(p for p in profiles.iterdir() if p.is_dir())
688 
689     for prof in candidates:
690         cfg = configparser.RawConfigParser(strict=False)
691         try:
692             cfg.read(prof / "basic.ini", encoding="utf-8-sig")
693         except (configparser.Error, OSError):
694             continue
695         mode = (cfg.get("Output", "Mode", fallback="") or "").lower()
696         if mode.startswith("adv"):
697             keys = [("AdvOut", "RecFilePath"), ("AdvOut", "FFFilePath"),
698                     ("SimpleOutput", "FilePath")]
699         else:
700             keys = [("SimpleOutput", "FilePath"), ("AdvOut", "RecFilePath")]
701         for section, key in keys:
702             raw = cfg.get(section, key, fallback=None)
703             if not raw:
704                 continue
705             path = Path(raw.replace("\\\\", "\\").strip())
706             if path.is_dir():
707                 return path
708     return None
709 
710 
711 class _Limited:
712     """Feeds copyfile only the bytes belonging to the requested range."""
713 
714     def __init__(self, fh, remaining):
715         self.fh, self.remaining = fh, remaining
716 
717     def read(self, size=-1):
718         if self.remaining <= 0:
719             return b""
720         if size is None or size < 0:
721             size = self.remaining
722         data = self.fh.read(min(size, self.remaining))
723         self.remaining -= len(data)
724         return data
725 
726     def close(self):
727         self.fh.close()
728 
729 
730 class RangeHandler(SimpleHTTPRequestHandler):
731     """Static files with HTTP Range support.
732 
733     Browsers refuse to seek in a <video> unless the server answers range
734     requests, and Python's stock handler does not - so scrubbing a clip, which
735     is most of what mining involves, would silently not work.
736     """
737 
738     def log_message(self, *a):
739         pass
740 
741     def end_headers(self):
742         self.send_header("Accept-Ranges", "bytes")
743         SimpleHTTPRequestHandler.end_headers(self)
744 
745     def send_head(self):
746         rng = self.headers.get("Range")
747         if not rng:
748             return SimpleHTTPRequestHandler.send_head(self)
749 
750         path = self.translate_path(self.path)
751         if os.path.isdir(path):
752             return SimpleHTTPRequestHandler.send_head(self)
753         try:
754             fh = open(path, "rb")
755         except OSError:
756             self.send_error(404, "File not found")
757             return None
758 
759         size = os.fstat(fh.fileno()).st_size
760         m = re.match(r"bytes=(\d*)-(\d*)\s*$", rng.strip())
761         if not m or (not m.group(1) and not m.group(2)):
762             fh.close()
763             self.send_error(400, "Malformed Range")
764             return None
765 
766         if not m.group(1):                       # bytes=-N  (last N bytes)
767             start, end = max(0, size - int(m.group(2))), size - 1
768         else:
769             start = int(m.group(1))
770             end = int(m.group(2)) if m.group(2) else size - 1
771         end = min(end, size - 1)
772 
773         if start >= size or start > end:
774             fh.close()
775             self.send_response(416)
776             self.send_header("Content-Range", "bytes */%d" % size)
777             self.send_header("Content-Length", "0")
778             self.end_headers()
779             return None
780 
781         self.send_response(206)
782         self.send_header("Content-Type", self.guess_type(path))
783         self.send_header("Content-Range", "bytes %d-%d/%d" % (start, end, size))
784         self.send_header("Content-Length", str(end - start + 1))
785         self.end_headers()
786         fh.seek(start)
787         return _Limited(fh, end - start + 1)
788 
789 
790 def serve(folder, port):
791     from functools import partial
792 
793     handler = partial(RangeHandler, directory=str(folder))
794     srv = ThreadingHTTPServer(("127.0.0.1", port), handler)
795     threading.Thread(target=srv.serve_forever, daemon=True).start()
796     log("serving %s at http://127.0.0.1:%d/  (video seeking enabled)" % (folder, port))
797     pages = sorted(p.name for p in Path(folder).glob("*.html"))
798     for name in pages[-10:]:
799         log("   http://127.0.0.1:%d/%s" % (port, name))
800     return srv
801 
802 
803 def stable(path, checks=3, delay=1.0):
804     """Wait until a file stops growing, so we do not read a half-written clip."""
805     last = -1
806     for _ in range(120):
807         try:
808             size = path.stat().st_size
809         except OSError:
810             return False
811         if size == last:
812             checks -= 1
813             if checks <= 0:
814                 return size > 0
815         else:
816             checks = 3
817             last = size
818         time.sleep(delay)
819     return False
820 
821 
822 def watch(sub, folder, args):
823     folder = Path(folder)
824     if not folder.is_dir():
825         raise SystemExit("not a folder: %s" % folder)
826     seen = {p.resolve() for p in folder.iterdir()
827             if p.suffix.lower() in VIDEO_EXT}
828     log("watching %s for new clips (%d already present) - Ctrl+C to stop"
829         % (folder, len(seen)))
830     while True:
831         try:
832             for p in sorted(folder.iterdir()):
833                 if p.suffix.lower() not in VIDEO_EXT:
834                     continue
835                 r = p.resolve()
836                 if r in seen:
837                     continue
838                 seen.add(r)
839                 log("new clip: %s" % p.name)
840                 if stable(p):
841                     sub.run(p)
842                 else:
843                     log("gave up waiting for %s to finish writing" % p.name)
844             time.sleep(args.poll)
845         except KeyboardInterrupt:
846             print()
847             log("stopped")
848             return
849 
850 
851 def main():
852     p = argparse.ArgumentParser(
853         description="Subtitle recordings and replay clips",
854         formatter_class=argparse.ArgumentDefaultsHelpFormatter)
855     p.add_argument("files", nargs="*", help="video/audio files to subtitle")
856     p.add_argument("--watch", nargs="?", const="auto", default=None, metavar="FOLDER",
857                    help="watch a folder and subtitle clips as they appear; "
858                         "with no folder, read OBS's own recording path")
859     p.add_argument("--poll", type=float, default=3.0, help="watch interval (s)")
860 
861     p.add_argument("--model", default="large-v3-turbo",
862                    help="quality matters more than speed here")
863     p.add_argument("--lang", default="fi", help="language code, or 'auto'")
864     p.add_argument("--device", default="cpu", choices=["cpu", "cuda"])
865     p.add_argument("--compute", default="int8")
866     p.add_argument("--threads", type=int, default=max(2, (os.cpu_count() or 8) - 2))
867     p.add_argument("--beam", type=int, default=5)
868     p.add_argument("--translate", action="store_true",
869                    help="write English subtitles instead of the original language")
870     p.add_argument("--translate-model", default=None,
871                    help="model used for translation; defaults to --model, except "
872                         "for turbo builds which cannot translate at all")
873 
874     p.add_argument("--format", default="srt", choices=sorted(WRITERS))
875     p.add_argument("--width", type=int, default=42, help="max characters per line")
876     p.add_argument("--max-chars", type=int, default=84, help="max characters per cue")
877     p.add_argument("--max-dur", type=float, default=6.0, help="max seconds per cue")
878     p.add_argument("--max-gap", type=float, default=0.8,
879                    help="silence (s) that forces a new cue")
880     p.add_argument("--overwrite", action="store_true",
881                    help="re-subtitle files that already have output")
882 
883     g = p.add_argument_group("mining")
884     g.add_argument("--mine", dest="mine", action="store_true", default=True,
885                    help="write a browser page with the clip and hoverable subtitles")
886     g.add_argument("--no-mine", dest="mine", action="store_false")
887     g.add_argument("--serve", nargs="?", type=int, const=8778, default=None,
888                    metavar="PORT",
889                    help="serve the mining pages over http with video seeking, "
890                         "so Yomitan works without file-URL permissions")
891 
892     g = p.add_argument_group("anki cards")
893     g.add_argument("--anki", action="store_true",
894                    help="also export one card per sentence, with clipped audio")
895     g.add_argument("--anki-translate", action="store_true",
896                    help="add an English translation to each card")
897     g.add_argument("--anki-images", dest="anki_images", action="store_true", default=True,
898                    help="grab a screenshot per card")
899     g.add_argument("--no-anki-images", dest="anki_images", action="store_false")
900     g.add_argument("--anki-image-width", type=int, default=640)
901     g.add_argument("--anki-pad", type=float, default=0.25,
902                    help="seconds of padding around each audio clip")
903     g.add_argument("--anki-media", default=None,
904                    help="write media straight into Anki's collection.media")
905 
906     args = p.parse_args()
907     if not args.files and not args.watch:
908         p.error("give some files, or --watch a folder")
909 
910     if args.watch == "auto":
911         found = obs_recording_folder()
912         if not found:
913             p.error("could not find OBS's recording folder; pass --watch FOLDER")
914         args.watch = str(found)
915         log("OBS records to: %s" % args.watch)
916 
917     sub = Subtitler(args)
918 
919     done = []
920     for pattern in args.files:
921         path = Path(pattern)
922         matches = [path] if path.exists() else sorted(Path().glob(pattern))
923         if not matches:
924             log("no such file: %s" % pattern)
925         for m in matches:
926             sub.run(m)
927             done.append(m)
928 
929     srv = None
930     if args.serve:
931         folder = Path(args.watch) if args.watch else (
932             done[0].parent if done else Path("."))
933         srv = serve(folder.resolve(), args.serve)
934 
935     if args.watch:
936         watch(sub, args.watch, args)
937     elif srv is not None:
938         log("Ctrl+C to stop serving")
939         try:
940             while True:
941                 time.sleep(3600)
942         except KeyboardInterrupt:
943             print()
944     return 0
945 
946 
947 if __name__ == "__main__":
948     sys.exit(main())