commit 40b2679948d4356b4fd5a48781ec409437eef4ab
equwal <truex@equwal.com>
2026-09-20 15:28:12 -0700
Three outputs: srt, clean mp4 for YouTube, mkv with subs embedded
The single mp4 carrying a mov_text track served neither use well. YouTube
wants captions uploaded separately as an .srt and handles embedded mov_text
unreliably; a local player wants one self-contained file.
So render once and remux:
- .srt the timing file alone (HoshiReader whispersync)
- .mp4 cover + audio, no subtitle track, for YouTube
- .mkv the same streams copied, with the SRT embedded and defaulted on,
for MPV/VLC/Jellyfin
MKV rather than MP4 for the embedded case because MKV stores SRT natively,
where MP4 must down-convert to mov_text. The mkv is a stream copy of the mp4,
so the second file costs seconds regardless of book length.
Also: surface the backend's real failure reason. subplz exits 0 even when a
sync fails, writing the cause to a .subfail file we ignored - so a broken
torchaudio install reported itself as "the audio and text may not match",
blaming the user's content for a library problem. Aligner.failure_reason()
reads that file, with the old guess kept only as a fallback.
backend/aligner.py | 29 +++++++++++++++++++++++++++++
backend/api.py | 2 +-
backend/render.py | 49 ++++++++++++++++++++++++++++++++++++++++++-------
backend/runner.py | 49 ++++++++++++++++++++++++++++++++++++++++++++++---
frontend/app.js | 17 +++++++++++++----
frontend/index.html | 2 +-
6 files changed, 132 insertions(+), 16 deletions(-)
diff --git a/backend/aligner.py b/backend/aligner.py
index 2f93d9b..b14da54 100644
--- a/backend/aligner.py
+++ b/backend/aligner.py
@@ -95,6 +95,14 @@ class Aligner(ABC):
"""Anything the user should know about this language, or None."""
return None
+ def failure_reason(self, req: "AlignRequest") -> str | None:
+ """Why the run produced nothing, if the backend left an explanation.
+
+ Needed because a backend may exit 0 and still have failed, so the exit
+ code alone cannot be trusted to mean success.
+ """
+ return None
+
# ---------------------------------------------------------------------------
# subplz
@@ -269,6 +277,27 @@ class SubPlzAligner(Aligner):
# Fall back in case upstream changes the convention.
return next(iter(sorted(req.out_dir.glob("*.srt"))), None)
+ def failure_reason(self, req: AlignRequest) -> str | None:
+ """Read subplz's own explanation out of the .subfail it leaves behind.
+
+ subplz exits 0 even when a sync fails, so the exit code cannot be
+ trusted. Without this the runner reports "the audio and text may not
+ match" for every failure - including ones that have nothing to do with
+ the content, such as a broken library.
+ """
+ for fail in sorted(req.out_dir.glob("*.subfail")):
+ try:
+ text = fail.read_text(encoding="utf-8", errors="replace").strip()
+ except OSError:
+ continue
+ if not text:
+ continue
+ # The file repeats the audio path on every line; keep the last
+ # line, which carries the actual reason.
+ lines = [ln.strip() for ln in text.splitlines() if ln.strip()]
+ return lines[-1] if lines else None
+ return None
+
def language_note(self, code: str) -> str | None:
lang = languages.get(code)
if lang is None or not lang.needs_nlp_flag:
diff --git a/backend/api.py b/backend/api.py
index 19a2a1e..c0347e3 100644
--- a/backend/api.py
+++ b/backend/api.py
@@ -488,7 +488,7 @@ def delete_job(
@router.get("/jobs/{job_id}/files/{kind}")
def download(
job_id: str,
- kind: Literal["srt", "video", "metadata", "log"],
+ kind: Literal["srt", "video", "video_embedded", "metadata", "log"],
account: Annotated[Account, Depends(get_account)],
session: Annotated[Session, Depends(get_session)],
):
diff --git a/backend/render.py b/backend/render.py
index b9172f1..a891816 100644
--- a/backend/render.py
+++ b/backend/render.py
@@ -181,20 +181,25 @@ def build_canvas(cover: Path | None, dest: Path) -> Path:
def render_video(
audio: Path,
- subtitles: Path,
canvas: Path,
dest: Path,
duration: float | None = None,
) -> Path:
- """Mux canvas + audio + subtitles into an MP4."""
+ """Cover image + audio into a clean MP4, carrying no subtitle track.
+
+ Subtitle-free on purpose. This is the file for YouTube, where captions are
+ attached separately as an .srt: YouTube's handling of an embedded mov_text
+ track is unreliable, and uploading the .srt is the supported route.
+
+ This is the expensive step - `mux_subtitles` reuses its output instead of
+ encoding a second time.
+ """
dest.parent.mkdir(parents=True, exist_ok=True)
cmd = [
"ffmpeg", "-hide_banner", "-v", "error", "-y",
"-loop", "1", "-r", str(settings.video_fps), "-i", str(canvas),
"-i", str(audio),
- # Declare the format: ffmpeg will not always sniff an srt correctly.
- "-f", "srt", "-i", str(subtitles),
]
if duration:
cmd += ["-t", f"{duration:.3f}"]
@@ -203,14 +208,12 @@ def render_video(
encoder = resolve_encoder()
cmd += [
- "-map", "0:v", "-map", "1:a", "-map", "2:s",
+ "-map", "0:v", "-map", "1:a",
# Carry chapter marks through; harmless where they are ignored.
"-map_chapters", "1",
"-c:v", encoder,
"-pix_fmt", "yuv420p",
"-c:a", "aac", "-b:a", "128k",
- # mov_text is the only subtitle codec MP4 carries.
- "-c:s", "mov_text",
# Lets a player start without reading the whole file first.
"-movflags", "+faststart",
]
@@ -229,3 +232,35 @@ def render_video(
tail = " | ".join(err[-3:]) if err else "no output"
raise RenderError(f"ffmpeg could not build the video: {tail}")
return dest
+
+
+def mux_subtitles(video: Path, subtitles: Path, dest: Path) -> Path:
+ """Copy `video` into an MKV carrying `subtitles` as a selectable track.
+
+ For local playback - MPV, VLC, Jellyfin - where one self-contained file
+ beats juggling a video and a sidecar .srt.
+
+ MKV rather than MP4 because MKV stores SRT natively, keeping the cue text
+ intact; MP4 must convert to mov_text, which is a lossy reduction. Nothing
+ is re-encoded here: the streams are copied, so this costs seconds however
+ long the book is.
+ """
+ dest.parent.mkdir(parents=True, exist_ok=True)
+ cmd = [
+ "ffmpeg", "-hide_banner", "-v", "error", "-y",
+ "-i", str(video),
+ # Declare the format; ffmpeg does not always sniff an srt correctly.
+ "-f", "srt", "-i", str(subtitles),
+ "-map", "0", "-map", "1",
+ "-c", "copy", "-c:s", "srt",
+ # Players pick this up automatically instead of needing it turned on.
+ "-disposition:s:0", "default",
+ "-metadata:s:s:0", "title=Aligned subtitles",
+ str(dest),
+ ]
+ proc = subprocess.run(cmd, capture_output=True, timeout=1800)
+ if proc.returncode != 0 or not dest.exists():
+ err = proc.stderr.decode("utf-8", errors="replace").strip().splitlines()
+ tail = " | ".join(err[-3:]) if err else "no output"
+ raise RenderError(f"ffmpeg could not embed the subtitles: {tail}")
+ return dest
diff --git a/backend/runner.py b/backend/runner.py
index 6f2fd2e..871a48a 100644
--- a/backend/runner.py
+++ b/backend/runner.py
@@ -470,6 +470,11 @@ def _collect_artifacts(job_id: str, paths: Paths, log_path: Path,
produced = aligner.locate_output(request)
if produced is None or not produced.exists():
+ # The backend may exit 0 and still have failed, so ask it why before
+ # falling back to guessing at the content.
+ reason = aligner.failure_reason(request)
+ if reason:
+ raise JobFailed(f"{aligner.name} failed: {reason}")
raise JobFailed(
f"{aligner.name} finished but produced no subtitle file. The audio "
"and text may not match, or the language may be wrong for this book."
@@ -485,22 +490,43 @@ def _collect_artifacts(job_id: str, paths: Paths, log_path: Path,
srt_key = f"{job_id}/{download_name}"
size = storage.put_file(srt_key, produced)
+ # Three deliverables, because they serve three different jobs:
+ # .srt - the timing file on its own (HoshiReader whispersync)
+ # .mp4 - clean video, no subtitle track, for YouTube (captions are
+ # uploaded separately there)
+ # .mkv - the same video with the subtitles embedded, for local players
+ # The mkv is a stream copy of the mp4, so the second file is nearly free.
video_name = video_key = None
- video_size = 0
+ embed_name = embed_key = None
+ video_size = embed_size = 0
+
if settings.render_video:
try:
_set(job_id, stage="Rendering video", progress=0.94)
scratch = paths.root / "video"
cover = render.extract_cover(request.text, scratch)
canvas = render.build_canvas(cover, scratch / "canvas.png")
+
video_name = f"{stem}.{language}.mp4"
out_video = scratch / video_name
render.render_video(
- audio=video_audio, subtitles=produced, canvas=canvas,
+ audio=video_audio, canvas=canvas,
dest=out_video, duration=duration,
)
video_key = f"{job_id}/{video_name}"
video_size = storage.put_file(video_key, out_video)
+
+ try:
+ _set(job_id, stage="Embedding subtitles", progress=0.97)
+ embed_name = f"{stem}.{language}.mkv"
+ out_embed = scratch / embed_name
+ render.mux_subtitles(out_video, produced, out_embed)
+ embed_key = f"{job_id}/{embed_name}"
+ embed_size = storage.put_file(embed_key, out_embed)
+ except Exception as exc: # noqa: BLE001
+ embed_name = embed_key = None
+ log.warning("job %s: subtitle embed failed: %s", job_id, exc)
+
except Exception as exc: # noqa: BLE001 - never fail a job over the video
# The subtitles are the product, so a failed render is not fatal -
# but it must not be silent either, or it looks like it never ran.
@@ -538,12 +564,24 @@ def _collect_artifacts(job_id: str, paths: Paths, log_path: Path,
{
"filename": video_name,
"container": "mp4",
- "subtitles": "soft track (mov_text)",
+ "subtitles": "none - upload the .srt to YouTube separately",
+ "for": "youtube",
"size_bytes": video_size,
}
if video_name
else None
),
+ "video_embedded": (
+ {
+ "filename": embed_name,
+ "container": "mkv",
+ "subtitles": "embedded SRT track, enabled by default",
+ "for": "local playback (MPV, VLC, Jellyfin)",
+ "size_bytes": embed_size,
+ }
+ if embed_name
+ else None
+ ),
}
meta_path = paths.root / "metadata.json"
meta_path.write_text(json.dumps(metadata, ensure_ascii=False, indent=2),
@@ -562,6 +600,11 @@ def _collect_artifacts(job_id: str, paths: Paths, log_path: Path,
Artifact(job_id=job_id, kind="video", filename=video_name,
storage_key=video_key, size_bytes=video_size)
)
+ if embed_name and embed_key:
+ rows.append(
+ Artifact(job_id=job_id, kind="video_embedded", filename=embed_name,
+ storage_key=embed_key, size_bytes=embed_size)
+ )
with SessionLocal() as s:
for r in rows:
s.add(r)
diff --git a/frontend/app.js b/frontend/app.js
index df8e96d..1541fc1 100644
--- a/frontend/app.js
+++ b/frontend/app.js
@@ -391,20 +391,29 @@ function jobCard(j) {
const files = j.artifacts || [];
if (files.length) {
- const order = { srt: 0, video: 1, metadata: 2, log: 3 };
+ const order = { srt: 0, video: 1, video_embedded: 2, metadata: 3, log: 4 };
+ // Three deliverables for three different uses; say which is which, because
+ // "two video files" is otherwise baffling.
const label = {
srt: '⬇ Subtitles (.srt)',
- video: '⬇ Video (.mp4)',
+ video: '⬇ Video for YouTube (.mp4)',
+ video_embedded: '⬇ Video with subs built in (.mkv)',
metadata: 'Metadata',
log: 'Run log',
};
- const primary = new Set(['srt', 'video']);
+ const hint = {
+ srt: 'For HoshiReader, or upload alongside the YouTube video',
+ video: 'No subtitles baked in — add the .srt in YouTube Studio',
+ video_embedded: 'Subtitles inside the file, for MPV/VLC',
+ };
+ const primary = new Set(['srt', 'video', 'video_embedded']);
html += '<div class="job-files">' + files
.slice()
.sort((a, b) => (order[a.kind] ?? 9) - (order[b.kind] ?? 9))
.map((a) => {
const size = a.size_bytes ? ` <span class="dl-size">${fmtBytes(a.size_bytes)}</span>` : '';
- return `<a class="dl ${primary.has(a.kind) ? '' : 'secondary'}"
+ const t = hint[a.kind] ? ` title="${escapeHtml(hint[a.kind])}"` : '';
+ return `<a class="dl ${primary.has(a.kind) ? '' : 'secondary'}"${t}
href="${a.url}" download>${label[a.kind] || a.kind}${size}</a>`;
})
.join('') + '</div>';
diff --git a/frontend/index.html b/frontend/index.html
index 6db962a..3a3bc72 100644
--- a/frontend/index.html
+++ b/frontend/index.html
@@ -130,6 +130,6 @@
</p>
</footer>
-<script src="/app.js?v=6"></script>
+<script src="/app.js?v=7"></script>
</body>
</html>