Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


backend/matching.py (13028 bytes)

1 """Does this text actually match this audio?
2 
3 subplz fails late and unhelpfully when it does not: it transcribes the whole
4 book first, then reports "the generated transcript and the provided text file
5 are too different" and writes a .subfail. On CPU that is a wasted hour.
6 
7 So we ask the same question up front, cheaply: transcribe a couple of short
8 samples from the start of a few chapters, and score them against the book with
9 the backend's own rule. Same arithmetic, same threshold, ~10 seconds instead of
10 an hour.
11 
12 What the number means, and what it does not: a good score says the text lines
13 up with what is being read. A bad score usually means the wrong book, the wrong
14 edition (abridged vs full), an audiobook that opens with a publisher
15 announcement the book does not contain, or a book buried under front matter.
16 """
17 
18 from __future__ import annotations
19 
20 import json
21 import logging
22 import subprocess
23 import zipfile
24 from dataclasses import dataclass, field
25 from functools import lru_cache
26 from pathlib import Path
27 
28 from . import chapters as chapters_mod
29 from .settings import settings
30 
31 log = logging.getLogger(__name__)
32 
33 # Enough speech to fingerprint a chapter confidently. Sampling is adaptive, so
34 # a book that matches well normally pays for one of these and stops.
35 SAMPLE_SECONDS = 120
36 MAX_SAMPLES = 3
37 
38 
39 @dataclass
40 class ChapterScore:
41     audio_chapter: int
42     best_score: float
43     best_text_chapter: int | None
44     runner_up: float = 0.0
45     # How far clear of the runner-up. 1.0 means every chapter looked alike,
46     # which is the signature of the wrong book rather than a poor recording.
47     confidence: float = 0.0
48     accepted: bool = False
49     transcript_head: str = ""
50 
51     @property
52     def matched(self) -> bool:
53         return self.accepted
54 
55 
56 @dataclass
57 class MatchReport:
58     threshold: float
59     scores: list[ChapterScore] = field(default_factory=list)
60     text_chapters: int = 0
61     audio_chapters: int = 0
62     accept_at: float = 0.0
63     noise_floor: float = 0.0
64     verdict: str = "unknown"  # good | marginal | poor | unknown
65     summary: str = ""
66     warnings: list[str] = field(default_factory=list)
67     skipped: str | None = None
68 
69     @property
70     def best(self) -> float:
71         return max((s.best_score for s in self.scores), default=0.0)
72 
73     @property
74     def worst(self) -> float:
75         return min((s.best_score for s in self.scores), default=0.0)
76 
77     @property
78     def matched(self) -> int:
79         return sum(1 for s in self.scores if s.matched)
80 
81     @property
82     def confidence(self) -> float:
83         return max((s.confidence for s in self.scores), default=0.0)
84 
85     def as_dict(self) -> dict:
86         return {
87             "threshold": round(self.accept_at, 1),
88             "noise_floor": round(self.noise_floor, 1),
89             "confidence": round(self.confidence, 2),
90             "verdict": self.verdict,
91             "summary": self.summary,
92             "best": round(self.best, 1),
93             "worst": round(self.worst, 1),
94             "sampled": len(self.scores),
95             "matched": self.matched,
96             "text_chapters": self.text_chapters,
97             "audio_chapters": self.audio_chapters,
98             "warnings": self.warnings,
99             "skipped": self.skipped,
100             "scores": [
101                 {
102                     "audio_chapter": s.audio_chapter,
103                     "score": round(s.best_score, 1),
104                     "runner_up": round(s.runner_up, 1),
105                     "confidence": round(s.confidence, 2),
106                     "text_chapter": s.best_text_chapter,
107                 }
108                 for s in self.scores
109             ],
110         }
111 
112 
113 # ---------------------------------------------------------------------------
114 # book text
115 # ---------------------------------------------------------------------------
116 
117 def book_chapters(text_path: Path) -> list[str]:
118     """The comparable units, split the way the backend will split them.
119 
120     This matters more than it looks. subplz gives an epub one text chapter per
121     spine document, but a .txt exactly one chapter for the whole file - so a
122     flat text file offers the matcher a single anchor no matter how many
123     chapters the audio has.
124     """
125     if text_path.suffix.lower() != ".epub":
126         return [text_path.read_text(encoding="utf-8", errors="replace")]
127 
128     try:
129         from bs4 import BeautifulSoup
130 
131         chapters: list[str] = []
132         with zipfile.ZipFile(text_path) as zf:
133             names = [
134                 n for n in zf.namelist()
135                 if n.lower().endswith((".xhtml", ".html", ".htm"))
136             ]
137             for name in sorted(names):
138                 try:
139                     raw = zf.read(name).decode("utf-8", errors="replace")
140                 except (KeyError, OSError):
141                     continue
142                 soup = BeautifulSoup(raw, "html.parser")
143                 for bad in soup(["script", "style"]):
144                     bad.decompose()
145                 # subplz joins paragraph texts with no separator, so match that.
146                 text = "".join(
147                     p.get_text(" ", strip=True) for p in soup.find_all("p")
148                 ) or soup.get_text(" ", strip=True)
149                 if text.strip():
150                     chapters.append(text)
151         return chapters
152     except zipfile.BadZipFile:
153         return []
154 
155 
156 # ---------------------------------------------------------------------------
157 # audio sampling
158 # ---------------------------------------------------------------------------
159 
160 def audio_chapter_starts(audio: Path) -> list[float]:
161     """Start time of each chapter, or [0.0] for a flat file."""
162     try:
163         out = subprocess.run(
164             ["ffprobe", "-v", "error", "-show_chapters", "-print_format", "json",
165              str(audio)],
166             capture_output=True, timeout=120, check=True,
167         )
168         chapters = json.loads(out.stdout.decode("utf-8")).get("chapters", [])
169         starts = [float(c["start_time"]) for c in chapters]
170         return starts or [0.0]
171     except (subprocess.SubprocessError, ValueError, OSError, KeyError):
172         return [0.0]
173 
174 
175 def read_samples(audio: Path, start: float, seconds: int):
176     """`seconds` of audio from `start`, as the float32 mono 16 kHz the model wants."""
177     import numpy as np
178 
179     proc = subprocess.run(
180         [
181             "ffmpeg", "-v", "error",
182             "-max_error_rate", "1.0", "-err_detect", "ignore_err",
183             "-ss", f"{start:.3f}", "-t", str(seconds), "-i", str(audio),
184             "-map", "0:a:0", "-f", "s16le", "-acodec", "pcm_s16le",
185             "-ac", "1", "-ar", "16000", "-",
186         ],
187         capture_output=True, timeout=300,
188     )
189     if proc.returncode != 0 or not proc.stdout:
190         return None
191     return np.frombuffer(proc.stdout, np.int16).astype(np.float32) / 32768.0
192 
193 
194 @lru_cache(maxsize=1)
195 def _model():
196     """The same tiny model the alignment itself uses, loaded once."""
197     from faster_whisper import WhisperModel
198 
199     return WhisperModel(settings.model, device="cpu", compute_type="int8")
200 
201 
202 def transcribe_sample(samples, language: str) -> str:
203     segments, _ = _model().transcribe(
204         samples, language=language, beam_size=5, without_timestamps=True
205     )
206     return "".join(seg.text for seg in segments)
207 
208 
209 # ---------------------------------------------------------------------------
210 # the check
211 # ---------------------------------------------------------------------------
212 
213 def check(audio: Path, text: Path, language: str, aligner) -> MatchReport:
214     """Score a sample of the audio against the book. Never raises."""
215     report = MatchReport(threshold=aligner.match_threshold)
216 
217     if not settings.match_check:
218         report.skipped = "disabled"
219         return report
220 
221     try:
222         chapters = book_chapters(text)
223         report.text_chapters = len(chapters)
224         if not chapters:
225             report.verdict = "poor"
226             report.summary = "No readable text could be found in the book."
227             return report
228 
229         starts = audio_chapter_starts(audio)
230         report.audio_chapters = len(starts)
231 
232         if len(chapters) == 1 and len(starts) > 1:
233             report.warnings.append(
234                 f"The book is one flat document but the audio has "
235                 f"{len(starts)} chapters. Alignment still works, but it has "
236                 f"only one place to anchor - an epub with real chapters aligns "
237                 f"more reliably."
238             )
239 
240         # Learn what this book scores by chance before judging any match
241         # against it. See backend/chapters.py for why a fixed threshold cannot
242         # work across scripts.
243         fingerprints = [chapters_mod.fingerprint(c) for c in chapters]
244         calibration = chapters_mod.calibrate(chapters)
245         report.accept_at = calibration.accept_at
246         report.noise_floor = calibration.floor
247 
248         # Sample chapters spread through the book, skipping the first: it is
249         # where publisher announcements and credits live, so it is the least
250         # representative chapter there is.
251         picks = _spread(len(starts), MAX_SAMPLES)
252 
253         for idx in picks:
254             samples = read_samples(audio, starts[idx], SAMPLE_SECONDS)
255             if samples is None or len(samples) < 16000:
256                 continue
257             transcript = transcribe_sample(samples, language)
258             if len(transcript.strip()) < 40:
259                 continue
260 
261             match = chapters_mod.best_match(
262                 chapters_mod.fingerprint(transcript), fingerprints, calibration
263             )
264             report.scores.append(
265                 ChapterScore(
266                     audio_chapter=idx,
267                     best_score=match.score,
268                     best_text_chapter=match.text_index,
269                     runner_up=match.runner_up,
270                     confidence=match.confidence,
271                     accepted=match.accepted,
272                     transcript_head=transcript.strip()[:160],
273                 )
274             )
275             # A confident match is enough; only keep sampling when unsure.
276             if match.accepted and match.confidence >= 2.0:
277                 break
278 
279         _verdict(report)
280         return report
281 
282     except Exception as exc:  # noqa: BLE001 - a check must never block an upload
283         log.warning("match check failed: %s", exc, exc_info=True)
284         report.skipped = str(exc)
285         return report
286 
287 
288 def _spread(n: int, k: int) -> list[int]:
289     """Up to k chapter indices spread across the book.
290 
291     Skips chapter 0 when there is anything else to choose. It is where
292     publisher announcements, credits and "read by" cards live - material that
293     is in the audio and not in the book - so it is the least representative
294     chapter there is, and the worst one to judge a whole book on.
295     """
296     if n <= 1:
297         return [0]
298     first = 1 if n > 3 else 0
299     usable = n - first
300     if usable <= k:
301         return list(range(first, n))
302     step = usable / k
303     return sorted({min(n - 1, first + int(i * step)) for i in range(k)})
304 
305 
306 def _verdict(report: MatchReport) -> None:
307     """Turn the measurements into advice.
308 
309     Two separate questions, and the second is the one a fixed threshold cannot
310     answer:
311 
312     * Is the best chapter similar enough to be a real match at all?
313     * Is it *distinctly* the best, or did every chapter score alike?
314 
315     A different book in the same language scores high on the first and fails
316     the second - its chapters all look equally plausible because they share a
317     language, not a story.
318     """
319     if not report.scores:
320         report.verdict = "unknown"
321         report.summary = "Could not sample enough audio to check the match."
322         return
323 
324     best = report.best
325     confidence = report.confidence
326     accepted = report.matched
327 
328     if accepted and confidence >= 2.0:
329         report.verdict = "good"
330         report.summary = (
331             f"Text and audio line up - the matching chapter scores {best:.0f}, "
332             f"{confidence:.1f}x clear of the next best."
333         )
334     elif accepted:
335         report.verdict = "marginal"
336         report.summary = (
337             f"Probable match ({best:.0f}), but only {confidence:.1f}x clear of "
338             f"the next best chapter. Expect some drift."
339         )
340     elif best >= report.accept_at:
341         # Similar enough, but nothing stood out.
342         report.verdict = "poor"
343         report.summary = (
344             f"Every chapter of this book scores about the same ({best:.0f} vs "
345             f"{report.scores[0].runner_up:.0f}), so nothing actually matches."
346         )
347         report.warnings.append(
348             "That pattern means a different book in the same language - the "
349             "words are familiar but the story is not. Check you uploaded the "
350             "right book and the right edition."
351         )
352     else:
353         report.verdict = "poor"
354         report.summary = (
355             f"This text does not look like this audio ({best:.0f}, and this "
356             f"book needs {report.accept_at:.0f} to count as a match)."
357         )