backend/matching.py (13028 bytes)
1 """Does this text actually match this audio? 2 3 subplz fails late and unhelpfully when it does not: it transcribes the whole 4 book first, then reports "the generated transcript and the provided text file 5 are too different" and writes a .subfail. On CPU that is a wasted hour. 6 7 So we ask the same question up front, cheaply: transcribe a couple of short 8 samples from the start of a few chapters, and score them against the book with 9 the backend's own rule. Same arithmetic, same threshold, ~10 seconds instead of 10 an hour. 11 12 What the number means, and what it does not: a good score says the text lines 13 up with what is being read. A bad score usually means the wrong book, the wrong 14 edition (abridged vs full), an audiobook that opens with a publisher 15 announcement the book does not contain, or a book buried under front matter. 16 """ 17 18 from __future__ import annotations 19 20 import json 21 import logging 22 import subprocess 23 import zipfile 24 from dataclasses import dataclass, field 25 from functools import lru_cache 26 from pathlib import Path 27 28 from . import chapters as chapters_mod 29 from .settings import settings 30 31 log = logging.getLogger(__name__) 32 33 # Enough speech to fingerprint a chapter confidently. Sampling is adaptive, so 34 # a book that matches well normally pays for one of these and stops. 35 SAMPLE_SECONDS = 120 36 MAX_SAMPLES = 3 37 38 39 @dataclass 40 class ChapterScore: 41 audio_chapter: int 42 best_score: float 43 best_text_chapter: int | None 44 runner_up: float = 0.0 45 # How far clear of the runner-up. 1.0 means every chapter looked alike, 46 # which is the signature of the wrong book rather than a poor recording. 47 confidence: float = 0.0 48 accepted: bool = False 49 transcript_head: str = "" 50 51 @property 52 def matched(self) -> bool: 53 return self.accepted 54 55 56 @dataclass 57 class MatchReport: 58 threshold: float 59 scores: list[ChapterScore] = field(default_factory=list) 60 text_chapters: int = 0 61 audio_chapters: int = 0 62 accept_at: float = 0.0 63 noise_floor: float = 0.0 64 verdict: str = "unknown" # good | marginal | poor | unknown 65 summary: str = "" 66 warnings: list[str] = field(default_factory=list) 67 skipped: str | None = None 68 69 @property 70 def best(self) -> float: 71 return max((s.best_score for s in self.scores), default=0.0) 72 73 @property 74 def worst(self) -> float: 75 return min((s.best_score for s in self.scores), default=0.0) 76 77 @property 78 def matched(self) -> int: 79 return sum(1 for s in self.scores if s.matched) 80 81 @property 82 def confidence(self) -> float: 83 return max((s.confidence for s in self.scores), default=0.0) 84 85 def as_dict(self) -> dict: 86 return { 87 "threshold": round(self.accept_at, 1), 88 "noise_floor": round(self.noise_floor, 1), 89 "confidence": round(self.confidence, 2), 90 "verdict": self.verdict, 91 "summary": self.summary, 92 "best": round(self.best, 1), 93 "worst": round(self.worst, 1), 94 "sampled": len(self.scores), 95 "matched": self.matched, 96 "text_chapters": self.text_chapters, 97 "audio_chapters": self.audio_chapters, 98 "warnings": self.warnings, 99 "skipped": self.skipped, 100 "scores": [ 101 { 102 "audio_chapter": s.audio_chapter, 103 "score": round(s.best_score, 1), 104 "runner_up": round(s.runner_up, 1), 105 "confidence": round(s.confidence, 2), 106 "text_chapter": s.best_text_chapter, 107 } 108 for s in self.scores 109 ], 110 } 111 112 113 # --------------------------------------------------------------------------- 114 # book text 115 # --------------------------------------------------------------------------- 116 117 def book_chapters(text_path: Path) -> list[str]: 118 """The comparable units, split the way the backend will split them. 119 120 This matters more than it looks. subplz gives an epub one text chapter per 121 spine document, but a .txt exactly one chapter for the whole file - so a 122 flat text file offers the matcher a single anchor no matter how many 123 chapters the audio has. 124 """ 125 if text_path.suffix.lower() != ".epub": 126 return [text_path.read_text(encoding="utf-8", errors="replace")] 127 128 try: 129 from bs4 import BeautifulSoup 130 131 chapters: list[str] = [] 132 with zipfile.ZipFile(text_path) as zf: 133 names = [ 134 n for n in zf.namelist() 135 if n.lower().endswith((".xhtml", ".html", ".htm")) 136 ] 137 for name in sorted(names): 138 try: 139 raw = zf.read(name).decode("utf-8", errors="replace") 140 except (KeyError, OSError): 141 continue 142 soup = BeautifulSoup(raw, "html.parser") 143 for bad in soup(["script", "style"]): 144 bad.decompose() 145 # subplz joins paragraph texts with no separator, so match that. 146 text = "".join( 147 p.get_text(" ", strip=True) for p in soup.find_all("p") 148 ) or soup.get_text(" ", strip=True) 149 if text.strip(): 150 chapters.append(text) 151 return chapters 152 except zipfile.BadZipFile: 153 return [] 154 155 156 # --------------------------------------------------------------------------- 157 # audio sampling 158 # --------------------------------------------------------------------------- 159 160 def audio_chapter_starts(audio: Path) -> list[float]: 161 """Start time of each chapter, or [0.0] for a flat file.""" 162 try: 163 out = subprocess.run( 164 ["ffprobe", "-v", "error", "-show_chapters", "-print_format", "json", 165 str(audio)], 166 capture_output=True, timeout=120, check=True, 167 ) 168 chapters = json.loads(out.stdout.decode("utf-8")).get("chapters", []) 169 starts = [float(c["start_time"]) for c in chapters] 170 return starts or [0.0] 171 except (subprocess.SubprocessError, ValueError, OSError, KeyError): 172 return [0.0] 173 174 175 def read_samples(audio: Path, start: float, seconds: int): 176 """`seconds` of audio from `start`, as the float32 mono 16 kHz the model wants.""" 177 import numpy as np 178 179 proc = subprocess.run( 180 [ 181 "ffmpeg", "-v", "error", 182 "-max_error_rate", "1.0", "-err_detect", "ignore_err", 183 "-ss", f"{start:.3f}", "-t", str(seconds), "-i", str(audio), 184 "-map", "0:a:0", "-f", "s16le", "-acodec", "pcm_s16le", 185 "-ac", "1", "-ar", "16000", "-", 186 ], 187 capture_output=True, timeout=300, 188 ) 189 if proc.returncode != 0 or not proc.stdout: 190 return None 191 return np.frombuffer(proc.stdout, np.int16).astype(np.float32) / 32768.0 192 193 194 @lru_cache(maxsize=1) 195 def _model(): 196 """The same tiny model the alignment itself uses, loaded once.""" 197 from faster_whisper import WhisperModel 198 199 return WhisperModel(settings.model, device="cpu", compute_type="int8") 200 201 202 def transcribe_sample(samples, language: str) -> str: 203 segments, _ = _model().transcribe( 204 samples, language=language, beam_size=5, without_timestamps=True 205 ) 206 return "".join(seg.text for seg in segments) 207 208 209 # --------------------------------------------------------------------------- 210 # the check 211 # --------------------------------------------------------------------------- 212 213 def check(audio: Path, text: Path, language: str, aligner) -> MatchReport: 214 """Score a sample of the audio against the book. Never raises.""" 215 report = MatchReport(threshold=aligner.match_threshold) 216 217 if not settings.match_check: 218 report.skipped = "disabled" 219 return report 220 221 try: 222 chapters = book_chapters(text) 223 report.text_chapters = len(chapters) 224 if not chapters: 225 report.verdict = "poor" 226 report.summary = "No readable text could be found in the book." 227 return report 228 229 starts = audio_chapter_starts(audio) 230 report.audio_chapters = len(starts) 231 232 if len(chapters) == 1 and len(starts) > 1: 233 report.warnings.append( 234 f"The book is one flat document but the audio has " 235 f"{len(starts)} chapters. Alignment still works, but it has " 236 f"only one place to anchor - an epub with real chapters aligns " 237 f"more reliably." 238 ) 239 240 # Learn what this book scores by chance before judging any match 241 # against it. See backend/chapters.py for why a fixed threshold cannot 242 # work across scripts. 243 fingerprints = [chapters_mod.fingerprint(c) for c in chapters] 244 calibration = chapters_mod.calibrate(chapters) 245 report.accept_at = calibration.accept_at 246 report.noise_floor = calibration.floor 247 248 # Sample chapters spread through the book, skipping the first: it is 249 # where publisher announcements and credits live, so it is the least 250 # representative chapter there is. 251 picks = _spread(len(starts), MAX_SAMPLES) 252 253 for idx in picks: 254 samples = read_samples(audio, starts[idx], SAMPLE_SECONDS) 255 if samples is None or len(samples) < 16000: 256 continue 257 transcript = transcribe_sample(samples, language) 258 if len(transcript.strip()) < 40: 259 continue 260 261 match = chapters_mod.best_match( 262 chapters_mod.fingerprint(transcript), fingerprints, calibration 263 ) 264 report.scores.append( 265 ChapterScore( 266 audio_chapter=idx, 267 best_score=match.score, 268 best_text_chapter=match.text_index, 269 runner_up=match.runner_up, 270 confidence=match.confidence, 271 accepted=match.accepted, 272 transcript_head=transcript.strip()[:160], 273 ) 274 ) 275 # A confident match is enough; only keep sampling when unsure. 276 if match.accepted and match.confidence >= 2.0: 277 break 278 279 _verdict(report) 280 return report 281 282 except Exception as exc: # noqa: BLE001 - a check must never block an upload 283 log.warning("match check failed: %s", exc, exc_info=True) 284 report.skipped = str(exc) 285 return report 286 287 288 def _spread(n: int, k: int) -> list[int]: 289 """Up to k chapter indices spread across the book. 290 291 Skips chapter 0 when there is anything else to choose. It is where 292 publisher announcements, credits and "read by" cards live - material that 293 is in the audio and not in the book - so it is the least representative 294 chapter there is, and the worst one to judge a whole book on. 295 """ 296 if n <= 1: 297 return [0] 298 first = 1 if n > 3 else 0 299 usable = n - first 300 if usable <= k: 301 return list(range(first, n)) 302 step = usable / k 303 return sorted({min(n - 1, first + int(i * step)) for i in range(k)}) 304 305 306 def _verdict(report: MatchReport) -> None: 307 """Turn the measurements into advice. 308 309 Two separate questions, and the second is the one a fixed threshold cannot 310 answer: 311 312 * Is the best chapter similar enough to be a real match at all? 313 * Is it *distinctly* the best, or did every chapter score alike? 314 315 A different book in the same language scores high on the first and fails 316 the second - its chapters all look equally plausible because they share a 317 language, not a story. 318 """ 319 if not report.scores: 320 report.verdict = "unknown" 321 report.summary = "Could not sample enough audio to check the match." 322 return 323 324 best = report.best 325 confidence = report.confidence 326 accepted = report.matched 327 328 if accepted and confidence >= 2.0: 329 report.verdict = "good" 330 report.summary = ( 331 f"Text and audio line up - the matching chapter scores {best:.0f}, " 332 f"{confidence:.1f}x clear of the next best." 333 ) 334 elif accepted: 335 report.verdict = "marginal" 336 report.summary = ( 337 f"Probable match ({best:.0f}), but only {confidence:.1f}x clear of " 338 f"the next best chapter. Expect some drift." 339 ) 340 elif best >= report.accept_at: 341 # Similar enough, but nothing stood out. 342 report.verdict = "poor" 343 report.summary = ( 344 f"Every chapter of this book scores about the same ({best:.0f} vs " 345 f"{report.scores[0].runner_up:.0f}), so nothing actually matches." 346 ) 347 report.warnings.append( 348 "That pattern means a different book in the same language - the " 349 "words are familiar but the story is not. Check you uploaded the " 350 "right book and the right edition." 351 ) 352 else: 353 report.verdict = "poor" 354 report.summary = ( 355 f"This text does not look like this audio ({best:.0f}, and this " 356 f"book needs {report.accept_at:.0f} to count as a match)." 357 )