backend/detect.py (5833 bytes)
1 """Work out what the user dropped on us. 2 3 Two questions, both answered without asking the user anything: 4 1. Which file is the audio and which is the book? (by extension) 5 2. What language is the book in? (lingua, over text pulled from the epub) 6 7 The detected language is a default, not a verdict - the UI always lets the user 8 override it, because getting this wrong wastes a long alignment run. 9 """ 10 11 from __future__ import annotations 12 13 import re 14 import zipfile 15 from dataclasses import dataclass 16 from functools import lru_cache 17 from pathlib import Path 18 19 from . import languages 20 from .runner import AUDIO_SUFFIXES, TEXT_SUFFIXES 21 22 # Enough text to be confident without reading a whole book into memory. 23 _SAMPLE_CHARS = 20_000 24 25 26 class DetectionError(ValueError): 27 pass 28 29 30 # Cover art for the rendered video. Optional, and gated behind sign-in. 31 IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".webp", ".bmp"} 32 33 34 @dataclass(frozen=True) 35 class Pairing: 36 # One entry for a single-file audiobook, many for a per-chapter set, in 37 # playback order. 38 audio_names: list[str] 39 text_name: str 40 # A picture to put behind the video, if one was dropped in. 41 cover_name: str | None = None 42 43 @property 44 def is_multipart(self) -> bool: 45 return len(self.audio_names) > 1 46 47 @property 48 def display_name(self) -> str: 49 if not self.is_multipart: 50 return self.audio_names[0] 51 return f"{self.audio_names[0]} + {len(self.audio_names) - 1} more" 52 53 54 def natural_key(name: str) -> tuple: 55 """Sort key that orders 2.mp3 before 10.mp3. 56 57 Chapter order is playback order, and lexical sorting gets it wrong as soon 58 as a set passes nine files. 59 """ 60 parts = re.split(r"(\d+)", Path(name).stem) 61 return tuple(int(p) if p.isdigit() else p.lower() for p in parts) 62 63 64 def classify(filenames: list[str]) -> Pairing: 65 """Split dropped filenames into the audio part(s), the text, and any cover.""" 66 audio = [n for n in filenames if Path(n).suffix.lower() in AUDIO_SUFFIXES] 67 text = [n for n in filenames if Path(n).suffix.lower() in TEXT_SUFFIXES] 68 images = [n for n in filenames if Path(n).suffix.lower() in IMAGE_SUFFIXES] 69 70 if not audio: 71 raise DetectionError( 72 "No audio file found. Add an audiobook " 73 "(m4b, mp3, m4a, opus, flac, wav...)." 74 ) 75 if not text: 76 raise DetectionError( 77 "No book file found. Add an epub (or a txt/srt/vtt/ass script)." 78 ) 79 if len(text) > 1: 80 raise DetectionError(f"Got {len(text)} text files. Drop one book at a time.") 81 82 # Mixed containers usually mean two different rips got dropped together. 83 suffixes = {Path(n).suffix.lower() for n in audio} 84 if len(suffixes) > 1: 85 raise DetectionError( 86 "The audio files are not all the same format (" 87 + ", ".join(sorted(suffixes)) 88 + "). Drop one audiobook at a time." 89 ) 90 91 return Pairing( 92 audio_names=sorted(audio, key=natural_key), 93 text_name=text[0], 94 # Last one wins, so re-dropping an image replaces the previous choice. 95 cover_name=images[-1] if images else None, 96 ) 97 98 99 def extract_text_sample(path: Path) -> str: 100 """Pull readable text out of an epub (or plain text file) for detection.""" 101 suffix = path.suffix.lower() 102 103 if suffix != ".epub": 104 try: 105 return path.read_text(encoding="utf-8", errors="replace")[:_SAMPLE_CHARS] 106 except OSError as exc: 107 raise DetectionError(f"Could not read {path.name}: {exc}") from exc 108 109 # Read the epub as a zip rather than via ebooklib: much faster, and it 110 # tolerates the malformed epubs that converted books often are. 111 try: 112 from bs4 import BeautifulSoup 113 114 chunks: list[str] = [] 115 total = 0 116 with zipfile.ZipFile(path) as zf: 117 names = [ 118 n for n in zf.namelist() 119 if n.lower().endswith((".xhtml", ".html", ".htm")) 120 ] 121 for name in sorted(names): 122 if total >= _SAMPLE_CHARS: 123 break 124 try: 125 raw = zf.read(name).decode("utf-8", errors="replace") 126 except (KeyError, OSError): 127 continue 128 text = BeautifulSoup(raw, "html.parser").get_text(" ", strip=True) 129 if text: 130 chunks.append(text) 131 total += len(text) 132 sample = " ".join(chunks)[:_SAMPLE_CHARS] 133 except zipfile.BadZipFile as exc: 134 raise DetectionError( 135 f"{path.name} is not a readable epub (bad zip archive)." 136 ) from exc 137 138 if not sample.strip(): 139 raise DetectionError( 140 f"No text could be read from {path.name}. " 141 "If it is a scanned/image-only book, it cannot be aligned." 142 ) 143 return sample 144 145 146 @lru_cache(maxsize=1) 147 def _detector(): 148 # Built once and cached: constructing this is the expensive part. 149 from lingua import LanguageDetectorBuilder 150 151 return LanguageDetectorBuilder.from_all_languages().build() 152 153 154 @dataclass(frozen=True) 155 class Detection: 156 code: str | None 157 name: str | None 158 confidence: float 159 supported: bool 160 161 162 def detect_language(sample: str) -> Detection: 163 """Best-guess ISO 639-1 code for a block of text.""" 164 if not sample.strip(): 165 return Detection(None, None, 0.0, False) 166 167 values = _detector().compute_language_confidence_values(sample) 168 if not values: 169 return Detection(None, None, 0.0, False) 170 171 best = values[0] 172 code = best.language.iso_code_639_1.name.lower() 173 known = languages.get(code) 174 return Detection( 175 code=code, 176 name=known.name if known else best.language.name.title(), 177 confidence=float(best.value), 178 supported=known is not None, 179 )