Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


backend/detect.py (5833 bytes)

1 """Work out what the user dropped on us.
2 
3 Two questions, both answered without asking the user anything:
4   1. Which file is the audio and which is the book? (by extension)
5   2. What language is the book in? (lingua, over text pulled from the epub)
6 
7 The detected language is a default, not a verdict - the UI always lets the user
8 override it, because getting this wrong wastes a long alignment run.
9 """
10 
11 from __future__ import annotations
12 
13 import re
14 import zipfile
15 from dataclasses import dataclass
16 from functools import lru_cache
17 from pathlib import Path
18 
19 from . import languages
20 from .runner import AUDIO_SUFFIXES, TEXT_SUFFIXES
21 
22 # Enough text to be confident without reading a whole book into memory.
23 _SAMPLE_CHARS = 20_000
24 
25 
26 class DetectionError(ValueError):
27     pass
28 
29 
30 # Cover art for the rendered video. Optional, and gated behind sign-in.
31 IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png", ".webp", ".bmp"}
32 
33 
34 @dataclass(frozen=True)
35 class Pairing:
36     # One entry for a single-file audiobook, many for a per-chapter set, in
37     # playback order.
38     audio_names: list[str]
39     text_name: str
40     # A picture to put behind the video, if one was dropped in.
41     cover_name: str | None = None
42 
43     @property
44     def is_multipart(self) -> bool:
45         return len(self.audio_names) > 1
46 
47     @property
48     def display_name(self) -> str:
49         if not self.is_multipart:
50             return self.audio_names[0]
51         return f"{self.audio_names[0]} + {len(self.audio_names) - 1} more"
52 
53 
54 def natural_key(name: str) -> tuple:
55     """Sort key that orders 2.mp3 before 10.mp3.
56 
57     Chapter order is playback order, and lexical sorting gets it wrong as soon
58     as a set passes nine files.
59     """
60     parts = re.split(r"(\d+)", Path(name).stem)
61     return tuple(int(p) if p.isdigit() else p.lower() for p in parts)
62 
63 
64 def classify(filenames: list[str]) -> Pairing:
65     """Split dropped filenames into the audio part(s), the text, and any cover."""
66     audio = [n for n in filenames if Path(n).suffix.lower() in AUDIO_SUFFIXES]
67     text = [n for n in filenames if Path(n).suffix.lower() in TEXT_SUFFIXES]
68     images = [n for n in filenames if Path(n).suffix.lower() in IMAGE_SUFFIXES]
69 
70     if not audio:
71         raise DetectionError(
72             "No audio file found. Add an audiobook "
73             "(m4b, mp3, m4a, opus, flac, wav...)."
74         )
75     if not text:
76         raise DetectionError(
77             "No book file found. Add an epub (or a txt/srt/vtt/ass script)."
78         )
79     if len(text) > 1:
80         raise DetectionError(f"Got {len(text)} text files. Drop one book at a time.")
81 
82     # Mixed containers usually mean two different rips got dropped together.
83     suffixes = {Path(n).suffix.lower() for n in audio}
84     if len(suffixes) > 1:
85         raise DetectionError(
86             "The audio files are not all the same format ("
87             + ", ".join(sorted(suffixes))
88             + "). Drop one audiobook at a time."
89         )
90 
91     return Pairing(
92         audio_names=sorted(audio, key=natural_key),
93         text_name=text[0],
94         # Last one wins, so re-dropping an image replaces the previous choice.
95         cover_name=images[-1] if images else None,
96     )
97 
98 
99 def extract_text_sample(path: Path) -> str:
100     """Pull readable text out of an epub (or plain text file) for detection."""
101     suffix = path.suffix.lower()
102 
103     if suffix != ".epub":
104         try:
105             return path.read_text(encoding="utf-8", errors="replace")[:_SAMPLE_CHARS]
106         except OSError as exc:
107             raise DetectionError(f"Could not read {path.name}: {exc}") from exc
108 
109     # Read the epub as a zip rather than via ebooklib: much faster, and it
110     # tolerates the malformed epubs that converted books often are.
111     try:
112         from bs4 import BeautifulSoup
113 
114         chunks: list[str] = []
115         total = 0
116         with zipfile.ZipFile(path) as zf:
117             names = [
118                 n for n in zf.namelist()
119                 if n.lower().endswith((".xhtml", ".html", ".htm"))
120             ]
121             for name in sorted(names):
122                 if total >= _SAMPLE_CHARS:
123                     break
124                 try:
125                     raw = zf.read(name).decode("utf-8", errors="replace")
126                 except (KeyError, OSError):
127                     continue
128                 text = BeautifulSoup(raw, "html.parser").get_text(" ", strip=True)
129                 if text:
130                     chunks.append(text)
131                     total += len(text)
132         sample = " ".join(chunks)[:_SAMPLE_CHARS]
133     except zipfile.BadZipFile as exc:
134         raise DetectionError(
135             f"{path.name} is not a readable epub (bad zip archive)."
136         ) from exc
137 
138     if not sample.strip():
139         raise DetectionError(
140             f"No text could be read from {path.name}. "
141             "If it is a scanned/image-only book, it cannot be aligned."
142         )
143     return sample
144 
145 
146 @lru_cache(maxsize=1)
147 def _detector():
148     # Built once and cached: constructing this is the expensive part.
149     from lingua import LanguageDetectorBuilder
150 
151     return LanguageDetectorBuilder.from_all_languages().build()
152 
153 
154 @dataclass(frozen=True)
155 class Detection:
156     code: str | None
157     name: str | None
158     confidence: float
159     supported: bool
160 
161 
162 def detect_language(sample: str) -> Detection:
163     """Best-guess ISO 639-1 code for a block of text."""
164     if not sample.strip():
165         return Detection(None, None, 0.0, False)
166 
167     values = _detector().compute_language_confidence_values(sample)
168     if not values:
169         return Detection(None, None, 0.0, False)
170 
171     best = values[0]
172     code = best.language.iso_code_639_1.name.lower()
173     known = languages.get(code)
174     return Detection(
175         code=code,
176         name=known.name if known else best.language.name.title(),
177         confidence=float(best.value),
178         supported=known is not None,
179     )