Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


backend/convert.py (9126 bytes)

1 """Accept fb2, mobi and azw3 by turning them into a chaptered epub.
2 
3 Chapters are the whole point, and it took a measurement to see why. subplz
4 matches audio to text by scoring the *opening* of each audio chapter against
5 the opening of each text chapter, and it splits an epub into one text chapter
6 per spine document but a .txt into exactly one chapter for the entire file.
7 
8 Measured on Moskva-Petushki (Russian, 4h40m):
9 
10     same book, per chapter   69.1
11     same book, one flat file 38.6   <- subplz's threshold is 40
12 
13 So converting to plain text would push a perfectly good book *below* the
14 threshold and produce the "transcript and text are too different" failure. The
15 structure has to survive the conversion.
16 
17 Nothing here parses an ebook format or writes a container by hand:
18 
19   azw3 / KF8    -> the `mobi` package unpacks it straight to epub
20   mobi (older)  -> `mobi` unpacks to HTML; ebooklib rebuilds the epub
21   fb2, fb2.zip  -> it is XML; lxml reads it, ebooklib writes the epub
22 
23 `mobi` is the only added dependency; lxml, BeautifulSoup and ebooklib already
24 ship with subplz.
25 """
26 
27 from __future__ import annotations
28 
29 import html
30 import shutil
31 import zipfile
32 from pathlib import Path
33 
34 # Accepted and converted on upload. Keep in sync with runner.TEXT_SUFFIXES
35 # and the frontend accept list.
36 CONVERTIBLE_SUFFIXES = {".fb2", ".mobi", ".azw", ".azw3", ".prc"}
37 
38 _MOBI_SUFFIXES = {".mobi", ".azw", ".azw3", ".prc"}
39 
40 # fb2 bodies named this way hold footnotes, not the story.
41 _FB2_SKIP_BODIES = {"notes", "comments"}
42 
43 # Text blocks worth keeping: prose, verse lines, subheadings.
44 _FB2_BLOCKS = {"p", "v", "subtitle"}
45 
46 
47 class ConversionError(ValueError):
48     pass
49 
50 
51 def needs_conversion(name: str) -> bool:
52     lowered = name.lower()
53     return lowered.endswith(".fb2.zip") or Path(lowered).suffix in CONVERTIBLE_SUFFIXES
54 
55 
56 def to_readable(src: Path, out_stem: Path) -> Path:
57     """Convert `src` into an epub subplz can align against."""
58     name = src.name.lower()
59 
60     if name.endswith(".epub"):
61         return src
62 
63     if name.endswith(".fb2.zip"):
64         return _fb2_to_epub(_unzip_fb2(src), src, out_stem)
65 
66     if name.endswith(".fb2"):
67         return _fb2_to_epub(src.read_bytes(), src, out_stem)
68 
69     if Path(name).suffix in _MOBI_SUFFIXES:
70         return _from_mobi(src, out_stem)
71 
72     raise ConversionError(f"Cannot read {src.name}.")
73 
74 
75 # ---------------------------------------------------------------------------
76 # fb2
77 # ---------------------------------------------------------------------------
78 
79 def _unzip_fb2(src: Path) -> bytes:
80     try:
81         with zipfile.ZipFile(src) as zf:
82             inner = next((n for n in zf.namelist() if n.lower().endswith(".fb2")), None)
83             if inner is None:
84                 raise ConversionError(f"{src.name} contains no .fb2 file.")
85             return zf.read(inner)
86     except zipfile.BadZipFile as exc:
87         raise ConversionError(f"{src.name} is not a readable zip archive.") from exc
88 
89 
90 def _fb2_to_epub(raw: bytes, src: Path, out_stem: Path) -> Path:
91     from lxml import etree
92 
93     # recover=True: fb2 in the wild is frequently not well-formed.
94     root = etree.fromstring(raw, etree.XMLParser(recover=True, huge_tree=True))
95     if root is None:
96         raise ConversionError(f"{src.name} could not be parsed as fb2.")
97 
98     def localname(el) -> str | None:
99         # Comments and processing instructions have a callable .tag, which
100         # QName rejects. Real fb2 files contain both.
101         return etree.QName(el).localname if isinstance(el.tag, str) else None
102 
103     def text_of(el) -> str:
104         return " ".join(t.strip() for t in el.itertext() if t and t.strip())
105 
106     title = _first_text(root, localname, "book-title") or src.stem
107     language = _first_text(root, localname, "lang") or "en"
108 
109     chapters: list[tuple[str, list[str]]] = []
110     for body in root.iter():
111         if localname(body) != "body" or body.get("name") in _FB2_SKIP_BODIES:
112             continue
113 
114         sections = [el for el in body if localname(el) == "section"]
115         targets = sections or [body]
116         for i, section in enumerate(targets, start=1):
117             heading = None
118             paragraphs: list[str] = []
119             last = None
120             for el in section.iter():
121                 tag = localname(el)
122                 if tag == "title" and heading is None:
123                     heading = text_of(el) or None
124                     continue
125                 if tag not in _FB2_BLOCKS:
126                     continue
127                 t = text_of(el)
128                 # fb2 nests <p> inside <title>, so a heading is reached twice.
129                 if t and t != last:
130                     paragraphs.append(t)
131                     last = t
132             if paragraphs:
133                 chapters.append((heading or f"Section {i}", paragraphs))
134 
135     if not chapters:
136         raise ConversionError(f"No text could be extracted from {src.name}.")
137 
138     return write_epub(out_stem.with_suffix(".epub"), title, language, chapters)
139 
140 
141 def _first_text(root, localname, tag: str) -> str | None:
142     for el in root.iter():
143         if localname(el) == tag and el.text and el.text.strip():
144             return el.text.strip()
145     return None
146 
147 
148 # ---------------------------------------------------------------------------
149 # mobi / azw3
150 # ---------------------------------------------------------------------------
151 
152 def _from_mobi(src: Path, out_stem: Path) -> Path:
153     try:
154         import mobi
155     except ImportError as exc:  # pragma: no cover - dependency is declared
156         raise ConversionError(
157             "mobi/azw3 support needs the `mobi` package: pip install mobi"
158         ) from exc
159 
160     tempdir = None
161     try:
162         # KF8 (azw3) unpacks straight to epub, chapters and all.
163         tempdir, produced = mobi.extract(str(src))
164     except Exception as exc:  # noqa: BLE001 - the library raises bare exceptions
165         if tempdir:
166             shutil.rmtree(tempdir, ignore_errors=True)
167         raise ConversionError(
168             f"Could not read {src.name}. If it is DRM-protected, it cannot be "
169             "converted."
170         ) from exc
171 
172     try:
173         out = Path(produced)
174         if not out.exists():
175             raise ConversionError(f"Nothing could be unpacked from {src.name}.")
176 
177         if out.suffix.lower() == ".epub":
178             dest = out_stem.with_suffix(".epub")
179             dest.parent.mkdir(parents=True, exist_ok=True)
180             shutil.copy2(out, dest)
181             return dest
182 
183         pages = [out] if out.suffix.lower() in {".html", ".xhtml", ".htm"} else []
184         pages += sorted(
185             p for p in Path(tempdir).rglob("*")
186             if p.suffix.lower() in {".html", ".xhtml", ".htm"} and p != out
187         )
188         if not pages:
189             raise ConversionError(f"No readable text found inside {src.name}.")
190 
191         chapters = []
192         for i, page in enumerate(pages, start=1):
193             paragraphs = _html_paragraphs(page)
194             if paragraphs:
195                 chapters.append((page.stem or f"Section {i}", paragraphs))
196         if not chapters:
197             raise ConversionError(f"No text could be extracted from {src.name}.")
198 
199         return write_epub(
200             out_stem.with_suffix(".epub"), src.stem, "en", chapters
201         )
202     finally:
203         if tempdir:
204             shutil.rmtree(tempdir, ignore_errors=True)
205 
206 
207 def _html_paragraphs(path: Path) -> list[str]:
208     from bs4 import BeautifulSoup
209 
210     soup = BeautifulSoup(
211         path.read_text(encoding="utf-8", errors="replace"), "html.parser"
212     )
213     for bad in soup(["script", "style"]):
214         bad.decompose()
215 
216     blocks = [
217         el.get_text(" ", strip=True)
218         for el in soup.find_all(["p", "h1", "h2", "h3", "h4", "blockquote"])
219     ]
220     blocks = [b for b in blocks if b]
221     if blocks:
222         return blocks
223     # Some MOBI files are one long <div> soup with no paragraph tags.
224     return [ln.strip() for ln in soup.get_text("\n").splitlines() if ln.strip()]
225 
226 
227 # ---------------------------------------------------------------------------
228 # epub output
229 # ---------------------------------------------------------------------------
230 
231 def write_epub(
232     dest: Path,
233     title: str,
234     language: str,
235     chapters: list[tuple[str, list[str]]],
236 ) -> Path:
237     """Write a chaptered epub with ebooklib - one spine document per chapter."""
238     from ebooklib import epub
239 
240     book = epub.EpubBook()
241     book.set_identifier(f"subplz-{abs(hash(title)) % 10**12}")
242     book.set_title(title)
243     book.set_language((language or "en")[:8])
244 
245     items = []
246     for i, (heading, paragraphs) in enumerate(chapters, start=1):
247         doc = epub.EpubHtml(
248             title=heading, file_name=f"chapter{i:04d}.xhtml", lang=language
249         )
250         body = "\n".join(f"<p>{html.escape(p)}</p>" for p in paragraphs)
251         doc.content = f"<h1>{html.escape(heading)}</h1>\n{body}"
252         book.add_item(doc)
253         items.append(doc)
254 
255     book.toc = tuple(items)
256     book.spine = ["nav", *items]
257     book.add_item(epub.EpubNcx())
258     book.add_item(epub.EpubNav())
259 
260     dest.parent.mkdir(parents=True, exist_ok=True)
261     epub.write_epub(str(dest), book)
262     return dest