backend/convert.py (9126 bytes)
1 """Accept fb2, mobi and azw3 by turning them into a chaptered epub. 2 3 Chapters are the whole point, and it took a measurement to see why. subplz 4 matches audio to text by scoring the *opening* of each audio chapter against 5 the opening of each text chapter, and it splits an epub into one text chapter 6 per spine document but a .txt into exactly one chapter for the entire file. 7 8 Measured on Moskva-Petushki (Russian, 4h40m): 9 10 same book, per chapter 69.1 11 same book, one flat file 38.6 <- subplz's threshold is 40 12 13 So converting to plain text would push a perfectly good book *below* the 14 threshold and produce the "transcript and text are too different" failure. The 15 structure has to survive the conversion. 16 17 Nothing here parses an ebook format or writes a container by hand: 18 19 azw3 / KF8 -> the `mobi` package unpacks it straight to epub 20 mobi (older) -> `mobi` unpacks to HTML; ebooklib rebuilds the epub 21 fb2, fb2.zip -> it is XML; lxml reads it, ebooklib writes the epub 22 23 `mobi` is the only added dependency; lxml, BeautifulSoup and ebooklib already 24 ship with subplz. 25 """ 26 27 from __future__ import annotations 28 29 import html 30 import shutil 31 import zipfile 32 from pathlib import Path 33 34 # Accepted and converted on upload. Keep in sync with runner.TEXT_SUFFIXES 35 # and the frontend accept list. 36 CONVERTIBLE_SUFFIXES = {".fb2", ".mobi", ".azw", ".azw3", ".prc"} 37 38 _MOBI_SUFFIXES = {".mobi", ".azw", ".azw3", ".prc"} 39 40 # fb2 bodies named this way hold footnotes, not the story. 41 _FB2_SKIP_BODIES = {"notes", "comments"} 42 43 # Text blocks worth keeping: prose, verse lines, subheadings. 44 _FB2_BLOCKS = {"p", "v", "subtitle"} 45 46 47 class ConversionError(ValueError): 48 pass 49 50 51 def needs_conversion(name: str) -> bool: 52 lowered = name.lower() 53 return lowered.endswith(".fb2.zip") or Path(lowered).suffix in CONVERTIBLE_SUFFIXES 54 55 56 def to_readable(src: Path, out_stem: Path) -> Path: 57 """Convert `src` into an epub subplz can align against.""" 58 name = src.name.lower() 59 60 if name.endswith(".epub"): 61 return src 62 63 if name.endswith(".fb2.zip"): 64 return _fb2_to_epub(_unzip_fb2(src), src, out_stem) 65 66 if name.endswith(".fb2"): 67 return _fb2_to_epub(src.read_bytes(), src, out_stem) 68 69 if Path(name).suffix in _MOBI_SUFFIXES: 70 return _from_mobi(src, out_stem) 71 72 raise ConversionError(f"Cannot read {src.name}.") 73 74 75 # --------------------------------------------------------------------------- 76 # fb2 77 # --------------------------------------------------------------------------- 78 79 def _unzip_fb2(src: Path) -> bytes: 80 try: 81 with zipfile.ZipFile(src) as zf: 82 inner = next((n for n in zf.namelist() if n.lower().endswith(".fb2")), None) 83 if inner is None: 84 raise ConversionError(f"{src.name} contains no .fb2 file.") 85 return zf.read(inner) 86 except zipfile.BadZipFile as exc: 87 raise ConversionError(f"{src.name} is not a readable zip archive.") from exc 88 89 90 def _fb2_to_epub(raw: bytes, src: Path, out_stem: Path) -> Path: 91 from lxml import etree 92 93 # recover=True: fb2 in the wild is frequently not well-formed. 94 root = etree.fromstring(raw, etree.XMLParser(recover=True, huge_tree=True)) 95 if root is None: 96 raise ConversionError(f"{src.name} could not be parsed as fb2.") 97 98 def localname(el) -> str | None: 99 # Comments and processing instructions have a callable .tag, which 100 # QName rejects. Real fb2 files contain both. 101 return etree.QName(el).localname if isinstance(el.tag, str) else None 102 103 def text_of(el) -> str: 104 return " ".join(t.strip() for t in el.itertext() if t and t.strip()) 105 106 title = _first_text(root, localname, "book-title") or src.stem 107 language = _first_text(root, localname, "lang") or "en" 108 109 chapters: list[tuple[str, list[str]]] = [] 110 for body in root.iter(): 111 if localname(body) != "body" or body.get("name") in _FB2_SKIP_BODIES: 112 continue 113 114 sections = [el for el in body if localname(el) == "section"] 115 targets = sections or [body] 116 for i, section in enumerate(targets, start=1): 117 heading = None 118 paragraphs: list[str] = [] 119 last = None 120 for el in section.iter(): 121 tag = localname(el) 122 if tag == "title" and heading is None: 123 heading = text_of(el) or None 124 continue 125 if tag not in _FB2_BLOCKS: 126 continue 127 t = text_of(el) 128 # fb2 nests <p> inside <title>, so a heading is reached twice. 129 if t and t != last: 130 paragraphs.append(t) 131 last = t 132 if paragraphs: 133 chapters.append((heading or f"Section {i}", paragraphs)) 134 135 if not chapters: 136 raise ConversionError(f"No text could be extracted from {src.name}.") 137 138 return write_epub(out_stem.with_suffix(".epub"), title, language, chapters) 139 140 141 def _first_text(root, localname, tag: str) -> str | None: 142 for el in root.iter(): 143 if localname(el) == tag and el.text and el.text.strip(): 144 return el.text.strip() 145 return None 146 147 148 # --------------------------------------------------------------------------- 149 # mobi / azw3 150 # --------------------------------------------------------------------------- 151 152 def _from_mobi(src: Path, out_stem: Path) -> Path: 153 try: 154 import mobi 155 except ImportError as exc: # pragma: no cover - dependency is declared 156 raise ConversionError( 157 "mobi/azw3 support needs the `mobi` package: pip install mobi" 158 ) from exc 159 160 tempdir = None 161 try: 162 # KF8 (azw3) unpacks straight to epub, chapters and all. 163 tempdir, produced = mobi.extract(str(src)) 164 except Exception as exc: # noqa: BLE001 - the library raises bare exceptions 165 if tempdir: 166 shutil.rmtree(tempdir, ignore_errors=True) 167 raise ConversionError( 168 f"Could not read {src.name}. If it is DRM-protected, it cannot be " 169 "converted." 170 ) from exc 171 172 try: 173 out = Path(produced) 174 if not out.exists(): 175 raise ConversionError(f"Nothing could be unpacked from {src.name}.") 176 177 if out.suffix.lower() == ".epub": 178 dest = out_stem.with_suffix(".epub") 179 dest.parent.mkdir(parents=True, exist_ok=True) 180 shutil.copy2(out, dest) 181 return dest 182 183 pages = [out] if out.suffix.lower() in {".html", ".xhtml", ".htm"} else [] 184 pages += sorted( 185 p for p in Path(tempdir).rglob("*") 186 if p.suffix.lower() in {".html", ".xhtml", ".htm"} and p != out 187 ) 188 if not pages: 189 raise ConversionError(f"No readable text found inside {src.name}.") 190 191 chapters = [] 192 for i, page in enumerate(pages, start=1): 193 paragraphs = _html_paragraphs(page) 194 if paragraphs: 195 chapters.append((page.stem or f"Section {i}", paragraphs)) 196 if not chapters: 197 raise ConversionError(f"No text could be extracted from {src.name}.") 198 199 return write_epub( 200 out_stem.with_suffix(".epub"), src.stem, "en", chapters 201 ) 202 finally: 203 if tempdir: 204 shutil.rmtree(tempdir, ignore_errors=True) 205 206 207 def _html_paragraphs(path: Path) -> list[str]: 208 from bs4 import BeautifulSoup 209 210 soup = BeautifulSoup( 211 path.read_text(encoding="utf-8", errors="replace"), "html.parser" 212 ) 213 for bad in soup(["script", "style"]): 214 bad.decompose() 215 216 blocks = [ 217 el.get_text(" ", strip=True) 218 for el in soup.find_all(["p", "h1", "h2", "h3", "h4", "blockquote"]) 219 ] 220 blocks = [b for b in blocks if b] 221 if blocks: 222 return blocks 223 # Some MOBI files are one long <div> soup with no paragraph tags. 224 return [ln.strip() for ln in soup.get_text("\n").splitlines() if ln.strip()] 225 226 227 # --------------------------------------------------------------------------- 228 # epub output 229 # --------------------------------------------------------------------------- 230 231 def write_epub( 232 dest: Path, 233 title: str, 234 language: str, 235 chapters: list[tuple[str, list[str]]], 236 ) -> Path: 237 """Write a chaptered epub with ebooklib - one spine document per chapter.""" 238 from ebooklib import epub 239 240 book = epub.EpubBook() 241 book.set_identifier(f"subplz-{abs(hash(title)) % 10**12}") 242 book.set_title(title) 243 book.set_language((language or "en")[:8]) 244 245 items = [] 246 for i, (heading, paragraphs) in enumerate(chapters, start=1): 247 doc = epub.EpubHtml( 248 title=heading, file_name=f"chapter{i:04d}.xhtml", lang=language 249 ) 250 body = "\n".join(f"<p>{html.escape(p)}</p>" for p in paragraphs) 251 doc.content = f"<h1>{html.escape(heading)}</h1>\n{body}" 252 book.add_item(doc) 253 items.append(doc) 254 255 book.toc = tuple(items) 256 book.spine = ["nav", *items] 257 book.add_item(epub.EpubNcx()) 258 book.add_item(epub.EpubNav()) 259 260 dest.parent.mkdir(parents=True, exist_ok=True) 261 epub.write_epub(str(dest), book) 262 return dest