backend/languages.py (1836 bytes)
1 """Language registry. 2 3 subplz defaults to Japanese and splits sentences with pysbd, which only knows 23 4 languages - pysbd raises ValueError on anything else (Portuguese and Finnish 5 included). subplz can use stanza instead when passed --nlp, which covers 74 more. 6 7 So: resolve the language here, decide the splitter here, and never let a request 8 reach subplz with a language its splitter cannot handle. 9 """ 10 11 import json 12 from dataclasses import dataclass 13 from functools import lru_cache 14 from pathlib import Path 15 16 _REGISTRY = Path(__file__).with_name("languages.json") 17 18 19 @dataclass(frozen=True) 20 class Language: 21 code: str 22 name: str 23 splitter: str # "pysbd" | "stanza" 24 25 @property 26 def needs_nlp_flag(self) -> bool: 27 """stanza languages require subplz's --nlp flag (and a one-off model download).""" 28 return self.splitter == "stanza" 29 30 31 @lru_cache(maxsize=1) 32 def _table() -> dict[str, Language]: 33 raw = json.loads(_REGISTRY.read_text(encoding="utf-8")) 34 return {e["code"]: Language(e["code"], e["name"], e["splitter"]) for e in raw["languages"]} 35 36 37 def all_languages() -> list[Language]: 38 # Sort by display name so the dropdown reads naturally. 39 return sorted(_table().values(), key=lambda l: l.name.lower()) 40 41 42 def get(code: str) -> Language | None: 43 return _table().get((code or "").strip().lower()) 44 45 46 def is_supported(code: str) -> bool: 47 return get(code) is not None 48 49 50 class UnsupportedLanguage(ValueError): 51 def __init__(self, code: str): 52 super().__init__( 53 f"Language {code!r} is not supported. subplz can segment " 54 f"{len(_table())} languages; see GET /api/languages for the list." 55 ) 56 self.code = code 57 58 59 def require(code: str) -> Language: 60 lang = get(code) 61 if lang is None: 62 raise UnsupportedLanguage(code) 63 return lang