dickt_convert.py (10978 bytes)
1 """ 2 Kenkyusha 新和英大辞典 — English → Portuguese translator 3 Uses Google Translate (no API key needed). 4 5 Run: python translate_kenkyusha.py <input_zip_or_dir> <output_dir> 6 7 Speed improvements over v1: 8 • Sentinel-joined batches — segments joined with a rare delimiter, one HTTP call per batch 9 • Thread pool — WORKERS concurrent requests 10 • Exponential backoff — on 429 / transient errors 11 • tqdm progress bar (optional; pip install tqdm) 12 """ 13 14 import json 15 import re 16 import sys 17 import time 18 import urllib.request 19 import urllib.parse 20 import urllib.error 21 import zipfile 22 import tempfile 23 import shutil 24 from pathlib import Path 25 from concurrent.futures import ThreadPoolExecutor, as_completed 26 27 try: 28 from tqdm import tqdm 29 HAS_TQDM = True 30 except ImportError: 31 HAS_TQDM = False 32 33 # ── Config ──────────────────────────────────────────────────────────────────── 34 35 TARGET_LANG = sys.argv[3] if len(sys.argv) > 3 else None 36 BATCH_SIZE = 40 # segments joined per HTTP call 37 WORKERS = 8 # concurrent threads 38 RETRY_LIMIT = 6 # max attempts per batch 39 BACKOFF_BASE = 1.5 # seconds; doubles each retry 40 REQUEST_GAP = 0.02 # polite pause after each successful request 41 42 # Sentinel must survive a round-trip through Google Translate unchanged. 43 # This Unicode private-use sequence is invisible and never appears in dictionary text. 44 SEP = "\uE000|\uE001" 45 46 # ── Google Translate (gtx endpoint, sentinel-batching) ──────────────────────── 47 48 _HEADERS = { 49 "User-Agent": ( 50 "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " 51 "AppleWebKit/537.36 (KHTML, like Gecko) " 52 "Chrome/124.0.0.0 Safari/537.36" 53 ) 54 } 55 56 57 def translate_batch(texts: list[str]) -> list[str]: 58 """ 59 Translate a list of strings in one HTTP call by joining them with SEP. 60 Splits the translated result back on the same separator. 61 Falls back to originals on permanent failure. 62 """ 63 if not texts: 64 return texts 65 66 joined = SEP.join(t[:4000] for t in texts) 67 encoded = urllib.parse.quote(joined) 68 url = ( 69 f"https://translate.googleapis.com/translate_a/single" 70 f"?client=gtx&sl=en&tl={TARGET_LANG}&dt=t&q={encoded}" 71 ) 72 req = urllib.request.Request(url, headers=_HEADERS) 73 74 delay = BACKOFF_BASE 75 for attempt in range(RETRY_LIMIT): 76 try: 77 with urllib.request.urlopen(req, timeout=15) as resp: 78 result = json.loads(resp.read().decode("utf-8")) 79 80 # result[0] is a list of [translated_chunk, original_chunk, ...] 81 translated_joined = "".join( 82 item[0] for item in result[0] if isinstance(item[0], str) 83 ) 84 85 # Split on any variant the translator may have introduced 86 # (it sometimes adds spaces around the separator) 87 parts = re.split(r"\s*\uE000\s*\|\s*\uE001\s*", translated_joined) 88 89 # Pad or trim to match input length 90 while len(parts) < len(texts): 91 parts.append(texts[len(parts)]) 92 93 time.sleep(REQUEST_GAP) 94 return parts[:len(texts)] 95 96 except urllib.error.HTTPError as e: 97 code = e.code 98 print(f"\n ⚠ HTTP {code} (attempt {attempt+1}); backing off {delay:.1f}s …", flush=True) 99 time.sleep(delay) 100 delay *= 2 101 102 except Exception as e: 103 print(f"\n ⚠ Error (attempt {attempt+1}): {e}; backing off {delay:.1f}s …", flush=True) 104 time.sleep(delay) 105 delay *= 2 106 107 print(f" ✗ Giving up on batch of {len(texts)} — keeping originals", flush=True) 108 return texts 109 110 111 # ── Per-line segment extraction ─────────────────────────────────────────────── 112 113 def extract_english_segments(text: str): 114 """ 115 Return list of (start, end, english_text) for all English portions. 116 117 Two patterns in this dictionary: 118 A) Lines with ideographic space U+3000 — English is everything after it. 119 B) Standalone English lines (no U+3000, contains Latin letters, 120 not a JP/romaji header). 121 """ 122 segments = [] 123 pos = 0 124 for line in text.split("\n"): 125 line_len = len(line) 126 if "\u3000" in line: 127 idx = line.index("\u3000") + 1 128 eng = line[idx:] 129 if re.search(r"[a-zA-Z]{2,}", eng): 130 start = pos + idx 131 segments.append((start, start + len(eng), eng)) 132 else: 133 if (re.search(r"[a-zA-Z]{2,}", line) 134 and not line.strip().startswith("[ローマ字]") 135 and not re.match(r'^[\u4e00-\u9fff\u3040-\u30ff].*\[ローマ字\]', line)): 136 segments.append((pos, pos + line_len, line)) 137 pos += line_len + 1 138 return segments 139 140 141 def apply_translations(text: str, segments, translations) -> str: 142 """Splice translations back into the original string at the correct offsets.""" 143 if not segments: 144 return text 145 result = [] 146 prev = 0 147 for (start, end, _), t in zip(segments, translations): 148 result.append(text[prev:start]) 149 result.append(t) 150 prev = end 151 result.append(text[prev:]) 152 return "".join(result) 153 154 155 # ── Batched + parallel translation of a flat list ──────────────────────────── 156 157 def translate_all(flat_eng: list[str]) -> list[str]: 158 """ 159 Translate every string in flat_eng using WORKERS threads and BATCH_SIZE 160 grouping. Returns a list of the same length with translated strings. 161 """ 162 total = len(flat_eng) 163 if total == 0: 164 return [] 165 166 batches = [] 167 for start in range(0, total, BATCH_SIZE): 168 batches.append((start, flat_eng[start:start + BATCH_SIZE])) 169 170 results = [""] * total 171 172 pbar = tqdm(total=total, unit="seg", dynamic_ncols=True, leave=False) if HAS_TQDM else None 173 174 def worker(start_idx, texts): 175 translated = translate_batch(texts) 176 return start_idx, translated 177 178 with ThreadPoolExecutor(max_workers=WORKERS) as pool: 179 futures = {pool.submit(worker, s, t): s for s, t in batches} 180 for fut in as_completed(futures): 181 start_idx, translated = fut.result() 182 for i, t in enumerate(translated): 183 results[start_idx + i] = t 184 if pbar: 185 pbar.update(len(translated)) 186 else: 187 done = sum(1 for r in results if r) 188 print(f" {done}/{total} segments done \r", end="", flush=True) 189 190 if pbar: 191 pbar.close() 192 else: 193 print() 194 195 return results 196 197 198 # ── File processor ──────────────────────────────────────────────────────────── 199 200 def process_term_bank(input_path: Path, output_path: Path): 201 print(f"\n📖 {input_path.name}", flush=True) 202 with open(input_path, "r", encoding="utf-8") as f: 203 data = json.load(f) 204 205 all_segs = [] 206 for i, entry in enumerate(data): 207 if (isinstance(entry, list) and len(entry) > 5 208 and isinstance(entry[5], list) and entry[5] 209 and isinstance(entry[5][0], str)): 210 segs = extract_english_segments(entry[5][0]) 211 if segs: 212 all_segs.append((i, segs)) 213 214 flat_eng = [] 215 flat_index = [] 216 for ai, (_, segs) in enumerate(all_segs): 217 for si, (_, _, eng) in enumerate(segs): 218 flat_eng.append(eng) 219 flat_index.append((ai, si)) 220 221 total = len(flat_eng) 222 print(f" {len(data)} entries — {total} English segments to translate", flush=True) 223 224 flat_translated = translate_all(flat_eng) 225 226 trans_map = {idx: t for idx, t in zip(flat_index, flat_translated)} 227 for ai, (entry_i, segs) in enumerate(all_segs): 228 seg_trans = [trans_map.get((ai, si), segs[si][2]) for si in range(len(segs))] 229 data[entry_i][5][0] = apply_translations(data[entry_i][5][0], segs, seg_trans) 230 231 output_path.parent.mkdir(parents=True, exist_ok=True) 232 with open(output_path, "w", encoding="utf-8") as f: 233 json.dump(data, f, ensure_ascii=False, indent=2) 234 print(f" ✓ saved → {output_path.name}", flush=True) 235 236 237 # ── Main ────────────────────────────────────────────────────────────────────── 238 239 def main(): 240 if len(sys.argv) < 4: 241 print("Usage: python translate_kenkyusha.py <input_zip> <output_dir> <target_lang>") 242 sys.exit(1) 243 244 input_path = Path(sys.argv[1]) 245 output_dir = Path(sys.argv[2]) 246 output_dir.mkdir(parents=True, exist_ok=True) 247 248 print(f"⚙ Workers={WORKERS} BatchSize={BATCH_SIZE} Target={TARGET_LANG}") 249 250 if input_path.suffix.lower() != ".zip": 251 print(f"Error: {input_path} is not a .zip file") 252 sys.exit(1) 253 print(f"📦 Extracting {input_path.name} …", flush=True) 254 _tmpdir = tempfile.mkdtemp(prefix="kenkyusha_") 255 with zipfile.ZipFile(input_path, "r") as zf: 256 zf.extractall(_tmpdir) 257 candidates = list(Path(_tmpdir).rglob("term_bank_1.json")) 258 input_dir = candidates[0].parent if candidates else Path(_tmpdir) 259 print(f" → unpacked to {input_dir}", flush=True) 260 261 try: 262 idx_src = input_dir / "index.json" 263 if idx_src.exists(): 264 with open(idx_src, encoding="utf-8") as f: 265 idx = json.load(f) 266 idx["title"] = idx.get("title", "Dictionary") + f" ({TARGET_LANG.upper()})" 267 idx["revision"] = idx.get("revision", "1") + f"-{TARGET_LANG}" 268 with open(output_dir / "index.json", "w", encoding="utf-8") as f: 269 json.dump(idx, f, ensure_ascii=False, indent=2) 270 print("📝 index.json updated") 271 272 term_banks = sorted(input_dir.glob("term_bank_*.json"), 273 key=lambda p: int(re.search(r'\d+', p.name).group())) 274 print(f"Found {len(term_banks)} term_bank files") 275 276 if not term_banks: 277 print("⚠ No term_bank_*.json files found — check the zip structure.") 278 sys.exit(1) 279 280 t0 = time.time() 281 for tb in term_banks: 282 process_term_bank(tb, output_dir / tb.name) 283 284 elapsed = int(time.time() - t0) 285 print(f"\n✅ Done in {elapsed//60}m {elapsed%60}s") 286 print(f"📁 Output: {output_dir}") 287 print("\nNext: zip the output folder contents and import into Yomitan.") 288 289 finally: 290 if _tmpdir: 291 shutil.rmtree(_tmpdir, ignore_errors=True) 292 293 294 if __name__ == "__main__": 295 main()