Recently Written · git

vibeslop-dickt

make your own vibeslop dickt in just a few minutes

git clone https://github.com/equwal/vibeslop-dickt

Log | Files | Refs


dickt_convert.py (10978 bytes)

1 """
2 Kenkyusha 新和英大辞典 — English → Portuguese translator
3 Uses Google Translate (no API key needed).
4 
5 Run:  python translate_kenkyusha.py <input_zip_or_dir> <output_dir>
6 
7 Speed improvements over v1:
8   • Sentinel-joined batches  — segments joined with a rare delimiter, one HTTP call per batch
9   • Thread pool              — WORKERS concurrent requests
10   • Exponential backoff      — on 429 / transient errors
11   • tqdm progress bar (optional; pip install tqdm)
12 """
13 
14 import json
15 import re
16 import sys
17 import time
18 import urllib.request
19 import urllib.parse
20 import urllib.error
21 import zipfile
22 import tempfile
23 import shutil
24 from pathlib import Path
25 from concurrent.futures import ThreadPoolExecutor, as_completed
26 
27 try:
28     from tqdm import tqdm
29     HAS_TQDM = True
30 except ImportError:
31     HAS_TQDM = False
32 
33 # ── Config ────────────────────────────────────────────────────────────────────
34 
35 TARGET_LANG    = sys.argv[3] if len(sys.argv) > 3 else None
36 BATCH_SIZE     = 40      # segments joined per HTTP call
37 WORKERS        = 8       # concurrent threads
38 RETRY_LIMIT    = 6       # max attempts per batch
39 BACKOFF_BASE   = 1.5     # seconds; doubles each retry
40 REQUEST_GAP    = 0.02    # polite pause after each successful request
41 
42 # Sentinel must survive a round-trip through Google Translate unchanged.
43 # This Unicode private-use sequence is invisible and never appears in dictionary text.
44 SEP = "\uE000|\uE001"
45 
46 # ── Google Translate (gtx endpoint, sentinel-batching) ────────────────────────
47 
48 _HEADERS = {
49     "User-Agent": (
50         "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
51         "AppleWebKit/537.36 (KHTML, like Gecko) "
52         "Chrome/124.0.0.0 Safari/537.36"
53     )
54 }
55 
56 
57 def translate_batch(texts: list[str]) -> list[str]:
58     """
59     Translate a list of strings in one HTTP call by joining them with SEP.
60     Splits the translated result back on the same separator.
61     Falls back to originals on permanent failure.
62     """
63     if not texts:
64         return texts
65 
66     joined = SEP.join(t[:4000] for t in texts)
67     encoded = urllib.parse.quote(joined)
68     url = (
69         f"https://translate.googleapis.com/translate_a/single"
70         f"?client=gtx&sl=en&tl={TARGET_LANG}&dt=t&q={encoded}"
71     )
72     req = urllib.request.Request(url, headers=_HEADERS)
73 
74     delay = BACKOFF_BASE
75     for attempt in range(RETRY_LIMIT):
76         try:
77             with urllib.request.urlopen(req, timeout=15) as resp:
78                 result = json.loads(resp.read().decode("utf-8"))
79 
80             # result[0] is a list of [translated_chunk, original_chunk, ...]
81             translated_joined = "".join(
82                 item[0] for item in result[0] if isinstance(item[0], str)
83             )
84 
85             # Split on any variant the translator may have introduced
86             # (it sometimes adds spaces around the separator)
87             parts = re.split(r"\s*\uE000\s*\|\s*\uE001\s*", translated_joined)
88 
89             # Pad or trim to match input length
90             while len(parts) < len(texts):
91                 parts.append(texts[len(parts)])
92 
93             time.sleep(REQUEST_GAP)
94             return parts[:len(texts)]
95 
96         except urllib.error.HTTPError as e:
97             code = e.code
98             print(f"\n  ⚠  HTTP {code} (attempt {attempt+1}); backing off {delay:.1f}s …", flush=True)
99             time.sleep(delay)
100             delay *= 2
101 
102         except Exception as e:
103             print(f"\n  ⚠  Error (attempt {attempt+1}): {e}; backing off {delay:.1f}s …", flush=True)
104             time.sleep(delay)
105             delay *= 2
106 
107     print(f"  ✗  Giving up on batch of {len(texts)} — keeping originals", flush=True)
108     return texts
109 
110 
111 # ── Per-line segment extraction ───────────────────────────────────────────────
112 
113 def extract_english_segments(text: str):
114     """
115     Return list of (start, end, english_text) for all English portions.
116 
117     Two patterns in this dictionary:
118       A) Lines with ideographic space U+3000 — English is everything after it.
119       B) Standalone English lines (no U+3000, contains Latin letters,
120          not a JP/romaji header).
121     """
122     segments = []
123     pos = 0
124     for line in text.split("\n"):
125         line_len = len(line)
126         if "\u3000" in line:
127             idx = line.index("\u3000") + 1
128             eng = line[idx:]
129             if re.search(r"[a-zA-Z]{2,}", eng):
130                 start = pos + idx
131                 segments.append((start, start + len(eng), eng))
132         else:
133             if (re.search(r"[a-zA-Z]{2,}", line)
134                     and not line.strip().startswith("[ローマ字]")
135                     and not re.match(r'^[\u4e00-\u9fff\u3040-\u30ff].*\[ローマ字\]', line)):
136                 segments.append((pos, pos + line_len, line))
137         pos += line_len + 1
138     return segments
139 
140 
141 def apply_translations(text: str, segments, translations) -> str:
142     """Splice translations back into the original string at the correct offsets."""
143     if not segments:
144         return text
145     result = []
146     prev = 0
147     for (start, end, _), t in zip(segments, translations):
148         result.append(text[prev:start])
149         result.append(t)
150         prev = end
151     result.append(text[prev:])
152     return "".join(result)
153 
154 
155 # ── Batched + parallel translation of a flat list ────────────────────────────
156 
157 def translate_all(flat_eng: list[str]) -> list[str]:
158     """
159     Translate every string in flat_eng using WORKERS threads and BATCH_SIZE
160     grouping.  Returns a list of the same length with translated strings.
161     """
162     total = len(flat_eng)
163     if total == 0:
164         return []
165 
166     batches = []
167     for start in range(0, total, BATCH_SIZE):
168         batches.append((start, flat_eng[start:start + BATCH_SIZE]))
169 
170     results = [""] * total
171 
172     pbar = tqdm(total=total, unit="seg", dynamic_ncols=True, leave=False) if HAS_TQDM else None
173 
174     def worker(start_idx, texts):
175         translated = translate_batch(texts)
176         return start_idx, translated
177 
178     with ThreadPoolExecutor(max_workers=WORKERS) as pool:
179         futures = {pool.submit(worker, s, t): s for s, t in batches}
180         for fut in as_completed(futures):
181             start_idx, translated = fut.result()
182             for i, t in enumerate(translated):
183                 results[start_idx + i] = t
184             if pbar:
185                 pbar.update(len(translated))
186             else:
187                 done = sum(1 for r in results if r)
188                 print(f"    {done}/{total} segments done   \r", end="", flush=True)
189 
190     if pbar:
191         pbar.close()
192     else:
193         print()
194 
195     return results
196 
197 
198 # ── File processor ────────────────────────────────────────────────────────────
199 
200 def process_term_bank(input_path: Path, output_path: Path):
201     print(f"\n📖  {input_path.name}", flush=True)
202     with open(input_path, "r", encoding="utf-8") as f:
203         data = json.load(f)
204 
205     all_segs = []
206     for i, entry in enumerate(data):
207         if (isinstance(entry, list) and len(entry) > 5
208                 and isinstance(entry[5], list) and entry[5]
209                 and isinstance(entry[5][0], str)):
210             segs = extract_english_segments(entry[5][0])
211             if segs:
212                 all_segs.append((i, segs))
213 
214     flat_eng   = []
215     flat_index = []
216     for ai, (_, segs) in enumerate(all_segs):
217         for si, (_, _, eng) in enumerate(segs):
218             flat_eng.append(eng)
219             flat_index.append((ai, si))
220 
221     total = len(flat_eng)
222     print(f"    {len(data)} entries — {total} English segments to translate", flush=True)
223 
224     flat_translated = translate_all(flat_eng)
225 
226     trans_map = {idx: t for idx, t in zip(flat_index, flat_translated)}
227     for ai, (entry_i, segs) in enumerate(all_segs):
228         seg_trans = [trans_map.get((ai, si), segs[si][2]) for si in range(len(segs))]
229         data[entry_i][5][0] = apply_translations(data[entry_i][5][0], segs, seg_trans)
230 
231     output_path.parent.mkdir(parents=True, exist_ok=True)
232     with open(output_path, "w", encoding="utf-8") as f:
233         json.dump(data, f, ensure_ascii=False, indent=2)
234     print(f"    ✓ saved → {output_path.name}", flush=True)
235 
236 
237 # ── Main ──────────────────────────────────────────────────────────────────────
238 
239 def main():
240     if len(sys.argv) < 4:
241         print("Usage: python translate_kenkyusha.py <input_zip> <output_dir> <target_lang>")
242         sys.exit(1)
243 
244     input_path = Path(sys.argv[1])
245     output_dir = Path(sys.argv[2])
246     output_dir.mkdir(parents=True, exist_ok=True)
247 
248     print(f"⚙  Workers={WORKERS}  BatchSize={BATCH_SIZE}  Target={TARGET_LANG}")
249 
250     if input_path.suffix.lower() != ".zip":
251         print(f"Error: {input_path} is not a .zip file")
252         sys.exit(1)
253     print(f"📦  Extracting {input_path.name} …", flush=True)
254     _tmpdir = tempfile.mkdtemp(prefix="kenkyusha_")
255     with zipfile.ZipFile(input_path, "r") as zf:
256         zf.extractall(_tmpdir)
257     candidates = list(Path(_tmpdir).rglob("term_bank_1.json"))
258     input_dir = candidates[0].parent if candidates else Path(_tmpdir)
259     print(f"    → unpacked to {input_dir}", flush=True)
260 
261     try:
262         idx_src = input_dir / "index.json"
263         if idx_src.exists():
264             with open(idx_src, encoding="utf-8") as f:
265                 idx = json.load(f)
266             idx["title"]    = idx.get("title", "Dictionary") + f" ({TARGET_LANG.upper()})"
267             idx["revision"] = idx.get("revision", "1") + f"-{TARGET_LANG}"
268             with open(output_dir / "index.json", "w", encoding="utf-8") as f:
269                 json.dump(idx, f, ensure_ascii=False, indent=2)
270             print("📝  index.json updated")
271 
272         term_banks = sorted(input_dir.glob("term_bank_*.json"),
273                             key=lambda p: int(re.search(r'\d+', p.name).group()))
274         print(f"Found {len(term_banks)} term_bank files")
275 
276         if not term_banks:
277             print("⚠  No term_bank_*.json files found — check the zip structure.")
278             sys.exit(1)
279 
280         t0 = time.time()
281         for tb in term_banks:
282             process_term_bank(tb, output_dir / tb.name)
283 
284         elapsed = int(time.time() - t0)
285         print(f"\n✅  Done in {elapsed//60}m {elapsed%60}s")
286         print(f"📁  Output: {output_dir}")
287         print("\nNext: zip the output folder contents and import into Yomitan.")
288 
289     finally:
290         if _tmpdir:
291             shutil.rmtree(_tmpdir, ignore_errors=True)
292 
293 
294 if __name__ == "__main__":
295     main()