Recently Written · git

subplz-web

git clone https://github.com/equwal/subplz-web

Log | Files | Refs


tools/client.py (9055 bytes)

1 """Command-line client for the SubPlz web API.
2 
3 Useful on its own for batch work, and it doubles as a worked example of the
4 three-call flow the browser uses.
5 
6     python tools/client.py submit "book.m4b" "book.epub" [--language ru] [--wait]
7     python tools/client.py list
8     python tools/client.py get <job_id>
9     python tools/client.py download <job_id> [--dir out/]
10 
11 Uses only the standard library so it runs anywhere, and sends multipart bodies
12 itself so non-ASCII filenames survive (curl on a non-UTF-8 console mangles them).
13 """
14 
15 from __future__ import annotations
16 
17 import argparse
18 import http.client
19 import json
20 import mimetypes
21 import os
22 import sys
23 import time
24 import urllib.parse
25 import uuid
26 from pathlib import Path
27 
28 DEFAULT_BASE = os.environ.get("SUBPLZ_WEB_URL", "http://127.0.0.1:8420")
29 COOKIE_FILE = Path.home() / ".subplz_web_cookie"
30 
31 
32 class Client:
33     def __init__(self, base: str):
34         parts = urllib.parse.urlsplit(base)
35         self.host = parts.hostname or "127.0.0.1"
36         self.port = parts.port or (443 if parts.scheme == "https" else 80)
37         self.https = parts.scheme == "https"
38         self.cookie = COOKIE_FILE.read_text().strip() if COOKIE_FILE.exists() else ""
39 
40     def _conn(self):
41         cls = http.client.HTTPSConnection if self.https else http.client.HTTPConnection
42         return cls(self.host, self.port, timeout=900)
43 
44     def _headers(self, extra: dict | None = None) -> dict:
45         h = {"Accept": "application/json"}
46         if self.cookie:
47             h["Cookie"] = self.cookie
48         h.update(extra or {})
49         return h
50 
51     def _remember_cookie(self, resp) -> None:
52         raw = resp.getheader("set-cookie")
53         if raw:
54             self.cookie = raw.split(";", 1)[0]
55             try:
56                 COOKIE_FILE.write_text(self.cookie)
57             except OSError:
58                 pass
59 
60     def request(self, method: str, path: str, body=None, headers=None):
61         conn = self._conn()
62         conn.request(method, path, body, self._headers(headers))
63         resp = conn.getresponse()
64         raw = resp.read().decode("utf-8", errors="replace")
65         self._remember_cookie(resp)
66         conn.close()
67 
68         if resp.status >= 400:
69             try:
70                 detail = json.loads(raw).get("detail", raw)
71             except json.JSONDecodeError:
72                 detail = raw[:500]
73             raise SystemExit(f"error {resp.status}: {detail}")
74         return json.loads(raw) if raw else None
75 
76     def upload(self, audio: list[Path], text: Path) -> dict:
77         boundary = "----subplz" + uuid.uuid4().hex
78         body = bytearray()
79         for path in [*audio, text]:
80             ctype = mimetypes.guess_type(path.name)[0] or "application/octet-stream"
81             disposition = (
82                 f'--{boundary}\r\n'
83                 f'Content-Disposition: form-data; name="files"; '
84                 f'filename="{path.name}"\r\n'
85                 f"Content-Type: {ctype}\r\n\r\n"
86             )
87             # The header is encoded as UTF-8, which is what browsers send and
88             # what Starlette decodes - this is the bit curl gets wrong.
89             body += disposition.encode("utf-8")
90             body += path.read_bytes()
91             body += b"\r\n"
92         body += f"--{boundary}--\r\n".encode()
93 
94         return self.request(
95             "POST", "/api/uploads", bytes(body),
96             {"Content-Type": f"multipart/form-data; boundary={boundary}"},
97         )
98 
99     def start(self, job_id: str, language: str | None) -> dict:
100         payload = json.dumps({"language": language} if language else {})
101         return self.request(
102             "POST", f"/api/jobs/{job_id}/start",
103             payload.encode(), {"Content-Type": "application/json"},
104         )
105 
106     def get(self, job_id: str) -> dict:
107         return self.request("GET", f"/api/jobs/{job_id}")
108 
109     def list(self) -> list:
110         return self.request("GET", "/api/jobs")
111 
112     def download(self, job_id: str, kind: str, out_dir: Path) -> Path | None:
113         job = self.get(job_id)
114         art = next((a for a in job["artifacts"] if a["kind"] == kind), None)
115         if art is None:
116             return None
117         conn = self._conn()
118         conn.request("GET", art["url"], None, self._headers())
119         resp = conn.getresponse()
120         data = resp.read()
121         conn.close()
122         if resp.status >= 400:
123             raise SystemExit(f"download failed: {resp.status}")
124         out_dir.mkdir(parents=True, exist_ok=True)
125         dest = out_dir / art["filename"]
126         dest.write_bytes(data)
127         return dest
128 
129 
130 def _fmt(job: dict) -> str:
131     pct = int(job["progress"] * 100)
132     return (
133         f"{job['id']}  {job['status']:<10} {pct:>3}%  "
134         f"{job['stage']:<34} {job['language_name']}  {job['audio_filename']}"
135     )
136 
137 
138 def wait(client: Client, job_id: str) -> dict:
139     last = None
140     while True:
141         job = client.get(job_id)
142         line = _fmt(job)
143         if line != last:
144             print(line, flush=True)
145             last = line
146         if job["status"] in ("succeeded", "failed", "canceled"):
147             return job
148         time.sleep(3)
149 
150 
151 def main() -> int:
152     ap = argparse.ArgumentParser(description=__doc__)
153     ap.add_argument("--base", default=DEFAULT_BASE)
154     sub = ap.add_subparsers(dest="cmd", required=True)
155 
156     s = sub.add_parser("submit", help="upload a pair and start aligning")
157     s.add_argument("audio")
158     s.add_argument("text")
159     s.add_argument("--language", help="override the detected language")
160     s.add_argument("--wait", action="store_true")
161     s.add_argument("--dir", default="subplz-out", help="where to save on --wait")
162 
163     sub.add_parser("list", help="list your jobs")
164 
165     g = sub.add_parser("get", help="show one job")
166     g.add_argument("job_id")
167 
168     w = sub.add_parser("watch", help="follow a job to completion")
169     w.add_argument("job_id")
170 
171     d = sub.add_parser("download", help="save a finished job's files")
172     d.add_argument("job_id")
173     d.add_argument("--dir", default="subplz-out")
174 
175     args = ap.parse_args()
176     client = Client(args.base)
177 
178     if args.cmd == "submit":
179         audio_arg, text = Path(args.audio), Path(args.text)
180         if not text.exists():
181             raise SystemExit(f"no such file: {text}")
182 
183         if audio_arg.is_dir():
184             # A folder of per-chapter files. Order is settled server-side.
185             exts = {".m4b", ".m4a", ".mp3", ".opus", ".ogg", ".oga", ".flac",
186                     ".wav", ".aac", ".wma", ".mka", ".mkv", ".mp4", ".webm"}
187             audio = sorted(
188                 p for p in audio_arg.iterdir()
189                 if p.is_file() and p.suffix.lower() in exts
190             )
191             if not audio:
192                 raise SystemExit(f"no audio files in {audio_arg}")
193         else:
194             if not audio_arg.exists():
195                 raise SystemExit(f"no such file: {audio_arg}")
196             audio = [audio_arg]
197 
198         size_mb = (sum(p.stat().st_size for p in audio) + text.stat().st_size) / 1024**2
199         label = f"{len(audio)} audio files" if len(audio) > 1 else audio[0].name
200         print(f"uploading {label}, {size_mb:.0f} MB ...", flush=True)
201         res = client.upload(audio, text)
202 
203         job, det = res["job"], res["detected"]
204         print(f"paired: audio={job['audio_filename']} text={job['text_filename']}")
205         if det["code"]:
206             print(
207                 f"detected language: {det['name']} ({det['code']}) "
208                 f"{det['confidence'] * 100:.1f}% confident"
209                 + ("" if det["supported"] else "  [NOT SUPPORTED - pass --language]")
210             )
211         lang = args.language or job["language"]
212         print(f"starting with language={lang}")
213         started = client.start(job["id"], args.language)
214         print(_fmt(started))
215 
216         if args.wait:
217             final = wait(client, job["id"])
218             if final["status"] == "succeeded":
219                 for kind in ("srt", "metadata"):
220                     dest = client.download(job["id"], kind, Path(args.dir))
221                     if dest:
222                         print(f"saved {dest}")
223                 return 0
224             print(f"job {final['status']}: {final.get('error') or ''}")
225             return 1
226         return 0
227 
228     if args.cmd == "list":
229         jobs = client.list()
230         if not jobs:
231             print("no jobs")
232         for j in jobs:
233             print(_fmt(j))
234         return 0
235 
236     if args.cmd == "get":
237         print(json.dumps(client.get(args.job_id), indent=2, ensure_ascii=False))
238         return 0
239 
240     if args.cmd == "watch":
241         final = wait(client, args.job_id)
242         return 0 if final["status"] == "succeeded" else 1
243 
244     if args.cmd == "download":
245         any_saved = False
246         for kind in ("srt", "metadata", "log"):
247             dest = client.download(args.job_id, kind, Path(args.dir))
248             if dest:
249                 print(f"saved {dest}")
250                 any_saved = True
251         if not any_saved:
252             print("nothing to download yet")
253         return 0
254 
255     return 1
256 
257 
258 if __name__ == "__main__":
259     sys.exit(main())