talkthrough 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
talkthrough/capture.py ADDED
@@ -0,0 +1,344 @@
1
+ """Capture a page into a folder: page.png (the whole page), boxes.json (its sections), page.txt, page.json."""
2
+ import asyncio
3
+ import base64
4
+ import io
5
+ import json
6
+ import os
7
+ import platform
8
+ import re
9
+ import shutil
10
+ import socket
11
+ import subprocess
12
+ import sys
13
+ import tempfile
14
+ import time
15
+ import urllib.parse
16
+ import urllib.request
17
+ from pathlib import Path
18
+
19
+ from PIL import Image
20
+
21
+ Image.MAX_IMAGE_PIXELS = None
22
+
23
+
24
+ def find_chrome():
25
+ if os.environ.get("CHROME"):
26
+ return os.environ["CHROME"]
27
+ candidates = {
28
+ "Darwin": ["/Applications/Google Chrome.app/Contents/MacOS/Google Chrome",
29
+ "/Applications/Chromium.app/Contents/MacOS/Chromium",
30
+ "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge"],
31
+ "Windows": [r"C:\Program Files\Google\Chrome\Application\chrome.exe",
32
+ r"C:\Program Files (x86)\Google\Chrome\Application\chrome.exe",
33
+ r"C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe"],
34
+ }.get(platform.system(), [])
35
+ for c in candidates:
36
+ if os.path.exists(c):
37
+ return c
38
+ for name in ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome", "msedge"):
39
+ p = shutil.which(name)
40
+ if p:
41
+ return p
42
+ sys.exit("No Chrome, Chromium or Edge found. Install one, or set CHROME=/path/to/browser.")
43
+
44
+
45
+ BOXES_JS = r"""
46
+ (() => {
47
+ const sel = 'h1,h2,h3,header,section,article,figure,svg,table,pre,img,canvas,blockquote,ol,ul,p,details,' +
48
+ '[class*=slide],[class*=card],[class*=tile],[class*=callout],[class*=seg],[class*=diagram],[class*=box],' +
49
+ 'aside,[class*=good],[class*=warn],[class*=note],[class*=tip],[class*=info],[class*=alert],[class*=panel]';
50
+ const seen = new Set(), out = [];
51
+ for (const el of document.querySelectorAll(sel)) {
52
+ if (el.closest('svg') && el.tagName.toLowerCase() !== 'svg') continue;
53
+ const r = el.getBoundingClientRect();
54
+ if (r.width < 80 || r.height < 24) continue;
55
+ const st = getComputedStyle(el);
56
+ if (st.visibility === 'hidden' || st.display === 'none' || +st.opacity === 0) continue;
57
+ const x = Math.round(r.left + scrollX), y = Math.round(r.top + scrollY);
58
+ const key = [x, y, Math.round(r.width), Math.round(r.height)].join(',');
59
+ if (seen.has(key)) continue;
60
+ seen.add(key);
61
+ let depth = 0; for (let p = el.parentElement; p; p = p.parentElement) depth++;
62
+ const text = (el.innerText || el.textContent || '').replace(/\s+/g, ' ').trim().slice(0, 140);
63
+ out.push({tag: el.tagName.toLowerCase(), cls: (el.getAttribute('class') || '').slice(0, 60),
64
+ rect: [x, y, Math.round(r.width), Math.round(r.height)], depth, text});
65
+ if (out.length >= 600) break;
66
+ }
67
+ return out;
68
+ })()
69
+ """
70
+
71
+ FREEZE_JS = r"""
72
+ (() => {
73
+ const target = location.hash ? document.getElementById(decodeURIComponent(location.hash.slice(1))) : null;
74
+ if (target) {
75
+ if (target.tagName === 'DETAILS') target.open = true;
76
+ for (let d = target.closest('details'); d; d = d.parentElement && d.parentElement.closest('details')) d.open = true;
77
+ }
78
+ const s = document.createElement('style');
79
+ s.textContent = '*,*::before,*::after{animation:none!important;transition:none!important;caret-color:transparent!important}';
80
+ document.head.appendChild(s);
81
+ for (const el of document.querySelectorAll('*')) {
82
+ const p = getComputedStyle(el).position;
83
+ if (p === 'fixed') el.style.setProperty('display', 'none', 'important');
84
+ else if (p === 'sticky') el.style.position = 'static';
85
+ }
86
+ return true;
87
+ })()
88
+ """
89
+
90
+
91
+ # ---------- capture (Chrome over the DevTools protocol) ----------
92
+
93
+ class CDP:
94
+ def __init__(self, ws):
95
+ self.ws, self.n, self.events = ws, 0, []
96
+
97
+ async def call(self, method, **params):
98
+ self.n += 1
99
+ mid = self.n
100
+ await self.ws.send(json.dumps({"id": mid, "method": method, "params": params}))
101
+ while True:
102
+ msg = json.loads(await self.ws.recv())
103
+ if msg.get("id") == mid:
104
+ if "error" in msg:
105
+ raise RuntimeError(f"{method}: {msg['error']}")
106
+ return msg.get("result", {})
107
+ self.events.append(msg)
108
+
109
+ async def wait_event(self, name, timeout=30):
110
+ end = time.time() + timeout
111
+ while time.time() < end:
112
+ if any(e.get("method") == name for e in self.events):
113
+ return
114
+ try:
115
+ msg = json.loads(await asyncio.wait_for(self.ws.recv(), timeout=1))
116
+ self.events.append(msg)
117
+ except asyncio.TimeoutError:
118
+ pass
119
+ raise TimeoutError(name)
120
+
121
+ async def js(self, expr):
122
+ r = await self.call("Runtime.evaluate", expression=expr, returnByValue=True, awaitPromise=True)
123
+ return r.get("result", {}).get("value")
124
+
125
+
126
+ def _free_port():
127
+ with socket.socket() as s:
128
+ s.bind(("127.0.0.1", 0))
129
+ return s.getsockname()[1]
130
+
131
+
132
+ async def _capture(url, width, scale):
133
+ import websockets
134
+ port = _free_port()
135
+ profile = tempfile.mkdtemp(prefix="walkthrough-chrome-")
136
+ proc = subprocess.Popen([find_chrome(), "--headless=new", f"--remote-debugging-port={port}", f"--user-data-dir={profile}",
137
+ "--hide-scrollbars", "--no-first-run", "--no-default-browser-check", "about:blank"],
138
+ stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
139
+ try:
140
+ ws_url = None
141
+ for _ in range(100):
142
+ try:
143
+ targets = json.load(urllib.request.urlopen(f"http://127.0.0.1:{port}/json/list", timeout=1))
144
+ ws_url = next(t["webSocketDebuggerUrl"] for t in targets if t.get("type") == "page")
145
+ break
146
+ except Exception:
147
+ time.sleep(0.1)
148
+ if not ws_url:
149
+ sys.exit("The browser did not start. Set CHROME=/path/to/browser.")
150
+ async with websockets.connect(ws_url, max_size=None) as ws:
151
+ c = CDP(ws)
152
+ await c.call("Page.enable")
153
+ await c.call("Emulation.setDeviceMetricsOverride", width=width, height=1200,
154
+ deviceScaleFactor=scale, mobile=False)
155
+ await c.call("Page.navigate", url=url)
156
+ await c.wait_event("Page.loadEventFired")
157
+ for _ in range(30):
158
+ title = await c.js("document.title + ' ' + (document.body ? document.body.innerText.slice(0, 200) : '')")
159
+ if not re.search(r"just a moment|checking your browser|attention required|verify you are human",
160
+ title or "", re.I):
161
+ break
162
+ await asyncio.sleep(1)
163
+ else:
164
+ print("warning: the page still shows a bot check; capture a saved copy instead", file=sys.stderr)
165
+ await c.js("document.fonts ? document.fonts.ready.then(() => true) : true")
166
+ await asyncio.sleep(0.8)
167
+ await c.js(FREEZE_JS)
168
+ await asyncio.sleep(0.3)
169
+ height = int(await c.js("Math.ceil(document.documentElement.scrollHeight)"))
170
+ boxes = [b for b in await c.js(BOXES_JS) if b["rect"][1] + b["rect"][3] <= height + 2]
171
+ text = await c.js("document.body.innerText")
172
+ title = await c.js("document.title")
173
+ bg = await c.js("getComputedStyle(document.body).backgroundColor")
174
+ await c.call("Emulation.setDeviceMetricsOverride", width=width, height=min(height, 8000),
175
+ deviceScaleFactor=scale, mobile=False)
176
+ await asyncio.sleep(0.3)
177
+ page = Image.new("RGB", (width * scale, height * scale), "white")
178
+ tile = 3000
179
+ for y in range(0, height, tile):
180
+ h = min(tile, height - y)
181
+ shot = await c.call("Page.captureScreenshot", format="png", captureBeyondViewport=True,
182
+ clip={"x": 0, "y": y, "width": width, "height": h, "scale": 1})
183
+ im = Image.open(io.BytesIO(base64.b64decode(shot["data"]))).convert("RGB")
184
+ page.paste(im, (0, y * scale))
185
+ try:
186
+ await c.call("Browser.close")
187
+ except Exception:
188
+ pass
189
+ finally:
190
+ proc.terminate()
191
+ try:
192
+ proc.wait(timeout=5)
193
+ except subprocess.TimeoutExpired:
194
+ proc.kill()
195
+ shutil.rmtree(profile, ignore_errors=True)
196
+
197
+ meta = {"url": url, "title": title, "width": width, "height": height, "scale": scale, "bg": bg, "kind": "html"}
198
+ return page, text, meta, boxes
199
+
200
+
201
+ PDF_GAP = 24
202
+
203
+
204
+ def _profile(mask, box, axis):
205
+ """Which rows (axis=0) or columns (axis=1) of box contain ink."""
206
+ x0, y0, x1, y1 = box
207
+ region = mask.crop(box)
208
+ size = (1, y1 - y0) if axis == 0 else (x1 - x0, 1)
209
+ return [v > 0 for v in region.resize(size, Image.Resampling.BOX).getdata()]
210
+
211
+
212
+ def _runs(ink, min_gap):
213
+ """Spans of ink separated by at least min_gap empty cells."""
214
+ spans, start, gap = [], None, 0
215
+ for i, on in enumerate(ink):
216
+ if on:
217
+ if start is None:
218
+ start = i
219
+ gap = 0
220
+ end = i + 1
221
+ elif start is not None:
222
+ gap += 1
223
+ if gap >= min_gap:
224
+ spans.append((start, end))
225
+ start = None
226
+ if start is not None:
227
+ spans.append((start, end))
228
+ return spans
229
+
230
+
231
+ def _xy_cut(mask, box, depth, out, gap_y=10, gap_x=14):
232
+ """Recursive whitespace cut: bands of content, then columns inside a band. Records every node."""
233
+ x0, y0, x1, y1 = box
234
+ rows = _runs(_profile(mask, box, 0), 1)
235
+ cols = _runs(_profile(mask, box, 1), 1)
236
+ if not rows or not cols:
237
+ return
238
+ box = x0 + cols[0][0], y0 + rows[0][0], x0 + cols[-1][1], y0 + rows[-1][1]
239
+ x0, y0, x1, y1 = box
240
+ if x1 - x0 >= 40 and y1 - y0 >= 12:
241
+ out.append((box, depth))
242
+ if depth >= 4:
243
+ return
244
+ bands = _runs(_profile(mask, box, 0), gap_y)
245
+ if len(bands) > 1:
246
+ for a, b in bands:
247
+ _xy_cut(mask, (x0, y0 + a, x1, y0 + b), depth + 1, out, gap_y, gap_x)
248
+ return
249
+ columns = _runs(_profile(mask, box, 1), gap_x)
250
+ if len(columns) > 1:
251
+ for a, b in columns:
252
+ _xy_cut(mask, (x0 + a, y0, x0 + b, y1), depth + 1, out, gap_y, gap_x)
253
+
254
+
255
+ def _undouble(word):
256
+ """Outlined SVG text often extracts with every glyph twice ("TTHHEE"); collapse such words."""
257
+ if len(word) >= 4 and len(word) % 2 == 0 and all(word[i] == word[i + 1] for i in range(0, len(word), 2)):
258
+ return word[::2]
259
+ return word
260
+
261
+
262
+ def _ink_mask(img, bg):
263
+ """White where the page differs from its background colour."""
264
+ from PIL import ImageChops
265
+ diff = ImageChops.difference(img, Image.new("RGB", img.size, bg)).convert("L")
266
+ return diff.point(lambda v: 255 if v > 18 else 0)
267
+
268
+
269
+ def capture_pdf(path, out, scale, name=None):
270
+ import pypdfium2 as pdfium
271
+ pdf = pdfium.PdfDocument(str(path))
272
+ sizes = [pdf[i].get_size() for i in range(len(pdf))]
273
+ width = int(max(s[0] for s in sizes)) + 2 * PDF_GAP
274
+ height = int(sum(s[1] for s in sizes) + PDF_GAP * (len(sizes) + 1))
275
+ sheet = Image.new("RGB", (width * scale, height * scale), (232, 232, 235))
276
+ boxes, texts, y = [], [], PDF_GAP
277
+ for i in range(len(pdf)):
278
+ page = pdf[i]
279
+ pw, ph = sizes[i]
280
+ x = (width - pw) / 2
281
+ img = page.render(scale=scale).to_pil().convert("RGB")
282
+ sheet.paste(img, (int(x * scale), int(y * scale)))
283
+ tp = page.get_textpage()
284
+ text = tp.get_text_range()
285
+ texts.append(f"--- page {i + 1} ---\n{text}")
286
+ boxes.append({"tag": "page", "cls": f"page {i + 1}", "rect": [round(x), round(y), round(pw), round(ph)],
287
+ "depth": 0, "text": " ".join(text.split())[:140]})
288
+ small = img.resize((round(pw), round(ph)), Image.Resampling.BOX)
289
+ bg = max(small.getcolors(small.width * small.height), key=lambda c: c[0])[1]
290
+ nodes = []
291
+ _xy_cut(_ink_mask(small, bg), (0, 0, small.width, small.height), 1, nodes)
292
+ seen = set()
293
+ for (bx0, by0, bx1, by1), depth in nodes:
294
+ if (bx0, by0, bx1, by1) in seen or (bx1 - bx0) * (by1 - by0) > 0.9 * pw * ph:
295
+ continue
296
+ seen.add((bx0, by0, bx1, by1))
297
+ bt = " ".join(_undouble(w) for w in tp.get_text_bounded(bx0, ph - by1, bx1, ph - by0).split())
298
+ boxes.append({"tag": "region", "cls": f"page {i + 1}", "depth": depth, "text": bt[:140],
299
+ "rect": [round(x + bx0), round(y + by0), bx1 - bx0, by1 - by0]})
300
+ y += ph + PDF_GAP
301
+ title = (pdf.get_metadata_dict().get("Title") or "").strip() or re.sub(r"[-_]+", " ", name or Path(path).stem)
302
+ meta = {"url": Path(path).resolve().as_uri(), "title": title, "width": width, "height": height,
303
+ "scale": scale, "bg": "rgb(232, 232, 235)", "kind": "pdf", "pages": len(pdf)}
304
+ return sheet, "\n\n".join(texts), meta, boxes
305
+
306
+
307
+ def _remote_pdf(url):
308
+ """Download url to a temp file if it serves a PDF (arXiv's /pdf/ links have no .pdf suffix), else None."""
309
+ req = urllib.request.Request(url, headers={"User-Agent": "talkthrough"})
310
+ with urllib.request.urlopen(req, timeout=60) as r:
311
+ if "pdf" not in (r.headers.get("Content-Type") or "").lower() and not url.lower().split("?")[0].endswith(".pdf"):
312
+ return None
313
+ fd, tmp = tempfile.mkstemp(suffix=".pdf", prefix="talkthrough-")
314
+ with os.fdopen(fd, "wb") as f:
315
+ shutil.copyfileobj(r, f)
316
+ return Path(tmp)
317
+
318
+
319
+ def capture(src, out, width=820, scale=2):
320
+ out = Path(out)
321
+ src = str(src)
322
+ if not re.match(r"^(https?|file):", src) and not Path(src).exists():
323
+ sys.exit(f"no such file: {src}")
324
+ if src.startswith("file:") and not Path(urllib.request.url2pathname(src[5:].split("#")[0])).exists():
325
+ sys.exit(f"no such file: {src}")
326
+ remote_pdf = _remote_pdf(src) if re.match(r"^https?:", src) else None
327
+ if remote_pdf:
328
+ name = urllib.parse.unquote(src.split("?")[0].rstrip("/").rsplit("/", 1)[-1])
329
+ page, text, meta, boxes = capture_pdf(remote_pdf, out, scale, name=re.sub(r"\.pdf$", "", name, flags=re.I))
330
+ meta["url"] = src
331
+ remote_pdf.unlink()
332
+ elif src.lower().endswith(".pdf") and not src.startswith("file:"):
333
+ page, text, meta, boxes = capture_pdf(src, out, scale)
334
+ else:
335
+ url = src if re.match(r"^(https?|file):", src) else Path(src).resolve().as_uri()
336
+ page, text, meta, boxes = asyncio.run(_capture(url, width, scale))
337
+ for i, b in enumerate(boxes):
338
+ b["id"] = f"b{i}"
339
+ out.mkdir(parents=True, exist_ok=True)
340
+ page.save(out / "page.png")
341
+ (out / "page.txt").write_text(text or "")
342
+ (out / "page.json").write_text(json.dumps(meta, indent=1))
343
+ (out / "boxes.json").write_text(json.dumps(boxes, indent=1, ensure_ascii=False))
344
+ return meta, boxes
talkthrough/cli.py ADDED
@@ -0,0 +1,200 @@
1
+ """talkthrough: turn an HTML page or a PDF into a narrated walkthrough video of itself."""
2
+ import argparse
3
+ import json
4
+ import os
5
+ import platform
6
+ import plistlib
7
+ import shutil
8
+ import subprocess
9
+ import sys
10
+ import tempfile
11
+ import webbrowser
12
+ from pathlib import Path
13
+
14
+ from . import __version__
15
+
16
+
17
+ def _job_path(folder):
18
+ return Path(folder) / "shots.json"
19
+
20
+
21
+ def cmd_capture(a):
22
+ from .capture import capture
23
+ meta, boxes = capture(a.page, a.out, a.width, a.scale)
24
+ print(f"captured {meta['title']!r}: {meta['width']}x{meta['height']} css px at {meta['scale']}x, "
25
+ f"{len(boxes)} boxes -> {a.out}")
26
+
27
+
28
+ def _need_planner():
29
+ try:
30
+ import anthropic # noqa: F401
31
+ except ImportError:
32
+ sys.exit("planning needs the anthropic package: pip install "
33
+ "'talkthrough[plan] @ git+https://github.com/amishah1998/talkthrough'")
34
+ if not (os.environ.get("ANTHROPIC_API_KEY") or os.environ.get("ANTHROPIC_AUTH_TOKEN")):
35
+ sys.exit("planning needs ANTHROPIC_API_KEY. Without one, ask Claude Code for the video (the plugin plans "
36
+ "it for you), or run capture, write shots.json yourself, then preview and render.")
37
+
38
+
39
+ def cmd_plan(a):
40
+ _need_planner()
41
+ from .plan import plan
42
+ shots = plan(a.dir, seconds=a.seconds, model=a.model, extra=a.note or "")
43
+ job = {"format": a.format, "tts": a.tts, "shots": shots}
44
+ _job_path(a.dir).write_text(json.dumps(job, indent=1, ensure_ascii=False))
45
+ print(f"{len(shots)} shots -> {_job_path(a.dir)}")
46
+
47
+
48
+ def cmd_preview(a):
49
+ from .render import preview
50
+ preview(a.dir)
51
+
52
+
53
+ def _mac_default_browser():
54
+ """Bundle id of the app that handles https links, or None if the user never changed it (Safari)."""
55
+ prefs = Path.home() / "Library/Preferences/com.apple.LaunchServices/com.apple.launchservices.secure.plist"
56
+ try:
57
+ handlers = plistlib.loads(prefs.read_bytes()).get("LSHandlers", [])
58
+ except (OSError, plistlib.InvalidFileException):
59
+ return None
60
+ for scheme in ("https", "http"):
61
+ for h in handlers:
62
+ if h.get("LSHandlerURLScheme") == scheme and h.get("LSHandlerRoleAll"):
63
+ return h["LSHandlerRoleAll"]
64
+ return None
65
+
66
+
67
+ def open_in_browser(video):
68
+ """Play the finished video in the default web browser.
69
+
70
+ On macOS a plain open hands an .mp4 to QuickTime, so the browser is looked up and named explicitly."""
71
+ video = Path(video).resolve()
72
+ if platform.system() == "Darwin":
73
+ bundle = _mac_default_browser()
74
+ args = ["open", "-b", bundle, str(video)] if bundle else ["open", "-a", "Safari", str(video)]
75
+ if subprocess.run(args, capture_output=True).returncode == 0:
76
+ return
77
+ if not webbrowser.open(video.as_uri()):
78
+ print(f"could not open a browser; the video is at {video}", file=sys.stderr)
79
+
80
+
81
+ def cmd_render(a):
82
+ from .render import render
83
+ out = render(a.dir, a.out, quality="share" if a.share else None, speed=a.speed)
84
+ if a.open:
85
+ open_in_browser(out)
86
+
87
+
88
+ def cmd_make(a):
89
+ _need_planner()
90
+ from .capture import capture
91
+ from .plan import plan
92
+ from .render import preview, render
93
+ folder = Path(a.keep) if a.keep else Path(tempfile.mkdtemp(prefix="talkthrough-"))
94
+ capture(a.page, folder, a.width, a.scale)
95
+ shots = plan(folder, seconds=a.seconds, model=a.model, extra=a.note or "")
96
+ _job_path(folder).write_text(json.dumps({"format": a.format, "tts": a.tts, "shots": shots}, indent=1,
97
+ ensure_ascii=False))
98
+ preview(folder)
99
+ out = Path(a.out) if a.out else Path.cwd() / (Path(a.page).stem + "-walkthrough.mp4")
100
+ render(folder, out, quality="share" if a.share else None, speed=a.speed)
101
+ if not a.keep:
102
+ shutil.rmtree(folder, ignore_errors=True)
103
+ if a.open:
104
+ open_in_browser(out)
105
+
106
+
107
+ def cmd_doctor(a):
108
+ from .capture import find_chrome
109
+ ok = True
110
+
111
+ def row(name, good, note):
112
+ nonlocal ok
113
+ ok &= good or name.startswith("(optional)")
114
+ print(f" {'ok ' if good else '-- '} {name}: {note}")
115
+
116
+ print(f"talkthrough {__version__} on {platform.system()}")
117
+ try:
118
+ row("browser", True, find_chrome())
119
+ except SystemExit as e:
120
+ row("browser", False, str(e))
121
+ row("ffmpeg", bool(shutil.which("ffmpeg")), shutil.which("ffmpeg") or "install ffmpeg")
122
+ row("ffprobe", bool(shutil.which("ffprobe")), shutil.which("ffprobe") or "comes with ffmpeg")
123
+ from .voice import ORDER, kokoro_available
124
+ voices = [name for name, env in ORDER if os.environ.get(env)]
125
+ if kokoro_available():
126
+ voices.append("kokoro (local)")
127
+ if platform.system() == "Darwin" and shutil.which("say"):
128
+ voices.append("say (macOS)")
129
+ row("voice", bool(voices), ", ".join(voices) or "none: set CARTESIA_API_KEY, ELEVENLABS_API_KEY or OPENAI_API_KEY, "
130
+ "or pip install 'talkthrough[local] @ git+https://github.com/amishah1998/talkthrough'")
131
+ try:
132
+ import anthropic # noqa: F401
133
+ has_sdk = True
134
+ except ImportError:
135
+ has_sdk = False
136
+ auto = has_sdk and bool(os.environ.get("ANTHROPIC_API_KEY") or os.environ.get("ANTHROPIC_AUTH_TOKEN"))
137
+ row("(optional) auto-plan", auto, "ready" if auto else
138
+ "needs pip install 'talkthrough[plan] @ git+https://github.com/amishah1998/talkthrough' and ANTHROPIC_API_KEY; not needed when an agent writes shots.json")
139
+ sys.exit(0 if ok else 1)
140
+
141
+
142
+ def main(argv=None):
143
+ ap = argparse.ArgumentParser(prog="talkthrough", description=__doc__)
144
+ ap.add_argument("--version", action="version", version=__version__)
145
+ sub = ap.add_subparsers(dest="cmd", required=True)
146
+
147
+ def capture_args(p):
148
+ p.add_argument("--width", type=int, default=820, help="viewport width in css px for HTML (default 820)")
149
+ p.add_argument("--scale", type=int, default=2, help="capture pixel density (default 2)")
150
+
151
+ def plan_args(p):
152
+ p.add_argument("--seconds", type=int, default=90, help="target length (default 90)")
153
+ p.add_argument("--format", default="portrait", choices=["portrait", "landscape", "square"])
154
+ p.add_argument("--tts", default="auto", choices=["auto", "cartesia", "elevenlabs", "openai", "kokoro", "say"])
155
+ p.add_argument("--model", default="claude-opus-5", help="Claude model that plans the shots")
156
+ p.add_argument("--note", help="extra direction for the planner, e.g. 'focus on the pricing table'")
157
+
158
+ p = sub.add_parser("capture", help="screenshot the page and list its sections")
159
+ p.add_argument("page", help=".html, .pdf, or an http(s) URL")
160
+ p.add_argument("--out", required=True, help="work folder")
161
+ capture_args(p)
162
+ p.set_defaults(fn=cmd_capture)
163
+
164
+ p = sub.add_parser("plan", help="let Claude write shots.json for a captured page")
165
+ p.add_argument("dir")
166
+ plan_args(p)
167
+ p.set_defaults(fn=cmd_plan)
168
+
169
+ p = sub.add_parser("preview", help="one spotlighted frame per shot, as preview.png")
170
+ p.add_argument("dir")
171
+ p.set_defaults(fn=cmd_preview)
172
+
173
+ p = sub.add_parser("render", help="voice, camera and captions into an MP4")
174
+ p.add_argument("dir")
175
+ p.add_argument("--out")
176
+ p.add_argument("--share", action="store_true", help="smaller file for chat apps (about half the size)")
177
+ p.add_argument("--speed", type=float, help="voice pace, e.g. 1.25 or 1.5 (default 1.0)")
178
+ p.add_argument("--open", action="store_true", help="play the video in your web browser when it is done")
179
+ p.set_defaults(fn=cmd_render)
180
+
181
+ p = sub.add_parser("make", help="capture, plan, preview and render in one go")
182
+ p.add_argument("page")
183
+ p.add_argument("--out", help="output .mp4 (default: <page>-walkthrough.mp4)")
184
+ p.add_argument("--keep", help="keep the work folder here")
185
+ p.add_argument("--share", action="store_true", help="smaller file for chat apps (about half the size)")
186
+ p.add_argument("--speed", type=float, help="voice pace, e.g. 1.25 or 1.5 (default 1.0)")
187
+ p.add_argument("--open", action="store_true", help="play the video in your web browser when it is done")
188
+ capture_args(p)
189
+ plan_args(p)
190
+ p.set_defaults(fn=cmd_make)
191
+
192
+ p = sub.add_parser("doctor", help="check what is installed")
193
+ p.set_defaults(fn=cmd_doctor)
194
+
195
+ a = ap.parse_args(argv)
196
+ a.fn(a)
197
+
198
+
199
+ if __name__ == "__main__":
200
+ main()
@@ -0,0 +1,46 @@
1
+ """The free local voice: Kokoro-82M (Apache-2.0 weights) through kokoro-onnx. No torch, no GPU.
2
+ The model is downloaded once into the user cache folder."""
3
+ import os
4
+ import sys
5
+ import urllib.request
6
+ from pathlib import Path
7
+
8
+ RELEASE = "https://github.com/thewh1teagle/kokoro-onnx/releases/download/model-files-v1.0/"
9
+ FILES = {"model": "kokoro-v1.0.onnx", "voices": "voices-v1.0.bin"}
10
+ _engine = None
11
+
12
+
13
+ def cache_dir():
14
+ base = os.environ.get("TALKTHROUGH_CACHE") or os.environ.get("XDG_CACHE_HOME") or Path.home() / ".cache"
15
+ d = Path(base) / "talkthrough" / "kokoro"
16
+ d.mkdir(parents=True, exist_ok=True)
17
+ return d
18
+
19
+
20
+ def _fetch(name):
21
+ path = cache_dir() / name
22
+ if path.exists() and path.stat().st_size > 0:
23
+ return path
24
+ print(f"downloading {name} (one time) ...", file=sys.stderr, flush=True)
25
+ tmp = path.with_suffix(".part")
26
+ urllib.request.urlretrieve(RELEASE + name, tmp)
27
+ tmp.rename(path)
28
+ return path
29
+
30
+
31
+ def engine():
32
+ global _engine
33
+ if _engine is None:
34
+ from kokoro_onnx import Kokoro
35
+ _engine = Kokoro(str(_fetch(FILES["model"])), str(_fetch(FILES["voices"])))
36
+ return _engine
37
+
38
+
39
+ def speak(text, voice, speed, wav, sample_rate):
40
+ import soundfile as sf
41
+ samples, sr = engine().create(text, voice=voice, speed=speed, lang="en-us")
42
+ tmp = Path(wav).with_suffix(".raw.wav")
43
+ sf.write(tmp, samples, sr)
44
+ from .voice import _ffmpeg
45
+ _ffmpeg("-i", tmp, "-ar", sample_rate, "-ac", 1, wav)
46
+ tmp.unlink()