talkthrough 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- talkthrough/__init__.py +1 -0
- talkthrough/__main__.py +3 -0
- talkthrough/capture.py +344 -0
- talkthrough/cli.py +200 -0
- talkthrough/local_voice.py +46 -0
- talkthrough/plan.py +124 -0
- talkthrough/render.py +606 -0
- talkthrough/voice.py +231 -0
- talkthrough-0.1.0.dist-info/METADATA +183 -0
- talkthrough-0.1.0.dist-info/RECORD +13 -0
- talkthrough-0.1.0.dist-info/WHEEL +4 -0
- talkthrough-0.1.0.dist-info/entry_points.txt +2 -0
- talkthrough-0.1.0.dist-info/licenses/LICENSE +21 -0
talkthrough/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
talkthrough/__main__.py
ADDED
talkthrough/capture.py
ADDED
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
"""Capture a page into a folder: page.png (the whole page), boxes.json (its sections), page.txt, page.json."""
|
|
2
|
+
import asyncio
|
|
3
|
+
import base64
|
|
4
|
+
import io
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import platform
|
|
8
|
+
import re
|
|
9
|
+
import shutil
|
|
10
|
+
import socket
|
|
11
|
+
import subprocess
|
|
12
|
+
import sys
|
|
13
|
+
import tempfile
|
|
14
|
+
import time
|
|
15
|
+
import urllib.parse
|
|
16
|
+
import urllib.request
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from PIL import Image
|
|
20
|
+
|
|
21
|
+
Image.MAX_IMAGE_PIXELS = None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def find_chrome():
|
|
25
|
+
if os.environ.get("CHROME"):
|
|
26
|
+
return os.environ["CHROME"]
|
|
27
|
+
candidates = {
|
|
28
|
+
"Darwin": ["/Applications/Google Chrome.app/Contents/MacOS/Google Chrome",
|
|
29
|
+
"/Applications/Chromium.app/Contents/MacOS/Chromium",
|
|
30
|
+
"/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge"],
|
|
31
|
+
"Windows": [r"C:\Program Files\Google\Chrome\Application\chrome.exe",
|
|
32
|
+
r"C:\Program Files (x86)\Google\Chrome\Application\chrome.exe",
|
|
33
|
+
r"C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe"],
|
|
34
|
+
}.get(platform.system(), [])
|
|
35
|
+
for c in candidates:
|
|
36
|
+
if os.path.exists(c):
|
|
37
|
+
return c
|
|
38
|
+
for name in ("google-chrome", "google-chrome-stable", "chromium", "chromium-browser", "chrome", "msedge"):
|
|
39
|
+
p = shutil.which(name)
|
|
40
|
+
if p:
|
|
41
|
+
return p
|
|
42
|
+
sys.exit("No Chrome, Chromium or Edge found. Install one, or set CHROME=/path/to/browser.")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
BOXES_JS = r"""
|
|
46
|
+
(() => {
|
|
47
|
+
const sel = 'h1,h2,h3,header,section,article,figure,svg,table,pre,img,canvas,blockquote,ol,ul,p,details,' +
|
|
48
|
+
'[class*=slide],[class*=card],[class*=tile],[class*=callout],[class*=seg],[class*=diagram],[class*=box],' +
|
|
49
|
+
'aside,[class*=good],[class*=warn],[class*=note],[class*=tip],[class*=info],[class*=alert],[class*=panel]';
|
|
50
|
+
const seen = new Set(), out = [];
|
|
51
|
+
for (const el of document.querySelectorAll(sel)) {
|
|
52
|
+
if (el.closest('svg') && el.tagName.toLowerCase() !== 'svg') continue;
|
|
53
|
+
const r = el.getBoundingClientRect();
|
|
54
|
+
if (r.width < 80 || r.height < 24) continue;
|
|
55
|
+
const st = getComputedStyle(el);
|
|
56
|
+
if (st.visibility === 'hidden' || st.display === 'none' || +st.opacity === 0) continue;
|
|
57
|
+
const x = Math.round(r.left + scrollX), y = Math.round(r.top + scrollY);
|
|
58
|
+
const key = [x, y, Math.round(r.width), Math.round(r.height)].join(',');
|
|
59
|
+
if (seen.has(key)) continue;
|
|
60
|
+
seen.add(key);
|
|
61
|
+
let depth = 0; for (let p = el.parentElement; p; p = p.parentElement) depth++;
|
|
62
|
+
const text = (el.innerText || el.textContent || '').replace(/\s+/g, ' ').trim().slice(0, 140);
|
|
63
|
+
out.push({tag: el.tagName.toLowerCase(), cls: (el.getAttribute('class') || '').slice(0, 60),
|
|
64
|
+
rect: [x, y, Math.round(r.width), Math.round(r.height)], depth, text});
|
|
65
|
+
if (out.length >= 600) break;
|
|
66
|
+
}
|
|
67
|
+
return out;
|
|
68
|
+
})()
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
FREEZE_JS = r"""
|
|
72
|
+
(() => {
|
|
73
|
+
const target = location.hash ? document.getElementById(decodeURIComponent(location.hash.slice(1))) : null;
|
|
74
|
+
if (target) {
|
|
75
|
+
if (target.tagName === 'DETAILS') target.open = true;
|
|
76
|
+
for (let d = target.closest('details'); d; d = d.parentElement && d.parentElement.closest('details')) d.open = true;
|
|
77
|
+
}
|
|
78
|
+
const s = document.createElement('style');
|
|
79
|
+
s.textContent = '*,*::before,*::after{animation:none!important;transition:none!important;caret-color:transparent!important}';
|
|
80
|
+
document.head.appendChild(s);
|
|
81
|
+
for (const el of document.querySelectorAll('*')) {
|
|
82
|
+
const p = getComputedStyle(el).position;
|
|
83
|
+
if (p === 'fixed') el.style.setProperty('display', 'none', 'important');
|
|
84
|
+
else if (p === 'sticky') el.style.position = 'static';
|
|
85
|
+
}
|
|
86
|
+
return true;
|
|
87
|
+
})()
|
|
88
|
+
"""
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# ---------- capture (Chrome over the DevTools protocol) ----------
|
|
92
|
+
|
|
93
|
+
class CDP:
|
|
94
|
+
def __init__(self, ws):
|
|
95
|
+
self.ws, self.n, self.events = ws, 0, []
|
|
96
|
+
|
|
97
|
+
async def call(self, method, **params):
|
|
98
|
+
self.n += 1
|
|
99
|
+
mid = self.n
|
|
100
|
+
await self.ws.send(json.dumps({"id": mid, "method": method, "params": params}))
|
|
101
|
+
while True:
|
|
102
|
+
msg = json.loads(await self.ws.recv())
|
|
103
|
+
if msg.get("id") == mid:
|
|
104
|
+
if "error" in msg:
|
|
105
|
+
raise RuntimeError(f"{method}: {msg['error']}")
|
|
106
|
+
return msg.get("result", {})
|
|
107
|
+
self.events.append(msg)
|
|
108
|
+
|
|
109
|
+
async def wait_event(self, name, timeout=30):
|
|
110
|
+
end = time.time() + timeout
|
|
111
|
+
while time.time() < end:
|
|
112
|
+
if any(e.get("method") == name for e in self.events):
|
|
113
|
+
return
|
|
114
|
+
try:
|
|
115
|
+
msg = json.loads(await asyncio.wait_for(self.ws.recv(), timeout=1))
|
|
116
|
+
self.events.append(msg)
|
|
117
|
+
except asyncio.TimeoutError:
|
|
118
|
+
pass
|
|
119
|
+
raise TimeoutError(name)
|
|
120
|
+
|
|
121
|
+
async def js(self, expr):
|
|
122
|
+
r = await self.call("Runtime.evaluate", expression=expr, returnByValue=True, awaitPromise=True)
|
|
123
|
+
return r.get("result", {}).get("value")
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _free_port():
|
|
127
|
+
with socket.socket() as s:
|
|
128
|
+
s.bind(("127.0.0.1", 0))
|
|
129
|
+
return s.getsockname()[1]
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
async def _capture(url, width, scale):
|
|
133
|
+
import websockets
|
|
134
|
+
port = _free_port()
|
|
135
|
+
profile = tempfile.mkdtemp(prefix="walkthrough-chrome-")
|
|
136
|
+
proc = subprocess.Popen([find_chrome(), "--headless=new", f"--remote-debugging-port={port}", f"--user-data-dir={profile}",
|
|
137
|
+
"--hide-scrollbars", "--no-first-run", "--no-default-browser-check", "about:blank"],
|
|
138
|
+
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
|
139
|
+
try:
|
|
140
|
+
ws_url = None
|
|
141
|
+
for _ in range(100):
|
|
142
|
+
try:
|
|
143
|
+
targets = json.load(urllib.request.urlopen(f"http://127.0.0.1:{port}/json/list", timeout=1))
|
|
144
|
+
ws_url = next(t["webSocketDebuggerUrl"] for t in targets if t.get("type") == "page")
|
|
145
|
+
break
|
|
146
|
+
except Exception:
|
|
147
|
+
time.sleep(0.1)
|
|
148
|
+
if not ws_url:
|
|
149
|
+
sys.exit("The browser did not start. Set CHROME=/path/to/browser.")
|
|
150
|
+
async with websockets.connect(ws_url, max_size=None) as ws:
|
|
151
|
+
c = CDP(ws)
|
|
152
|
+
await c.call("Page.enable")
|
|
153
|
+
await c.call("Emulation.setDeviceMetricsOverride", width=width, height=1200,
|
|
154
|
+
deviceScaleFactor=scale, mobile=False)
|
|
155
|
+
await c.call("Page.navigate", url=url)
|
|
156
|
+
await c.wait_event("Page.loadEventFired")
|
|
157
|
+
for _ in range(30):
|
|
158
|
+
title = await c.js("document.title + ' ' + (document.body ? document.body.innerText.slice(0, 200) : '')")
|
|
159
|
+
if not re.search(r"just a moment|checking your browser|attention required|verify you are human",
|
|
160
|
+
title or "", re.I):
|
|
161
|
+
break
|
|
162
|
+
await asyncio.sleep(1)
|
|
163
|
+
else:
|
|
164
|
+
print("warning: the page still shows a bot check; capture a saved copy instead", file=sys.stderr)
|
|
165
|
+
await c.js("document.fonts ? document.fonts.ready.then(() => true) : true")
|
|
166
|
+
await asyncio.sleep(0.8)
|
|
167
|
+
await c.js(FREEZE_JS)
|
|
168
|
+
await asyncio.sleep(0.3)
|
|
169
|
+
height = int(await c.js("Math.ceil(document.documentElement.scrollHeight)"))
|
|
170
|
+
boxes = [b for b in await c.js(BOXES_JS) if b["rect"][1] + b["rect"][3] <= height + 2]
|
|
171
|
+
text = await c.js("document.body.innerText")
|
|
172
|
+
title = await c.js("document.title")
|
|
173
|
+
bg = await c.js("getComputedStyle(document.body).backgroundColor")
|
|
174
|
+
await c.call("Emulation.setDeviceMetricsOverride", width=width, height=min(height, 8000),
|
|
175
|
+
deviceScaleFactor=scale, mobile=False)
|
|
176
|
+
await asyncio.sleep(0.3)
|
|
177
|
+
page = Image.new("RGB", (width * scale, height * scale), "white")
|
|
178
|
+
tile = 3000
|
|
179
|
+
for y in range(0, height, tile):
|
|
180
|
+
h = min(tile, height - y)
|
|
181
|
+
shot = await c.call("Page.captureScreenshot", format="png", captureBeyondViewport=True,
|
|
182
|
+
clip={"x": 0, "y": y, "width": width, "height": h, "scale": 1})
|
|
183
|
+
im = Image.open(io.BytesIO(base64.b64decode(shot["data"]))).convert("RGB")
|
|
184
|
+
page.paste(im, (0, y * scale))
|
|
185
|
+
try:
|
|
186
|
+
await c.call("Browser.close")
|
|
187
|
+
except Exception:
|
|
188
|
+
pass
|
|
189
|
+
finally:
|
|
190
|
+
proc.terminate()
|
|
191
|
+
try:
|
|
192
|
+
proc.wait(timeout=5)
|
|
193
|
+
except subprocess.TimeoutExpired:
|
|
194
|
+
proc.kill()
|
|
195
|
+
shutil.rmtree(profile, ignore_errors=True)
|
|
196
|
+
|
|
197
|
+
meta = {"url": url, "title": title, "width": width, "height": height, "scale": scale, "bg": bg, "kind": "html"}
|
|
198
|
+
return page, text, meta, boxes
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
PDF_GAP = 24
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _profile(mask, box, axis):
|
|
205
|
+
"""Which rows (axis=0) or columns (axis=1) of box contain ink."""
|
|
206
|
+
x0, y0, x1, y1 = box
|
|
207
|
+
region = mask.crop(box)
|
|
208
|
+
size = (1, y1 - y0) if axis == 0 else (x1 - x0, 1)
|
|
209
|
+
return [v > 0 for v in region.resize(size, Image.Resampling.BOX).getdata()]
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _runs(ink, min_gap):
|
|
213
|
+
"""Spans of ink separated by at least min_gap empty cells."""
|
|
214
|
+
spans, start, gap = [], None, 0
|
|
215
|
+
for i, on in enumerate(ink):
|
|
216
|
+
if on:
|
|
217
|
+
if start is None:
|
|
218
|
+
start = i
|
|
219
|
+
gap = 0
|
|
220
|
+
end = i + 1
|
|
221
|
+
elif start is not None:
|
|
222
|
+
gap += 1
|
|
223
|
+
if gap >= min_gap:
|
|
224
|
+
spans.append((start, end))
|
|
225
|
+
start = None
|
|
226
|
+
if start is not None:
|
|
227
|
+
spans.append((start, end))
|
|
228
|
+
return spans
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _xy_cut(mask, box, depth, out, gap_y=10, gap_x=14):
|
|
232
|
+
"""Recursive whitespace cut: bands of content, then columns inside a band. Records every node."""
|
|
233
|
+
x0, y0, x1, y1 = box
|
|
234
|
+
rows = _runs(_profile(mask, box, 0), 1)
|
|
235
|
+
cols = _runs(_profile(mask, box, 1), 1)
|
|
236
|
+
if not rows or not cols:
|
|
237
|
+
return
|
|
238
|
+
box = x0 + cols[0][0], y0 + rows[0][0], x0 + cols[-1][1], y0 + rows[-1][1]
|
|
239
|
+
x0, y0, x1, y1 = box
|
|
240
|
+
if x1 - x0 >= 40 and y1 - y0 >= 12:
|
|
241
|
+
out.append((box, depth))
|
|
242
|
+
if depth >= 4:
|
|
243
|
+
return
|
|
244
|
+
bands = _runs(_profile(mask, box, 0), gap_y)
|
|
245
|
+
if len(bands) > 1:
|
|
246
|
+
for a, b in bands:
|
|
247
|
+
_xy_cut(mask, (x0, y0 + a, x1, y0 + b), depth + 1, out, gap_y, gap_x)
|
|
248
|
+
return
|
|
249
|
+
columns = _runs(_profile(mask, box, 1), gap_x)
|
|
250
|
+
if len(columns) > 1:
|
|
251
|
+
for a, b in columns:
|
|
252
|
+
_xy_cut(mask, (x0 + a, y0, x0 + b, y1), depth + 1, out, gap_y, gap_x)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _undouble(word):
|
|
256
|
+
"""Outlined SVG text often extracts with every glyph twice ("TTHHEE"); collapse such words."""
|
|
257
|
+
if len(word) >= 4 and len(word) % 2 == 0 and all(word[i] == word[i + 1] for i in range(0, len(word), 2)):
|
|
258
|
+
return word[::2]
|
|
259
|
+
return word
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _ink_mask(img, bg):
|
|
263
|
+
"""White where the page differs from its background colour."""
|
|
264
|
+
from PIL import ImageChops
|
|
265
|
+
diff = ImageChops.difference(img, Image.new("RGB", img.size, bg)).convert("L")
|
|
266
|
+
return diff.point(lambda v: 255 if v > 18 else 0)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def capture_pdf(path, out, scale, name=None):
|
|
270
|
+
import pypdfium2 as pdfium
|
|
271
|
+
pdf = pdfium.PdfDocument(str(path))
|
|
272
|
+
sizes = [pdf[i].get_size() for i in range(len(pdf))]
|
|
273
|
+
width = int(max(s[0] for s in sizes)) + 2 * PDF_GAP
|
|
274
|
+
height = int(sum(s[1] for s in sizes) + PDF_GAP * (len(sizes) + 1))
|
|
275
|
+
sheet = Image.new("RGB", (width * scale, height * scale), (232, 232, 235))
|
|
276
|
+
boxes, texts, y = [], [], PDF_GAP
|
|
277
|
+
for i in range(len(pdf)):
|
|
278
|
+
page = pdf[i]
|
|
279
|
+
pw, ph = sizes[i]
|
|
280
|
+
x = (width - pw) / 2
|
|
281
|
+
img = page.render(scale=scale).to_pil().convert("RGB")
|
|
282
|
+
sheet.paste(img, (int(x * scale), int(y * scale)))
|
|
283
|
+
tp = page.get_textpage()
|
|
284
|
+
text = tp.get_text_range()
|
|
285
|
+
texts.append(f"--- page {i + 1} ---\n{text}")
|
|
286
|
+
boxes.append({"tag": "page", "cls": f"page {i + 1}", "rect": [round(x), round(y), round(pw), round(ph)],
|
|
287
|
+
"depth": 0, "text": " ".join(text.split())[:140]})
|
|
288
|
+
small = img.resize((round(pw), round(ph)), Image.Resampling.BOX)
|
|
289
|
+
bg = max(small.getcolors(small.width * small.height), key=lambda c: c[0])[1]
|
|
290
|
+
nodes = []
|
|
291
|
+
_xy_cut(_ink_mask(small, bg), (0, 0, small.width, small.height), 1, nodes)
|
|
292
|
+
seen = set()
|
|
293
|
+
for (bx0, by0, bx1, by1), depth in nodes:
|
|
294
|
+
if (bx0, by0, bx1, by1) in seen or (bx1 - bx0) * (by1 - by0) > 0.9 * pw * ph:
|
|
295
|
+
continue
|
|
296
|
+
seen.add((bx0, by0, bx1, by1))
|
|
297
|
+
bt = " ".join(_undouble(w) for w in tp.get_text_bounded(bx0, ph - by1, bx1, ph - by0).split())
|
|
298
|
+
boxes.append({"tag": "region", "cls": f"page {i + 1}", "depth": depth, "text": bt[:140],
|
|
299
|
+
"rect": [round(x + bx0), round(y + by0), bx1 - bx0, by1 - by0]})
|
|
300
|
+
y += ph + PDF_GAP
|
|
301
|
+
title = (pdf.get_metadata_dict().get("Title") or "").strip() or re.sub(r"[-_]+", " ", name or Path(path).stem)
|
|
302
|
+
meta = {"url": Path(path).resolve().as_uri(), "title": title, "width": width, "height": height,
|
|
303
|
+
"scale": scale, "bg": "rgb(232, 232, 235)", "kind": "pdf", "pages": len(pdf)}
|
|
304
|
+
return sheet, "\n\n".join(texts), meta, boxes
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _remote_pdf(url):
|
|
308
|
+
"""Download url to a temp file if it serves a PDF (arXiv's /pdf/ links have no .pdf suffix), else None."""
|
|
309
|
+
req = urllib.request.Request(url, headers={"User-Agent": "talkthrough"})
|
|
310
|
+
with urllib.request.urlopen(req, timeout=60) as r:
|
|
311
|
+
if "pdf" not in (r.headers.get("Content-Type") or "").lower() and not url.lower().split("?")[0].endswith(".pdf"):
|
|
312
|
+
return None
|
|
313
|
+
fd, tmp = tempfile.mkstemp(suffix=".pdf", prefix="talkthrough-")
|
|
314
|
+
with os.fdopen(fd, "wb") as f:
|
|
315
|
+
shutil.copyfileobj(r, f)
|
|
316
|
+
return Path(tmp)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def capture(src, out, width=820, scale=2):
|
|
320
|
+
out = Path(out)
|
|
321
|
+
src = str(src)
|
|
322
|
+
if not re.match(r"^(https?|file):", src) and not Path(src).exists():
|
|
323
|
+
sys.exit(f"no such file: {src}")
|
|
324
|
+
if src.startswith("file:") and not Path(urllib.request.url2pathname(src[5:].split("#")[0])).exists():
|
|
325
|
+
sys.exit(f"no such file: {src}")
|
|
326
|
+
remote_pdf = _remote_pdf(src) if re.match(r"^https?:", src) else None
|
|
327
|
+
if remote_pdf:
|
|
328
|
+
name = urllib.parse.unquote(src.split("?")[0].rstrip("/").rsplit("/", 1)[-1])
|
|
329
|
+
page, text, meta, boxes = capture_pdf(remote_pdf, out, scale, name=re.sub(r"\.pdf$", "", name, flags=re.I))
|
|
330
|
+
meta["url"] = src
|
|
331
|
+
remote_pdf.unlink()
|
|
332
|
+
elif src.lower().endswith(".pdf") and not src.startswith("file:"):
|
|
333
|
+
page, text, meta, boxes = capture_pdf(src, out, scale)
|
|
334
|
+
else:
|
|
335
|
+
url = src if re.match(r"^(https?|file):", src) else Path(src).resolve().as_uri()
|
|
336
|
+
page, text, meta, boxes = asyncio.run(_capture(url, width, scale))
|
|
337
|
+
for i, b in enumerate(boxes):
|
|
338
|
+
b["id"] = f"b{i}"
|
|
339
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
340
|
+
page.save(out / "page.png")
|
|
341
|
+
(out / "page.txt").write_text(text or "")
|
|
342
|
+
(out / "page.json").write_text(json.dumps(meta, indent=1))
|
|
343
|
+
(out / "boxes.json").write_text(json.dumps(boxes, indent=1, ensure_ascii=False))
|
|
344
|
+
return meta, boxes
|
talkthrough/cli.py
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
"""talkthrough: turn an HTML page or a PDF into a narrated walkthrough video of itself."""
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import platform
|
|
6
|
+
import plistlib
|
|
7
|
+
import shutil
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
import tempfile
|
|
11
|
+
import webbrowser
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from . import __version__
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _job_path(folder):
|
|
18
|
+
return Path(folder) / "shots.json"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def cmd_capture(a):
|
|
22
|
+
from .capture import capture
|
|
23
|
+
meta, boxes = capture(a.page, a.out, a.width, a.scale)
|
|
24
|
+
print(f"captured {meta['title']!r}: {meta['width']}x{meta['height']} css px at {meta['scale']}x, "
|
|
25
|
+
f"{len(boxes)} boxes -> {a.out}")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _need_planner():
|
|
29
|
+
try:
|
|
30
|
+
import anthropic # noqa: F401
|
|
31
|
+
except ImportError:
|
|
32
|
+
sys.exit("planning needs the anthropic package: pip install "
|
|
33
|
+
"'talkthrough[plan] @ git+https://github.com/amishah1998/talkthrough'")
|
|
34
|
+
if not (os.environ.get("ANTHROPIC_API_KEY") or os.environ.get("ANTHROPIC_AUTH_TOKEN")):
|
|
35
|
+
sys.exit("planning needs ANTHROPIC_API_KEY. Without one, ask Claude Code for the video (the plugin plans "
|
|
36
|
+
"it for you), or run capture, write shots.json yourself, then preview and render.")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def cmd_plan(a):
|
|
40
|
+
_need_planner()
|
|
41
|
+
from .plan import plan
|
|
42
|
+
shots = plan(a.dir, seconds=a.seconds, model=a.model, extra=a.note or "")
|
|
43
|
+
job = {"format": a.format, "tts": a.tts, "shots": shots}
|
|
44
|
+
_job_path(a.dir).write_text(json.dumps(job, indent=1, ensure_ascii=False))
|
|
45
|
+
print(f"{len(shots)} shots -> {_job_path(a.dir)}")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def cmd_preview(a):
|
|
49
|
+
from .render import preview
|
|
50
|
+
preview(a.dir)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _mac_default_browser():
|
|
54
|
+
"""Bundle id of the app that handles https links, or None if the user never changed it (Safari)."""
|
|
55
|
+
prefs = Path.home() / "Library/Preferences/com.apple.LaunchServices/com.apple.launchservices.secure.plist"
|
|
56
|
+
try:
|
|
57
|
+
handlers = plistlib.loads(prefs.read_bytes()).get("LSHandlers", [])
|
|
58
|
+
except (OSError, plistlib.InvalidFileException):
|
|
59
|
+
return None
|
|
60
|
+
for scheme in ("https", "http"):
|
|
61
|
+
for h in handlers:
|
|
62
|
+
if h.get("LSHandlerURLScheme") == scheme and h.get("LSHandlerRoleAll"):
|
|
63
|
+
return h["LSHandlerRoleAll"]
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def open_in_browser(video):
|
|
68
|
+
"""Play the finished video in the default web browser.
|
|
69
|
+
|
|
70
|
+
On macOS a plain open hands an .mp4 to QuickTime, so the browser is looked up and named explicitly."""
|
|
71
|
+
video = Path(video).resolve()
|
|
72
|
+
if platform.system() == "Darwin":
|
|
73
|
+
bundle = _mac_default_browser()
|
|
74
|
+
args = ["open", "-b", bundle, str(video)] if bundle else ["open", "-a", "Safari", str(video)]
|
|
75
|
+
if subprocess.run(args, capture_output=True).returncode == 0:
|
|
76
|
+
return
|
|
77
|
+
if not webbrowser.open(video.as_uri()):
|
|
78
|
+
print(f"could not open a browser; the video is at {video}", file=sys.stderr)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def cmd_render(a):
|
|
82
|
+
from .render import render
|
|
83
|
+
out = render(a.dir, a.out, quality="share" if a.share else None, speed=a.speed)
|
|
84
|
+
if a.open:
|
|
85
|
+
open_in_browser(out)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def cmd_make(a):
|
|
89
|
+
_need_planner()
|
|
90
|
+
from .capture import capture
|
|
91
|
+
from .plan import plan
|
|
92
|
+
from .render import preview, render
|
|
93
|
+
folder = Path(a.keep) if a.keep else Path(tempfile.mkdtemp(prefix="talkthrough-"))
|
|
94
|
+
capture(a.page, folder, a.width, a.scale)
|
|
95
|
+
shots = plan(folder, seconds=a.seconds, model=a.model, extra=a.note or "")
|
|
96
|
+
_job_path(folder).write_text(json.dumps({"format": a.format, "tts": a.tts, "shots": shots}, indent=1,
|
|
97
|
+
ensure_ascii=False))
|
|
98
|
+
preview(folder)
|
|
99
|
+
out = Path(a.out) if a.out else Path.cwd() / (Path(a.page).stem + "-walkthrough.mp4")
|
|
100
|
+
render(folder, out, quality="share" if a.share else None, speed=a.speed)
|
|
101
|
+
if not a.keep:
|
|
102
|
+
shutil.rmtree(folder, ignore_errors=True)
|
|
103
|
+
if a.open:
|
|
104
|
+
open_in_browser(out)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def cmd_doctor(a):
|
|
108
|
+
from .capture import find_chrome
|
|
109
|
+
ok = True
|
|
110
|
+
|
|
111
|
+
def row(name, good, note):
|
|
112
|
+
nonlocal ok
|
|
113
|
+
ok &= good or name.startswith("(optional)")
|
|
114
|
+
print(f" {'ok ' if good else '-- '} {name}: {note}")
|
|
115
|
+
|
|
116
|
+
print(f"talkthrough {__version__} on {platform.system()}")
|
|
117
|
+
try:
|
|
118
|
+
row("browser", True, find_chrome())
|
|
119
|
+
except SystemExit as e:
|
|
120
|
+
row("browser", False, str(e))
|
|
121
|
+
row("ffmpeg", bool(shutil.which("ffmpeg")), shutil.which("ffmpeg") or "install ffmpeg")
|
|
122
|
+
row("ffprobe", bool(shutil.which("ffprobe")), shutil.which("ffprobe") or "comes with ffmpeg")
|
|
123
|
+
from .voice import ORDER, kokoro_available
|
|
124
|
+
voices = [name for name, env in ORDER if os.environ.get(env)]
|
|
125
|
+
if kokoro_available():
|
|
126
|
+
voices.append("kokoro (local)")
|
|
127
|
+
if platform.system() == "Darwin" and shutil.which("say"):
|
|
128
|
+
voices.append("say (macOS)")
|
|
129
|
+
row("voice", bool(voices), ", ".join(voices) or "none: set CARTESIA_API_KEY, ELEVENLABS_API_KEY or OPENAI_API_KEY, "
|
|
130
|
+
"or pip install 'talkthrough[local] @ git+https://github.com/amishah1998/talkthrough'")
|
|
131
|
+
try:
|
|
132
|
+
import anthropic # noqa: F401
|
|
133
|
+
has_sdk = True
|
|
134
|
+
except ImportError:
|
|
135
|
+
has_sdk = False
|
|
136
|
+
auto = has_sdk and bool(os.environ.get("ANTHROPIC_API_KEY") or os.environ.get("ANTHROPIC_AUTH_TOKEN"))
|
|
137
|
+
row("(optional) auto-plan", auto, "ready" if auto else
|
|
138
|
+
"needs pip install 'talkthrough[plan] @ git+https://github.com/amishah1998/talkthrough' and ANTHROPIC_API_KEY; not needed when an agent writes shots.json")
|
|
139
|
+
sys.exit(0 if ok else 1)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def main(argv=None):
|
|
143
|
+
ap = argparse.ArgumentParser(prog="talkthrough", description=__doc__)
|
|
144
|
+
ap.add_argument("--version", action="version", version=__version__)
|
|
145
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
146
|
+
|
|
147
|
+
def capture_args(p):
|
|
148
|
+
p.add_argument("--width", type=int, default=820, help="viewport width in css px for HTML (default 820)")
|
|
149
|
+
p.add_argument("--scale", type=int, default=2, help="capture pixel density (default 2)")
|
|
150
|
+
|
|
151
|
+
def plan_args(p):
|
|
152
|
+
p.add_argument("--seconds", type=int, default=90, help="target length (default 90)")
|
|
153
|
+
p.add_argument("--format", default="portrait", choices=["portrait", "landscape", "square"])
|
|
154
|
+
p.add_argument("--tts", default="auto", choices=["auto", "cartesia", "elevenlabs", "openai", "kokoro", "say"])
|
|
155
|
+
p.add_argument("--model", default="claude-opus-5", help="Claude model that plans the shots")
|
|
156
|
+
p.add_argument("--note", help="extra direction for the planner, e.g. 'focus on the pricing table'")
|
|
157
|
+
|
|
158
|
+
p = sub.add_parser("capture", help="screenshot the page and list its sections")
|
|
159
|
+
p.add_argument("page", help=".html, .pdf, or an http(s) URL")
|
|
160
|
+
p.add_argument("--out", required=True, help="work folder")
|
|
161
|
+
capture_args(p)
|
|
162
|
+
p.set_defaults(fn=cmd_capture)
|
|
163
|
+
|
|
164
|
+
p = sub.add_parser("plan", help="let Claude write shots.json for a captured page")
|
|
165
|
+
p.add_argument("dir")
|
|
166
|
+
plan_args(p)
|
|
167
|
+
p.set_defaults(fn=cmd_plan)
|
|
168
|
+
|
|
169
|
+
p = sub.add_parser("preview", help="one spotlighted frame per shot, as preview.png")
|
|
170
|
+
p.add_argument("dir")
|
|
171
|
+
p.set_defaults(fn=cmd_preview)
|
|
172
|
+
|
|
173
|
+
p = sub.add_parser("render", help="voice, camera and captions into an MP4")
|
|
174
|
+
p.add_argument("dir")
|
|
175
|
+
p.add_argument("--out")
|
|
176
|
+
p.add_argument("--share", action="store_true", help="smaller file for chat apps (about half the size)")
|
|
177
|
+
p.add_argument("--speed", type=float, help="voice pace, e.g. 1.25 or 1.5 (default 1.0)")
|
|
178
|
+
p.add_argument("--open", action="store_true", help="play the video in your web browser when it is done")
|
|
179
|
+
p.set_defaults(fn=cmd_render)
|
|
180
|
+
|
|
181
|
+
p = sub.add_parser("make", help="capture, plan, preview and render in one go")
|
|
182
|
+
p.add_argument("page")
|
|
183
|
+
p.add_argument("--out", help="output .mp4 (default: <page>-walkthrough.mp4)")
|
|
184
|
+
p.add_argument("--keep", help="keep the work folder here")
|
|
185
|
+
p.add_argument("--share", action="store_true", help="smaller file for chat apps (about half the size)")
|
|
186
|
+
p.add_argument("--speed", type=float, help="voice pace, e.g. 1.25 or 1.5 (default 1.0)")
|
|
187
|
+
p.add_argument("--open", action="store_true", help="play the video in your web browser when it is done")
|
|
188
|
+
capture_args(p)
|
|
189
|
+
plan_args(p)
|
|
190
|
+
p.set_defaults(fn=cmd_make)
|
|
191
|
+
|
|
192
|
+
p = sub.add_parser("doctor", help="check what is installed")
|
|
193
|
+
p.set_defaults(fn=cmd_doctor)
|
|
194
|
+
|
|
195
|
+
a = ap.parse_args(argv)
|
|
196
|
+
a.fn(a)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
if __name__ == "__main__":
|
|
200
|
+
main()
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""The free local voice: Kokoro-82M (Apache-2.0 weights) through kokoro-onnx. No torch, no GPU.
|
|
2
|
+
The model is downloaded once into the user cache folder."""
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
import urllib.request
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
RELEASE = "https://github.com/thewh1teagle/kokoro-onnx/releases/download/model-files-v1.0/"
|
|
9
|
+
FILES = {"model": "kokoro-v1.0.onnx", "voices": "voices-v1.0.bin"}
|
|
10
|
+
_engine = None
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def cache_dir():
|
|
14
|
+
base = os.environ.get("TALKTHROUGH_CACHE") or os.environ.get("XDG_CACHE_HOME") or Path.home() / ".cache"
|
|
15
|
+
d = Path(base) / "talkthrough" / "kokoro"
|
|
16
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
17
|
+
return d
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _fetch(name):
|
|
21
|
+
path = cache_dir() / name
|
|
22
|
+
if path.exists() and path.stat().st_size > 0:
|
|
23
|
+
return path
|
|
24
|
+
print(f"downloading {name} (one time) ...", file=sys.stderr, flush=True)
|
|
25
|
+
tmp = path.with_suffix(".part")
|
|
26
|
+
urllib.request.urlretrieve(RELEASE + name, tmp)
|
|
27
|
+
tmp.rename(path)
|
|
28
|
+
return path
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def engine():
|
|
32
|
+
global _engine
|
|
33
|
+
if _engine is None:
|
|
34
|
+
from kokoro_onnx import Kokoro
|
|
35
|
+
_engine = Kokoro(str(_fetch(FILES["model"])), str(_fetch(FILES["voices"])))
|
|
36
|
+
return _engine
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def speak(text, voice, speed, wav, sample_rate):
|
|
40
|
+
import soundfile as sf
|
|
41
|
+
samples, sr = engine().create(text, voice=voice, speed=speed, lang="en-us")
|
|
42
|
+
tmp = Path(wav).with_suffix(".raw.wav")
|
|
43
|
+
sf.write(tmp, samples, sr)
|
|
44
|
+
from .voice import _ffmpeg
|
|
45
|
+
_ffmpeg("-i", tmp, "-ar", sample_rate, "-ac", 1, wav)
|
|
46
|
+
tmp.unlink()
|