screengraft 0.17.0 → 0.20.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -6
- package/package.json +5 -2
- package/scripts/grade.py +53 -26
- package/scripts/preflight.py +48 -0
- package/scripts/requirements.txt +1 -0
- package/scripts/ui.py +146 -5
- package/scripts/warp.py +288 -71
- package/skills/inject-screenshot/SKILL.md +27 -10
- package/ui/index.html +138 -78
package/scripts/warp.py
CHANGED
|
@@ -28,6 +28,8 @@ screen as it appears in the photo — order matters, it defines the mapping).
|
|
|
28
28
|
|
|
29
29
|
import argparse
|
|
30
30
|
import json
|
|
31
|
+
import os
|
|
32
|
+
import subprocess
|
|
31
33
|
import sys
|
|
32
34
|
|
|
33
35
|
import cv2
|
|
@@ -116,6 +118,140 @@ def shoelace_area(pts: np.ndarray) -> float:
|
|
|
116
118
|
return 0.5 * abs(np.dot(x, np.roll(y, 1)) - np.dot(y, np.roll(x, 1)))
|
|
117
119
|
|
|
118
120
|
|
|
121
|
+
class Plan:
|
|
122
|
+
"""Everything about a fit that does NOT change from frame to frame.
|
|
123
|
+
|
|
124
|
+
Built once for a still, once per clip for a video. The split exists so the
|
|
125
|
+
two paths cannot drift: `compose()` is a Plan plus one frame, and
|
|
126
|
+
`compose_video()` is the same Plan plus N frames, which makes "frame 0 of a
|
|
127
|
+
render equals the still composite, byte for byte" a property the tests can
|
|
128
|
+
assert rather than a thing we hope stays true.
|
|
129
|
+
|
|
130
|
+
What lives here is what a fixed photo and a fixed quad make constant:
|
|
131
|
+
|
|
132
|
+
- the prefilter's target size (the screenshot/frame is always the same
|
|
133
|
+
size, and the quad never moves, so the minification factor is fixed);
|
|
134
|
+
- the homography, which is derived from that rescaled size;
|
|
135
|
+
- the rounded source mask;
|
|
136
|
+
- the warped, antialiased destination mask, which is by far the most
|
|
137
|
+
expensive thing in compose() because it supersamples MASK_SS x over the
|
|
138
|
+
quad's bbox. Computing it once is most of the speed of a video render,
|
|
139
|
+
and it also means the screen's EDGE is pixel-identical in every frame,
|
|
140
|
+
so there is no edge crawl — the artefact that makes a composite read as
|
|
141
|
+
fake. A fixed photo is the one case where that comes for free;
|
|
142
|
+
- the grain sigma, measured from the photo, which does not change either.
|
|
143
|
+
|
|
144
|
+
The grade parameters are deliberately NOT built here: they need a frame to
|
|
145
|
+
measure against, so `bind_grade()` takes the fit frame and stores them.
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
def __init__(self, photo: np.ndarray, frame_shape, corners,
|
|
149
|
+
corner_radius: float = 0.0, grain: bool = False):
|
|
150
|
+
dst_quad = np.array(corners, dtype=np.float32)
|
|
151
|
+
if shoelace_area(dst_quad) < 1.0:
|
|
152
|
+
raise ValueError("degenerate quad (near-zero area) — check corner order TL,TR,BR,BL")
|
|
153
|
+
self.photo = photo
|
|
154
|
+
self.dst_quad = dst_quad
|
|
155
|
+
self.grain = grain
|
|
156
|
+
|
|
157
|
+
top = float(np.linalg.norm(dst_quad[1] - dst_quad[0]))
|
|
158
|
+
bottom = float(np.linalg.norm(dst_quad[2] - dst_quad[3]))
|
|
159
|
+
left = float(np.linalg.norm(dst_quad[3] - dst_quad[0]))
|
|
160
|
+
right = float(np.linalg.norm(dst_quad[2] - dst_quad[1]))
|
|
161
|
+
need_w, need_h = max(top, bottom), max(left, right)
|
|
162
|
+
sh0, sw0 = frame_shape[:2]
|
|
163
|
+
scale_x, scale_y = need_w / sw0, need_h / sh0
|
|
164
|
+
self.prescale = max(scale_x, scale_y)
|
|
165
|
+
if 0 < self.prescale < 0.95:
|
|
166
|
+
self.new_w = max(1, int(round(sw0 * self.prescale)))
|
|
167
|
+
self.new_h = max(1, int(round(sh0 * self.prescale)))
|
|
168
|
+
self.radius = corner_radius * (self.new_w / sw0)
|
|
169
|
+
else:
|
|
170
|
+
self.new_w, self.new_h = sw0, sh0
|
|
171
|
+
self.radius = corner_radius
|
|
172
|
+
self.prescale = 1.0
|
|
173
|
+
|
|
174
|
+
src_rect = np.array([[0, 0], [self.new_w, 0],
|
|
175
|
+
[self.new_w, self.new_h], [0, self.new_h]], dtype=np.float32)
|
|
176
|
+
self.H = cv2.getPerspectiveTransform(src_rect, dst_quad)
|
|
177
|
+
ph, pw = photo.shape[:2]
|
|
178
|
+
self.size = (pw, ph)
|
|
179
|
+
src_mask = rounded_mask(self.new_w, self.new_h, float(self.radius))
|
|
180
|
+
self.warped_mask = _warp_mask_antialiased(src_mask, self.H, pw, ph, dst_quad)
|
|
181
|
+
self.mask3 = cv2.merge([self.warped_mask] * 3).astype(np.float32) / 255.0
|
|
182
|
+
self.grain_sigma = (_grade.measure_grain(photo, _grade.surround_ring(self.warped_mask))
|
|
183
|
+
if grain else 0.0)
|
|
184
|
+
self.grade_params = None
|
|
185
|
+
# Integer bbox of the quad, clamped to the canvas and padded by a pixel
|
|
186
|
+
# so the antialiased edge is never clipped.
|
|
187
|
+
xs, ys = dst_quad[:, 0], dst_quad[:, 1]
|
|
188
|
+
bx0, by0 = max(0, int(np.floor(xs.min())) - 1), max(0, int(np.floor(ys.min())) - 1)
|
|
189
|
+
bx1, by1 = min(pw, int(np.ceil(xs.max())) + 2), min(ph, int(np.ceil(ys.max())) + 2)
|
|
190
|
+
self.bbox = (bx0, by0, bx1, by1) if bx1 > bx0 and by1 > by0 else None
|
|
191
|
+
|
|
192
|
+
def _prep(self, frame: np.ndarray, bbox=None) -> np.ndarray:
|
|
193
|
+
"""Warp one frame. With `bbox`, warp only that window of the canvas.
|
|
194
|
+
|
|
195
|
+
The window is an integer translation of the output grid, folded into the
|
|
196
|
+
homography, so every output pixel resolves to exactly the same source
|
|
197
|
+
coordinate as the full-canvas warp — byte-identical, and asserted as
|
|
198
|
+
such in test_video.py. Outside the quad the mask is zero and the blend
|
|
199
|
+
is the photo copied onto itself, so on a 2400x1792 photo whose screen
|
|
200
|
+
occupies 9% of the frame this is most of the per-frame cost removed.
|
|
201
|
+
"""
|
|
202
|
+
if (self.new_w, self.new_h) != (frame.shape[1], frame.shape[0]):
|
|
203
|
+
frame = cv2.resize(frame, (self.new_w, self.new_h), interpolation=cv2.INTER_AREA)
|
|
204
|
+
H, size = self.H, self.size
|
|
205
|
+
if bbox is not None:
|
|
206
|
+
x0, y0, x1, y1 = bbox
|
|
207
|
+
T = np.array([[1, 0, -x0], [0, 1, -y0], [0, 0, 1]], dtype=np.float64)
|
|
208
|
+
H, size = T @ H, (x1 - x0, y1 - y0)
|
|
209
|
+
return cv2.warpPerspective(frame, H, size,
|
|
210
|
+
flags=cv2.INTER_LANCZOS4,
|
|
211
|
+
borderMode=cv2.BORDER_REPLICATE)
|
|
212
|
+
|
|
213
|
+
def bind_grade(self, frame: np.ndarray, strength: float) -> None:
|
|
214
|
+
"""Measure the light correction once, from the frame the user fitted on."""
|
|
215
|
+
self.grade_params = _grade.light_params(
|
|
216
|
+
self.photo, self._prep(frame), self.warped_mask, strength) if strength > 0 else None
|
|
217
|
+
|
|
218
|
+
def render(self, frame: np.ndarray, screen_off: np.ndarray = None,
|
|
219
|
+
specular: float = 0.75, fast: bool = False) -> np.ndarray:
|
|
220
|
+
"""Composite one frame onto the photo.
|
|
221
|
+
|
|
222
|
+
`fast` confines the warp and the blend to the quad's bounding box. It is
|
|
223
|
+
off for stills, where a single frame's cost is irrelevant and the
|
|
224
|
+
simplest code is the one to trust, and on for video renders. Both
|
|
225
|
+
produce identical bytes; test_video.py asserts it rather than assuming.
|
|
226
|
+
"""
|
|
227
|
+
if fast and self.bbox is not None:
|
|
228
|
+
x0, y0, x1, y1 = self.bbox
|
|
229
|
+
warped_screen = self._prep(frame, self.bbox)
|
|
230
|
+
if self.grade_params is not None:
|
|
231
|
+
warped_screen = _grade.apply_light(warped_screen, self.grade_params)
|
|
232
|
+
out = self.photo.copy()
|
|
233
|
+
win = self.mask3[y0:y1, x0:x1]
|
|
234
|
+
out[y0:y1, x0:x1] = np.clip(
|
|
235
|
+
self.photo[y0:y1, x0:x1].astype(np.float32) * (1 - win)
|
|
236
|
+
+ warped_screen.astype(np.float32) * win, 0, 255).astype(np.uint8)
|
|
237
|
+
else:
|
|
238
|
+
warped_screen = self._prep(frame)
|
|
239
|
+
if self.grade_params is not None:
|
|
240
|
+
warped_screen = _grade.apply_light(warped_screen, self.grade_params)
|
|
241
|
+
out = (self.photo.astype(np.float32) * (1 - self.mask3)
|
|
242
|
+
+ warped_screen.astype(np.float32) * self.mask3)
|
|
243
|
+
out = np.clip(out, 0, 255).astype(np.uint8)
|
|
244
|
+
if self.grain:
|
|
245
|
+
# Seeded, so the grain is IDENTICAL in every frame. Over a still
|
|
246
|
+
# photograph that is what it must be: the background's own noise is
|
|
247
|
+
# frozen, and grain that crawled on the screen alone would read as a
|
|
248
|
+
# dirty window. It also keeps the render deterministic.
|
|
249
|
+
out = _grade.add_grain(out, self.warped_mask, self.grain_sigma)
|
|
250
|
+
if screen_off is not None:
|
|
251
|
+
out = _grade.specular_lift(out, screen_off, self.warped_mask, strength=specular)
|
|
252
|
+
return out
|
|
253
|
+
|
|
254
|
+
|
|
119
255
|
def compose(photo: np.ndarray, screenshot: np.ndarray, corners, corner_radius: float = 0.0,
|
|
120
256
|
grade: float = 0.0, grain: bool = False, screen_off: np.ndarray = None,
|
|
121
257
|
specular: float = 0.75) -> np.ndarray:
|
|
@@ -124,79 +260,160 @@ def compose(photo: np.ndarray, screenshot: np.ndarray, corners, corner_radius: f
|
|
|
124
260
|
Single resampling pass at the photo's resolution; deterministic. This is the
|
|
125
261
|
whole engine — the CLI below and ui.py both call it.
|
|
126
262
|
"""
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
#
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
263
|
+
# A Plan plus one frame. The long-form pipeline this replaced (prefilter,
|
|
264
|
+
# homography, rounded mask, antialiased mask warp, grade, blend, grain,
|
|
265
|
+
# specular) now lives in Plan, so the still and video paths run the SAME
|
|
266
|
+
# code and cannot drift apart. See test_video.py: frame 0 of a render is
|
|
267
|
+
# asserted byte-identical to this function's output.
|
|
268
|
+
plan = Plan(photo, screenshot.shape, corners, corner_radius, grain=grain)
|
|
269
|
+
plan.bind_grade(screenshot, grade)
|
|
270
|
+
return plan.render(screenshot, screen_off=screen_off, specular=specular)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def ffmpeg_exe() -> str:
|
|
274
|
+
"""Path to the ffmpeg binary, or raise with something a designer can act on.
|
|
275
|
+
|
|
276
|
+
imageio_ffmpeg ships a static binary as a wheel, so it installs into the
|
|
277
|
+
same venv OpenCV already lives in and the zero-configuration constraint
|
|
278
|
+
holds — no Homebrew, no PATH hunting. cv2.VideoWriter was rejected for this
|
|
279
|
+
job: the headless wheels carry a limited codec set, expose no control over
|
|
280
|
+
bitrate or pixel format, and behave differently per platform, none of which
|
|
281
|
+
is acceptable when the output is the deliverable.
|
|
282
|
+
"""
|
|
283
|
+
try:
|
|
284
|
+
import imageio_ffmpeg
|
|
285
|
+
return imageio_ffmpeg.get_ffmpeg_exe()
|
|
286
|
+
except Exception as e:
|
|
287
|
+
raise RuntimeError(
|
|
288
|
+
"ffmpeg is missing. Run `python3 scripts/preflight.py --install` to add it "
|
|
289
|
+
"to the screengraft venv (it ships as a wheel; nothing is installed "
|
|
290
|
+
"system-wide)."
|
|
291
|
+
) from e
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
# Output presets. The last stage of the pipeline is the only one that can undo
|
|
295
|
+
# the care taken in all the others: H.264's 4:2:0 chroma subsampling softens
|
|
296
|
+
# exactly the coloured text edges the INTER_AREA prefilter exists to protect.
|
|
297
|
+
# So "web" runs at CRF 16, which is near-visually-lossless rather than
|
|
298
|
+
# delivery-sized, and anything destined for a case study should use prores.
|
|
299
|
+
# BT.709 is tagged explicitly on both so players do not guess at the primaries
|
|
300
|
+
# and shift the colour we just matched to the room.
|
|
301
|
+
PRESETS = {
|
|
302
|
+
"web": ["-c:v", "libx264", "-preset", "slow", "-crf", "16",
|
|
303
|
+
"-pix_fmt", "yuv420p", "-movflags", "+faststart"],
|
|
304
|
+
"prores": ["-c:v", "prores_ks", "-profile:v", "3", "-pix_fmt", "yuv422p10le"],
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def probe_video(path: str):
|
|
309
|
+
"""(frame_count, fps, width, height) — read, never trusted blindly."""
|
|
310
|
+
cap = cv2.VideoCapture(path)
|
|
311
|
+
if not cap.isOpened():
|
|
312
|
+
raise RuntimeError(f"could not open video: {path}")
|
|
313
|
+
fps = float(cap.get(cv2.CAP_PROP_FPS)) or 0.0
|
|
314
|
+
n = int(cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0)
|
|
315
|
+
w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH) or 0)
|
|
316
|
+
h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT) or 0)
|
|
317
|
+
cap.release()
|
|
318
|
+
# CAP_PROP_FRAME_COUNT is a container hint and is wrong often enough that
|
|
319
|
+
# the render loop counts frames as it reads them instead. It is reported
|
|
320
|
+
# here only to drive a progress bar.
|
|
321
|
+
if not (0.1 <= fps <= 240):
|
|
322
|
+
fps = 30.0
|
|
323
|
+
return n, fps, w, h
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def read_frame_at(path: str, index: int = 0):
|
|
327
|
+
"""One frame, for fitting and for the poster image."""
|
|
328
|
+
cap = cv2.VideoCapture(path)
|
|
329
|
+
if not cap.isOpened():
|
|
330
|
+
raise RuntimeError(f"could not open video: {path}")
|
|
331
|
+
if index > 0:
|
|
332
|
+
cap.set(cv2.CAP_PROP_POS_FRAMES, index)
|
|
333
|
+
ok, frame = cap.read()
|
|
334
|
+
if not ok and index > 0:
|
|
335
|
+
# Seeking is unreliable on some containers; fall back to reading
|
|
336
|
+
# forward, which is slow and always right.
|
|
337
|
+
cap.release()
|
|
338
|
+
cap = cv2.VideoCapture(path)
|
|
339
|
+
for _ in range(index + 1):
|
|
340
|
+
ok, frame = cap.read()
|
|
341
|
+
if not ok:
|
|
342
|
+
break
|
|
343
|
+
cap.release()
|
|
344
|
+
if not ok or frame is None:
|
|
345
|
+
raise RuntimeError(f"could not read frame {index} of {path}")
|
|
346
|
+
return frame
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def compose_video(photo: np.ndarray, video_path: str, corners, output: str,
|
|
350
|
+
corner_radius: float = 0.0, grade: float = 0.0, grain: bool = False,
|
|
351
|
+
preset: str = "web", fit_frame: int = 0, audio: bool = True,
|
|
352
|
+
frames_dir: str = None, progress=None) -> dict:
|
|
353
|
+
"""Inject a VIDEO into a still photo. The photo does not move, so there is
|
|
354
|
+
exactly one homography and the whole of Plan is computed once.
|
|
355
|
+
|
|
356
|
+
Frames are piped to ffmpeg as raw BGR24 rather than written out as a PNG
|
|
357
|
+
sequence: a ten-second clip is several hundred frames and several GB of
|
|
358
|
+
intermediate PNGs, for no benefit. `frames_dir` still dumps them when a
|
|
359
|
+
test or a human needs to look at individual frames.
|
|
360
|
+
|
|
361
|
+
Time is deliberately NOT resampled. The output runs at the source's own
|
|
362
|
+
frame rate; converting fps by dropping or duplicating frames is judder, and
|
|
363
|
+
doing it properly means blending, which is a different feature. Note also
|
|
364
|
+
that a prototype recording has no motion blur — it renders discrete frames
|
|
365
|
+
— so a fast scroll will strobe. That is a property of the input, the same
|
|
366
|
+
way the source resolution is, not something this stage should paper over.
|
|
367
|
+
"""
|
|
368
|
+
n_hint, fps, vw, vh = probe_video(video_path)
|
|
369
|
+
first = read_frame_at(video_path, fit_frame)
|
|
370
|
+
plan = Plan(photo, first.shape, corners, corner_radius, grain=grain)
|
|
371
|
+
plan.bind_grade(first, grade)
|
|
162
372
|
|
|
163
|
-
sh, sw = screenshot.shape[:2]
|
|
164
|
-
src_rect = np.array([[0, 0], [sw, 0], [sw, sh], [0, sh]], dtype=np.float32)
|
|
165
|
-
H = cv2.getPerspectiveTransform(src_rect, dst_quad)
|
|
166
373
|
ph, pw = photo.shape[:2]
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
374
|
+
cmd = [ffmpeg_exe(), "-y", "-loglevel", "error",
|
|
375
|
+
"-f", "rawvideo", "-pix_fmt", "bgr24", "-s", f"{pw}x{ph}",
|
|
376
|
+
"-r", f"{fps}", "-i", "-"]
|
|
377
|
+
if audio:
|
|
378
|
+
# Optional by construction: `?` on the map means a source with no audio
|
|
379
|
+
# track (which a prototype recording usually is) is not an error.
|
|
380
|
+
cmd += ["-i", video_path, "-map", "0:v", "-map", "1:a?", "-c:a", "aac", "-shortest"]
|
|
381
|
+
cmd += PRESETS.get(preset, PRESETS["web"])
|
|
382
|
+
cmd += ["-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709",
|
|
383
|
+
output]
|
|
384
|
+
|
|
385
|
+
if frames_dir:
|
|
386
|
+
os.makedirs(frames_dir, exist_ok=True)
|
|
387
|
+
cap = cv2.VideoCapture(video_path)
|
|
388
|
+
proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stderr=subprocess.PIPE)
|
|
389
|
+
count = 0
|
|
390
|
+
try:
|
|
391
|
+
while True:
|
|
392
|
+
ok, frame = cap.read()
|
|
393
|
+
if not ok:
|
|
394
|
+
break
|
|
395
|
+
out = plan.render(frame, fast=True)
|
|
396
|
+
if frames_dir:
|
|
397
|
+
cv2.imwrite(os.path.join(frames_dir, f"{count:06d}.png"), out,
|
|
398
|
+
[cv2.IMWRITE_PNG_COMPRESSION, 1])
|
|
399
|
+
proc.stdin.write(out.tobytes())
|
|
400
|
+
count += 1
|
|
401
|
+
if progress and count % 10 == 0:
|
|
402
|
+
progress(count, n_hint)
|
|
403
|
+
finally:
|
|
404
|
+
cap.release()
|
|
405
|
+
try:
|
|
406
|
+
proc.stdin.close()
|
|
407
|
+
except (BrokenPipeError, OSError):
|
|
408
|
+
pass
|
|
409
|
+
err = proc.stderr.read().decode("utf-8", "replace")
|
|
410
|
+
proc.wait()
|
|
411
|
+
if proc.returncode != 0:
|
|
412
|
+
raise RuntimeError(f"ffmpeg failed ({proc.returncode}): {err.strip()[:400]}")
|
|
413
|
+
if count == 0:
|
|
414
|
+
raise RuntimeError(f"no frames could be read from {video_path}")
|
|
415
|
+
return {"frames": count, "fps": fps, "source_size": [vw, vh],
|
|
416
|
+
"output_size": [pw, ph], "preset": preset, "fit_frame": fit_frame}
|
|
200
417
|
|
|
201
418
|
|
|
202
419
|
def main() -> None:
|
|
@@ -1,17 +1,19 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: inject-screenshot
|
|
3
|
-
description: Injects a UI screenshot onto a photographed device screen (phone/tablet/laptop) at any angle, matching the perspective exactly via homography — geometry, not AI generation. Use when the user wants to composite a screen design into a real device photo for a portfolio, case study, or mockup, and needs the result to look like a genuine photo rather than a template. Opens a local browser UI for picking files, correcting corners and saving.
|
|
3
|
+
description: Injects a UI screenshot OR a screen recording onto a photographed device screen (phone/tablet/laptop) at any angle, matching the perspective exactly via homography — geometry, not AI generation. Use when the user wants to composite a screen design into a real device photo for a portfolio, case study, or mockup, or wants a prototype recording playing on a real device in a photograph, and needs the result to look like a genuine photo rather than a template. Handles video sources (mp4/mov/webm) by fitting once and rendering every frame. Opens a local browser UI for picking files, correcting corners, and saving or rendering.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Inject a screenshot onto a photographed device
|
|
7
7
|
|
|
8
|
-
**What ships (v0.
|
|
8
|
+
**What ships (v0.20):** a local browser UI (`scripts/ui.py`) that walks the designer through the whole job — pick the photo and the screen source, which may be an image **or a video** (recent Desktop/Downloads images, drag-drop, browse, path, or a **Figma frame link**), auto-detect the screen as a starting position, then **match the four edges** (drag an edge's middle to slide it, near an end to pivot; corners still draggable) with canvas zoom/pan and a rectified strip loupe. The fit and the composite sit **side by side and always have** — the result pane re-renders as you drag, which is how a corner gets judged, so it is the layout rather than a mode you can switch off. Then an on-by-default realism pass that colour-matches the source to the photo's light, **Save** (or **Render**, for a video) into the project folder (`--out-dir`), and a **Send to Claude** button that reaches you through the plugin's own MCP server. The UI is a hand port of the project's Figma design file — dark only.
|
|
9
9
|
|
|
10
10
|
The geometry is exact (`warp.py`); the detection is advisory (`detect.py`) and the human corrects it.
|
|
11
11
|
|
|
12
|
-
**The realism pass ships and is ON by default** (`grade.py
|
|
12
|
+
**The realism pass ships and is ON by default** (`grade.py`): it matches the injected screen's white balance and grain to the light around it, at a strength the designer sets in the rail. It can also lift the device's real specular highlights from a screen-off reference frame, though the UI cannot supply one yet. Off is a first-class choice and keeps the screenshot's colour exactly — say so if the user is reviewing brand colour.
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
**Video ships too.** The screen source can be a video (mp4/mov/webm) as well as a still — pick it exactly like a screenshot, choose which frame to match the edges on, and the primary button becomes **Render**. The photo does not move, so there is one homography and every frame gets the same geometry; the light match is measured once from the frame you fitted on, so the screen cannot pulse as the UI scrolls. Output is H.264 at CRF 16 (near-visually-lossless) or ProRes 422 HQ. This is what pairs with a prototype recording: record the prototype, then inject the recording into a real photograph.
|
|
15
|
+
|
|
16
|
+
Still missing: **no ML detection** (measured, and it segments the phone body rather than the glass, so it is not shipped), **no occluder handling** — a finger or glare in front of the screen gets painted over — and **no camera motion**: the photo must be a still, so a clip of a moving phone is not this. Also, detection **abstains** rather than guessing when the background is itself neutral (a pale tiled floor, a plain wall); the edges get placed by hand there, which is normal, not a failure. Say so if any of it matters for the photo.
|
|
15
17
|
|
|
16
18
|
**Runs on the user's Mac shell** (Desktop Commander `start_process` or equivalent). The sandboxed Linux shell can't open a browser or reach `~/Desktop`. Paths below are relative to the plugin root — two levels up from this file.
|
|
17
19
|
|
|
@@ -23,10 +25,22 @@ Still missing: **no ML detection** (M4 — measured, and it segments the phone b
|
|
|
23
25
|
python3 scripts/preflight.py
|
|
24
26
|
```
|
|
25
27
|
|
|
26
|
-
Read the JSON.
|
|
28
|
+
Read the JSON. Note `python` when `ready` is true — **use that interpreter for every command below** (it's the venv at `~/.screengraft/venv`, not system Python).
|
|
29
|
+
|
|
30
|
+
**Both dependency questions are asked HERE, before the browser opens, and never later.** Once the UI is up the user is in a browser tab, not in the chat, and an `AskUserQuestion` there is a prompt they have walked away from. Ask everything you might need in one exchange, then launch.
|
|
31
|
+
|
|
32
|
+
**If `ready` is false — OpenCV is missing:**
|
|
33
|
+
|
|
34
|
+
- Say plainly what's missing and that **OpenCV is the engine: without it nothing runs — not worse results, no results.**
|
|
35
|
+
- Ask with `AskUserQuestion` whether to run `python3 scripts/preflight.py --install` (creates `~/.screengraft/venv`, pip-installs `opencv-python-headless` + `numpy`, ~60 MB, touches nothing else). Run it only after a yes.
|
|
27
36
|
|
|
28
|
-
|
|
29
|
-
|
|
37
|
+
**Independently of that — check `optional` for ffmpeg every time.** This check runs whether or not `ready` is true, because the usual case is exactly the one that would be skipped otherwise: someone who installed screengraft before video existed has a working venv, `ready` is `true`, and **nothing** tells them ffmpeg is absent until a render fails at the very end of a job. The JSON gives `status`, `install_command` and `install_size`. If it says `missing`:
|
|
38
|
+
|
|
39
|
+
- It is **optional and stills are unaffected** — never describe it as broken. It encodes video renders and nothing else.
|
|
40
|
+
- Offer it with `AskUserQuestion` — combined with the OpenCV question into one card when both are missing: *"Add video support? About 25 MB, into screengraft's own environment; nothing system-wide. Stills work either way."*
|
|
41
|
+
- On a yes, run `python3 scripts/preflight.py --install-ffmpeg` — one wheel, not a reinstall of everything. On a no, launch anyway and mention that video renders are unavailable until it is added; the fitting all works regardless.
|
|
42
|
+
|
|
43
|
+
These two are the **only** interview questions this skill asks in chat, both are install consent, and both belong before the launch. Everything else happens in the UI.
|
|
30
44
|
|
|
31
45
|
### 1. Launch the UI — with the project folder as the output directory
|
|
32
46
|
|
|
@@ -48,14 +62,16 @@ It prints one JSON line — `url`, `session`, `job`, `result`, `out_dir` — and
|
|
|
48
62
|
|
|
49
63
|
> **What this does** — it computes the perspective between your photo and your screenshot, so the screenshot lands on the glass exactly. Geometry, not AI: nothing is invented and your pixels are unchanged.
|
|
50
64
|
>
|
|
51
|
-
> **1 · Choose a photo, then a screenshot.** Recent images from Desktop and Downloads are listed for you — or drag a file in, browse, paste a path, or paste a Figma frame link and I'll export it.
|
|
65
|
+
> **1 · Choose a photo, then a screenshot — or a screen recording.** Recent images from Desktop and Downloads are listed for you — or drag a file in, browse, paste a path, or paste a Figma frame link and I'll export it. A video source (mp4/mov/webm) works the same way; you'll pick which frame to match the edges on.
|
|
52
66
|
>
|
|
53
67
|
> **2 · Match the four edges to the screen.** Drag an edge's middle to slide it, or near an end to pivot — only that edge moves. The magnified strip below shows the boundary straightened, so aligned reads as flat. Arrow keys nudge 1px, Shift+arrow 10px, Tab moves to the next edge.
|
|
54
68
|
>
|
|
55
|
-
> **3 ·
|
|
69
|
+
> **3 · Watch the Result pane** — it re-renders as you drag, so you judge the fit against the composite. Then **Save** — **Render**, for a video — and **Send to Claude**, and I'll check the result and show it here. Saves go to `<out-dir>`.
|
|
56
70
|
>
|
|
57
71
|
> A detector proposes a starting quad, but it's only a guess — you confirm all four edges. That's deliberate: a confident-looking wrong result is the one failure this tool won't risk.
|
|
58
72
|
|
|
73
|
+
**If ffmpeg was missing and the user declined it, say so in this same message** — one line, that video renders are unavailable until it's added and everything else works. That is the last moment they are still reading the chat.
|
|
74
|
+
|
|
59
75
|
Adapt it: name the real output folder, and mention the realism pass only if it matters (it is on by default and changes the screenshot's colour, which is worth flagging if they are reviewing brand colour). Say it once, on launch — not again on every re-arm.
|
|
60
76
|
|
|
61
77
|
Do not build a chat *interview* — the page collects every input, and duplicating its questions is what the one-interview-surface rule forbids. Explaining the workflow is not an interview. The page scans `~/Desktop` and `~/Downloads` for recent images itself, every launch.
|
|
@@ -94,7 +110,8 @@ The user pressed Save. Read the output image back (you can see images). Check:
|
|
|
94
110
|
|
|
95
111
|
- The injected screen sits on the bezel edge all the way round — no sliver of the original screen showing, no UI poking past the glass. Zoom a corner if unsure.
|
|
96
112
|
- Text in the injected area is sharp. Soft means double resampling — that's a bug, not a setting.
|
|
97
|
-
- Nothing that was in front of the screen in the photo has been painted over (if it has, say so —
|
|
113
|
+
- Nothing that was in front of the screen in the photo has been painted over (if it has, say so — occluders are not handled yet).
|
|
114
|
+
- **For a video render**, the same checks on a frame, plus: play it and confirm the screen does not pulse or shift, and that the photo around it is perfectly static. The renderer guarantees the second by construction — everything outside the screen mask is the original photo's bytes — so movement there is a bug worth reporting, not a setting.
|
|
98
115
|
|
|
99
116
|
Then `present_files` the output. Report what you checked, not "done".
|
|
100
117
|
|