screengraft 0.17.0 → 0.20.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/scripts/warp.py CHANGED
@@ -28,6 +28,8 @@ screen as it appears in the photo — order matters, it defines the mapping).
28
28
 
29
29
  import argparse
30
30
  import json
31
+ import os
32
+ import subprocess
31
33
  import sys
32
34
 
33
35
  import cv2
@@ -116,6 +118,140 @@ def shoelace_area(pts: np.ndarray) -> float:
116
118
  return 0.5 * abs(np.dot(x, np.roll(y, 1)) - np.dot(y, np.roll(x, 1)))
117
119
 
118
120
 
121
+ class Plan:
122
+ """Everything about a fit that does NOT change from frame to frame.
123
+
124
+ Built once for a still, once per clip for a video. The split exists so the
125
+ two paths cannot drift: `compose()` is a Plan plus one frame, and
126
+ `compose_video()` is the same Plan plus N frames, which makes "frame 0 of a
127
+ render equals the still composite, byte for byte" a property the tests can
128
+ assert rather than a thing we hope stays true.
129
+
130
+ What lives here is what a fixed photo and a fixed quad make constant:
131
+
132
+ - the prefilter's target size (the screenshot/frame is always the same
133
+ size, and the quad never moves, so the minification factor is fixed);
134
+ - the homography, which is derived from that rescaled size;
135
+ - the rounded source mask;
136
+ - the warped, antialiased destination mask, which is by far the most
137
+ expensive thing in compose() because it supersamples MASK_SS x over the
138
+ quad's bbox. Computing it once is most of the speed of a video render,
139
+ and it also means the screen's EDGE is pixel-identical in every frame,
140
+ so there is no edge crawl — the artefact that makes a composite read as
141
+ fake. A fixed photo is the one case where that comes for free;
142
+ - the grain sigma, measured from the photo, which does not change either.
143
+
144
+ The grade parameters are deliberately NOT built here: they need a frame to
145
+ measure against, so `bind_grade()` takes the fit frame and stores them.
146
+ """
147
+
148
+ def __init__(self, photo: np.ndarray, frame_shape, corners,
149
+ corner_radius: float = 0.0, grain: bool = False):
150
+ dst_quad = np.array(corners, dtype=np.float32)
151
+ if shoelace_area(dst_quad) < 1.0:
152
+ raise ValueError("degenerate quad (near-zero area) — check corner order TL,TR,BR,BL")
153
+ self.photo = photo
154
+ self.dst_quad = dst_quad
155
+ self.grain = grain
156
+
157
+ top = float(np.linalg.norm(dst_quad[1] - dst_quad[0]))
158
+ bottom = float(np.linalg.norm(dst_quad[2] - dst_quad[3]))
159
+ left = float(np.linalg.norm(dst_quad[3] - dst_quad[0]))
160
+ right = float(np.linalg.norm(dst_quad[2] - dst_quad[1]))
161
+ need_w, need_h = max(top, bottom), max(left, right)
162
+ sh0, sw0 = frame_shape[:2]
163
+ scale_x, scale_y = need_w / sw0, need_h / sh0
164
+ self.prescale = max(scale_x, scale_y)
165
+ if 0 < self.prescale < 0.95:
166
+ self.new_w = max(1, int(round(sw0 * self.prescale)))
167
+ self.new_h = max(1, int(round(sh0 * self.prescale)))
168
+ self.radius = corner_radius * (self.new_w / sw0)
169
+ else:
170
+ self.new_w, self.new_h = sw0, sh0
171
+ self.radius = corner_radius
172
+ self.prescale = 1.0
173
+
174
+ src_rect = np.array([[0, 0], [self.new_w, 0],
175
+ [self.new_w, self.new_h], [0, self.new_h]], dtype=np.float32)
176
+ self.H = cv2.getPerspectiveTransform(src_rect, dst_quad)
177
+ ph, pw = photo.shape[:2]
178
+ self.size = (pw, ph)
179
+ src_mask = rounded_mask(self.new_w, self.new_h, float(self.radius))
180
+ self.warped_mask = _warp_mask_antialiased(src_mask, self.H, pw, ph, dst_quad)
181
+ self.mask3 = cv2.merge([self.warped_mask] * 3).astype(np.float32) / 255.0
182
+ self.grain_sigma = (_grade.measure_grain(photo, _grade.surround_ring(self.warped_mask))
183
+ if grain else 0.0)
184
+ self.grade_params = None
185
+ # Integer bbox of the quad, clamped to the canvas and padded by a pixel
186
+ # so the antialiased edge is never clipped.
187
+ xs, ys = dst_quad[:, 0], dst_quad[:, 1]
188
+ bx0, by0 = max(0, int(np.floor(xs.min())) - 1), max(0, int(np.floor(ys.min())) - 1)
189
+ bx1, by1 = min(pw, int(np.ceil(xs.max())) + 2), min(ph, int(np.ceil(ys.max())) + 2)
190
+ self.bbox = (bx0, by0, bx1, by1) if bx1 > bx0 and by1 > by0 else None
191
+
192
+ def _prep(self, frame: np.ndarray, bbox=None) -> np.ndarray:
193
+ """Warp one frame. With `bbox`, warp only that window of the canvas.
194
+
195
+ The window is an integer translation of the output grid, folded into the
196
+ homography, so every output pixel resolves to exactly the same source
197
+ coordinate as the full-canvas warp — byte-identical, and asserted as
198
+ such in test_video.py. Outside the quad the mask is zero and the blend
199
+ is the photo copied onto itself, so on a 2400x1792 photo whose screen
200
+ occupies 9% of the frame this is most of the per-frame cost removed.
201
+ """
202
+ if (self.new_w, self.new_h) != (frame.shape[1], frame.shape[0]):
203
+ frame = cv2.resize(frame, (self.new_w, self.new_h), interpolation=cv2.INTER_AREA)
204
+ H, size = self.H, self.size
205
+ if bbox is not None:
206
+ x0, y0, x1, y1 = bbox
207
+ T = np.array([[1, 0, -x0], [0, 1, -y0], [0, 0, 1]], dtype=np.float64)
208
+ H, size = T @ H, (x1 - x0, y1 - y0)
209
+ return cv2.warpPerspective(frame, H, size,
210
+ flags=cv2.INTER_LANCZOS4,
211
+ borderMode=cv2.BORDER_REPLICATE)
212
+
213
+ def bind_grade(self, frame: np.ndarray, strength: float) -> None:
214
+ """Measure the light correction once, from the frame the user fitted on."""
215
+ self.grade_params = _grade.light_params(
216
+ self.photo, self._prep(frame), self.warped_mask, strength) if strength > 0 else None
217
+
218
+ def render(self, frame: np.ndarray, screen_off: np.ndarray = None,
219
+ specular: float = 0.75, fast: bool = False) -> np.ndarray:
220
+ """Composite one frame onto the photo.
221
+
222
+ `fast` confines the warp and the blend to the quad's bounding box. It is
223
+ off for stills, where a single frame's cost is irrelevant and the
224
+ simplest code is the one to trust, and on for video renders. Both
225
+ produce identical bytes; test_video.py asserts it rather than assuming.
226
+ """
227
+ if fast and self.bbox is not None:
228
+ x0, y0, x1, y1 = self.bbox
229
+ warped_screen = self._prep(frame, self.bbox)
230
+ if self.grade_params is not None:
231
+ warped_screen = _grade.apply_light(warped_screen, self.grade_params)
232
+ out = self.photo.copy()
233
+ win = self.mask3[y0:y1, x0:x1]
234
+ out[y0:y1, x0:x1] = np.clip(
235
+ self.photo[y0:y1, x0:x1].astype(np.float32) * (1 - win)
236
+ + warped_screen.astype(np.float32) * win, 0, 255).astype(np.uint8)
237
+ else:
238
+ warped_screen = self._prep(frame)
239
+ if self.grade_params is not None:
240
+ warped_screen = _grade.apply_light(warped_screen, self.grade_params)
241
+ out = (self.photo.astype(np.float32) * (1 - self.mask3)
242
+ + warped_screen.astype(np.float32) * self.mask3)
243
+ out = np.clip(out, 0, 255).astype(np.uint8)
244
+ if self.grain:
245
+ # Seeded, so the grain is IDENTICAL in every frame. Over a still
246
+ # photograph that is what it must be: the background's own noise is
247
+ # frozen, and grain that crawled on the screen alone would read as a
248
+ # dirty window. It also keeps the render deterministic.
249
+ out = _grade.add_grain(out, self.warped_mask, self.grain_sigma)
250
+ if screen_off is not None:
251
+ out = _grade.specular_lift(out, screen_off, self.warped_mask, strength=specular)
252
+ return out
253
+
254
+
119
255
  def compose(photo: np.ndarray, screenshot: np.ndarray, corners, corner_radius: float = 0.0,
120
256
  grade: float = 0.0, grain: bool = False, screen_off: np.ndarray = None,
121
257
  specular: float = 0.75) -> np.ndarray:
@@ -124,79 +260,160 @@ def compose(photo: np.ndarray, screenshot: np.ndarray, corners, corner_radius: f
124
260
  Single resampling pass at the photo's resolution; deterministic. This is the
125
261
  whole engine — the CLI below and ui.py both call it.
126
262
  """
127
- dst_quad = np.array(corners, dtype=np.float32)
128
- if shoelace_area(dst_quad) < 1.0:
129
- raise ValueError("degenerate quad (near-zero area) — check corner order TL,TR,BR,BL")
130
-
131
- # --- prefilter for minification -------------------------------------
132
- # A UI screenshot is almost always far larger than the screen it lands
133
- # on: 1206x2622 into a 226x454 quad is 5.3x across and 5.8x along, so
134
- # each output pixel is the average of ~31 source pixels. warpPerspective
135
- # (like remap) does NOT area-average — INTER_LANCZOS4 samples a fixed 8x8
136
- # window around one source point no matter the scale, and Lanczos is a
137
- # sharpening kernel, so heavy minification came out aliased and crunchy
138
- # with text turned to noise (3 Sep 2026, reported from a real save).
139
- #
140
- # The fix is the mipmap principle: area-average DOWN to roughly the
141
- # destination footprint first, then warp at ~1:1. This is not the
142
- # "warp-then-scale" the build brief warns against — that's resampling an
143
- # already-warped result, which blurs. This is resampling the source with
144
- # the right filter before the only geometric pass, which is how you avoid
145
- # aliasing when minifying.
146
- top = float(np.linalg.norm(dst_quad[1] - dst_quad[0]))
147
- bottom = float(np.linalg.norm(dst_quad[2] - dst_quad[3]))
148
- left = float(np.linalg.norm(dst_quad[3] - dst_quad[0]))
149
- right = float(np.linalg.norm(dst_quad[2] - dst_quad[1]))
150
- # Use the LONGER opposing edge of each pair: under perspective the near
151
- # edge carries the most detail, and that's the resolution to preserve.
152
- need_w = max(top, bottom)
153
- need_h = max(left, right)
154
- sh0, sw0 = screenshot.shape[:2]
155
- scale_x, scale_y = need_w / sw0, need_h / sh0
156
- if 0 < max(scale_x, scale_y) < 0.95: # only ever downsample
157
- new_w = max(1, int(round(sw0 * max(scale_x, scale_y))))
158
- new_h = max(1, int(round(sh0 * max(scale_x, scale_y))))
159
- screenshot = cv2.resize(screenshot, (new_w, new_h), interpolation=cv2.INTER_AREA)
160
- corner_radius = corner_radius * (new_w / sw0) # radius is in source px
161
- # --------------------------------------------------------------------
263
+ # A Plan plus one frame. The long-form pipeline this replaced (prefilter,
264
+ # homography, rounded mask, antialiased mask warp, grade, blend, grain,
265
+ # specular) now lives in Plan, so the still and video paths run the SAME
266
+ # code and cannot drift apart. See test_video.py: frame 0 of a render is
267
+ # asserted byte-identical to this function's output.
268
+ plan = Plan(photo, screenshot.shape, corners, corner_radius, grain=grain)
269
+ plan.bind_grade(screenshot, grade)
270
+ return plan.render(screenshot, screen_off=screen_off, specular=specular)
271
+
272
+
273
+ def ffmpeg_exe() -> str:
274
+ """Path to the ffmpeg binary, or raise with something a designer can act on.
275
+
276
+ imageio_ffmpeg ships a static binary as a wheel, so it installs into the
277
+ same venv OpenCV already lives in and the zero-configuration constraint
278
+ holds — no Homebrew, no PATH hunting. cv2.VideoWriter was rejected for this
279
+ job: the headless wheels carry a limited codec set, expose no control over
280
+ bitrate or pixel format, and behave differently per platform, none of which
281
+ is acceptable when the output is the deliverable.
282
+ """
283
+ try:
284
+ import imageio_ffmpeg
285
+ return imageio_ffmpeg.get_ffmpeg_exe()
286
+ except Exception as e:
287
+ raise RuntimeError(
288
+ "ffmpeg is missing. Run `python3 scripts/preflight.py --install` to add it "
289
+ "to the screengraft venv (it ships as a wheel; nothing is installed "
290
+ "system-wide)."
291
+ ) from e
292
+
293
+
294
+ # Output presets. The last stage of the pipeline is the only one that can undo
295
+ # the care taken in all the others: H.264's 4:2:0 chroma subsampling softens
296
+ # exactly the coloured text edges the INTER_AREA prefilter exists to protect.
297
+ # So "web" runs at CRF 16, which is near-visually-lossless rather than
298
+ # delivery-sized, and anything destined for a case study should use prores.
299
+ # BT.709 is tagged explicitly on both so players do not guess at the primaries
300
+ # and shift the colour we just matched to the room.
301
+ PRESETS = {
302
+ "web": ["-c:v", "libx264", "-preset", "slow", "-crf", "16",
303
+ "-pix_fmt", "yuv420p", "-movflags", "+faststart"],
304
+ "prores": ["-c:v", "prores_ks", "-profile:v", "3", "-pix_fmt", "yuv422p10le"],
305
+ }
306
+
307
+
308
+ def probe_video(path: str):
309
+ """(frame_count, fps, width, height) — read, never trusted blindly."""
310
+ cap = cv2.VideoCapture(path)
311
+ if not cap.isOpened():
312
+ raise RuntimeError(f"could not open video: {path}")
313
+ fps = float(cap.get(cv2.CAP_PROP_FPS)) or 0.0
314
+ n = int(cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0)
315
+ w = int(cap.get(cv2.CAP_PROP_FRAME_WIDTH) or 0)
316
+ h = int(cap.get(cv2.CAP_PROP_FRAME_HEIGHT) or 0)
317
+ cap.release()
318
+ # CAP_PROP_FRAME_COUNT is a container hint and is wrong often enough that
319
+ # the render loop counts frames as it reads them instead. It is reported
320
+ # here only to drive a progress bar.
321
+ if not (0.1 <= fps <= 240):
322
+ fps = 30.0
323
+ return n, fps, w, h
324
+
325
+
326
+ def read_frame_at(path: str, index: int = 0):
327
+ """One frame, for fitting and for the poster image."""
328
+ cap = cv2.VideoCapture(path)
329
+ if not cap.isOpened():
330
+ raise RuntimeError(f"could not open video: {path}")
331
+ if index > 0:
332
+ cap.set(cv2.CAP_PROP_POS_FRAMES, index)
333
+ ok, frame = cap.read()
334
+ if not ok and index > 0:
335
+ # Seeking is unreliable on some containers; fall back to reading
336
+ # forward, which is slow and always right.
337
+ cap.release()
338
+ cap = cv2.VideoCapture(path)
339
+ for _ in range(index + 1):
340
+ ok, frame = cap.read()
341
+ if not ok:
342
+ break
343
+ cap.release()
344
+ if not ok or frame is None:
345
+ raise RuntimeError(f"could not read frame {index} of {path}")
346
+ return frame
347
+
348
+
349
+ def compose_video(photo: np.ndarray, video_path: str, corners, output: str,
350
+ corner_radius: float = 0.0, grade: float = 0.0, grain: bool = False,
351
+ preset: str = "web", fit_frame: int = 0, audio: bool = True,
352
+ frames_dir: str = None, progress=None) -> dict:
353
+ """Inject a VIDEO into a still photo. The photo does not move, so there is
354
+ exactly one homography and the whole of Plan is computed once.
355
+
356
+ Frames are piped to ffmpeg as raw BGR24 rather than written out as a PNG
357
+ sequence: a ten-second clip is several hundred frames and several GB of
358
+ intermediate PNGs, for no benefit. `frames_dir` still dumps them when a
359
+ test or a human needs to look at individual frames.
360
+
361
+ Time is deliberately NOT resampled. The output runs at the source's own
362
+ frame rate; converting fps by dropping or duplicating frames is judder, and
363
+ doing it properly means blending, which is a different feature. Note also
364
+ that a prototype recording has no motion blur — it renders discrete frames
365
+ — so a fast scroll will strobe. That is a property of the input, the same
366
+ way the source resolution is, not something this stage should paper over.
367
+ """
368
+ n_hint, fps, vw, vh = probe_video(video_path)
369
+ first = read_frame_at(video_path, fit_frame)
370
+ plan = Plan(photo, first.shape, corners, corner_radius, grain=grain)
371
+ plan.bind_grade(first, grade)
162
372
 
163
- sh, sw = screenshot.shape[:2]
164
- src_rect = np.array([[0, 0], [sw, 0], [sw, sh], [0, sh]], dtype=np.float32)
165
- H = cv2.getPerspectiveTransform(src_rect, dst_quad)
166
373
  ph, pw = photo.shape[:2]
167
- # Radius stays fractional: rounded_mask is analytic, and after the
168
- # prefilter rescale above a truncation here is up to a whole pixel of
169
- # radius thrown away at exactly the scale the viewer is looking at.
170
- src_mask = rounded_mask(sw, sh, float(corner_radius))
171
- # BORDER_REPLICATE, not BORDER_CONSTANT black: with an antialiased mask
172
- # the edge pixels are a genuine blend of screen and photo, and sampling
173
- # black just outside the screenshot would draw a dark fringe right where
174
- # the antialiasing is supposed to be doing its work. Replicating the edge
175
- # pixel means a half-covered pixel blends real screen colour instead.
176
- warped_screen = cv2.warpPerspective(
177
- screenshot, H, (pw, ph), flags=cv2.INTER_LANCZOS4,
178
- borderMode=cv2.BORDER_REPLICATE)
179
- warped_mask = _warp_mask_antialiased(src_mask, H, pw, ph, dst_quad)
180
-
181
- # --- M2: the realism pass -------------------------------------------
182
- # Order is not arbitrary. The light match runs on the warped screen BEFORE
183
- # compositing, so it measures and moves only screen pixels — grading after
184
- # the blend would drag the bezel with it. Grain and the specular lift run
185
- # AFTER, because both are things that happen to the finished surface, and
186
- # both are confined to the screen by the same mask.
187
- if grade > 0:
188
- warped_screen = _grade.match_light(photo, warped_screen, warped_mask, strength=grade)
189
-
190
- mask3 = cv2.merge([warped_mask] * 3).astype(np.float32) / 255.0
191
- out = photo.astype(np.float32) * (1 - mask3) + warped_screen.astype(np.float32) * mask3
192
- out = np.clip(out, 0, 255).astype(np.uint8)
193
-
194
- if grain:
195
- sigma = _grade.measure_grain(photo, _grade.surround_ring(warped_mask))
196
- out = _grade.add_grain(out, warped_mask, sigma)
197
- if screen_off is not None:
198
- out = _grade.specular_lift(out, screen_off, warped_mask, strength=specular)
199
- return out
374
+ cmd = [ffmpeg_exe(), "-y", "-loglevel", "error",
375
+ "-f", "rawvideo", "-pix_fmt", "bgr24", "-s", f"{pw}x{ph}",
376
+ "-r", f"{fps}", "-i", "-"]
377
+ if audio:
378
+ # Optional by construction: `?` on the map means a source with no audio
379
+ # track (which a prototype recording usually is) is not an error.
380
+ cmd += ["-i", video_path, "-map", "0:v", "-map", "1:a?", "-c:a", "aac", "-shortest"]
381
+ cmd += PRESETS.get(preset, PRESETS["web"])
382
+ cmd += ["-colorspace", "bt709", "-color_primaries", "bt709", "-color_trc", "bt709",
383
+ output]
384
+
385
+ if frames_dir:
386
+ os.makedirs(frames_dir, exist_ok=True)
387
+ cap = cv2.VideoCapture(video_path)
388
+ proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stderr=subprocess.PIPE)
389
+ count = 0
390
+ try:
391
+ while True:
392
+ ok, frame = cap.read()
393
+ if not ok:
394
+ break
395
+ out = plan.render(frame, fast=True)
396
+ if frames_dir:
397
+ cv2.imwrite(os.path.join(frames_dir, f"{count:06d}.png"), out,
398
+ [cv2.IMWRITE_PNG_COMPRESSION, 1])
399
+ proc.stdin.write(out.tobytes())
400
+ count += 1
401
+ if progress and count % 10 == 0:
402
+ progress(count, n_hint)
403
+ finally:
404
+ cap.release()
405
+ try:
406
+ proc.stdin.close()
407
+ except (BrokenPipeError, OSError):
408
+ pass
409
+ err = proc.stderr.read().decode("utf-8", "replace")
410
+ proc.wait()
411
+ if proc.returncode != 0:
412
+ raise RuntimeError(f"ffmpeg failed ({proc.returncode}): {err.strip()[:400]}")
413
+ if count == 0:
414
+ raise RuntimeError(f"no frames could be read from {video_path}")
415
+ return {"frames": count, "fps": fps, "source_size": [vw, vh],
416
+ "output_size": [pw, ph], "preset": preset, "fit_frame": fit_frame}
200
417
 
201
418
 
202
419
  def main() -> None:
@@ -1,17 +1,19 @@
1
1
  ---
2
2
  name: inject-screenshot
3
- description: Injects a UI screenshot onto a photographed device screen (phone/tablet/laptop) at any angle, matching the perspective exactly via homography — geometry, not AI generation. Use when the user wants to composite a screen design into a real device photo for a portfolio, case study, or mockup, and needs the result to look like a genuine photo rather than a template. Opens a local browser UI for picking files, correcting corners and saving.
3
+ description: Injects a UI screenshot OR a screen recording onto a photographed device screen (phone/tablet/laptop) at any angle, matching the perspective exactly via homography — geometry, not AI generation. Use when the user wants to composite a screen design into a real device photo for a portfolio, case study, or mockup, or wants a prototype recording playing on a real device in a photograph, and needs the result to look like a genuine photo rather than a template. Handles video sources (mp4/mov/webm) by fitting once and rendering every frame. Opens a local browser UI for picking files, correcting corners, and saving or rendering.
4
4
  ---
5
5
 
6
6
  # Inject a screenshot onto a photographed device
7
7
 
8
- **What ships (v0.13.1):** a local browser UI (`scripts/ui.py`) that walks the designer through the whole job — pick the photo and the screenshot (recent Desktop/Downloads images, drag-drop, browse, path, or a **Figma frame link**), auto-detect the screen as a starting position, then **match the four edges** (drag an edge's middle to slide it, near an end to pivot; corners still draggable) with canvas zoom/pan and a rectified strip loupe, Preview (in a popup, or continuously in the compare pane), an on-by-default realism pass that colour-matches the screenshot to the photo's light, Save into the project folder (`--out-dir`), and a **Send to Claude** button that reaches you through the plugin's own MCP server. The UI is a hand port of the project's Figma design file — dark only.
8
+ **What ships (v0.20):** a local browser UI (`scripts/ui.py`) that walks the designer through the whole job — pick the photo and the screen source, which may be an image **or a video** (recent Desktop/Downloads images, drag-drop, browse, path, or a **Figma frame link**), auto-detect the screen as a starting position, then **match the four edges** (drag an edge's middle to slide it, near an end to pivot; corners still draggable) with canvas zoom/pan and a rectified strip loupe. The fit and the composite sit **side by side and always have** — the result pane re-renders as you drag, which is how a corner gets judged, so it is the layout rather than a mode you can switch off. Then an on-by-default realism pass that colour-matches the source to the photo's light, **Save** (or **Render**, for a video) into the project folder (`--out-dir`), and a **Send to Claude** button that reaches you through the plugin's own MCP server. The UI is a hand port of the project's Figma design file — dark only.
9
9
 
10
10
  The geometry is exact (`warp.py`); the detection is advisory (`detect.py`) and the human corrects it.
11
11
 
12
- **The realism pass ships and is ON by default** (`grade.py`, M2): it matches the injected screen's white balance and grain to the light around it, at a strength the designer sets in the rail. It can also lift the device's real specular highlights from a screen-off reference frame, though the UI cannot supply one yet. Off is a first-class choice and keeps the screenshot's colour exactly — say so if the user is reviewing brand colour.
12
+ **The realism pass ships and is ON by default** (`grade.py`): it matches the injected screen's white balance and grain to the light around it, at a strength the designer sets in the rail. It can also lift the device's real specular highlights from a screen-off reference frame, though the UI cannot supply one yet. Off is a first-class choice and keeps the screenshot's colour exactly — say so if the user is reviewing brand colour.
13
13
 
14
- Still missing: **no ML detection** (M4 — measured, and it segments the phone body rather than the glass, so it is not shipped), **no occluder handling** — a finger or glare in front of the screen gets painted over (M5) — and **no video** (M3). Say so if it matters for the photo.
14
+ **Video ships too.** The screen source can be a video (mp4/mov/webm) as well as a still — pick it exactly like a screenshot, choose which frame to match the edges on, and the primary button becomes **Render**. The photo does not move, so there is one homography and every frame gets the same geometry; the light match is measured once from the frame you fitted on, so the screen cannot pulse as the UI scrolls. Output is H.264 at CRF 16 (near-visually-lossless) or ProRes 422 HQ. This is what pairs with a prototype recording: record the prototype, then inject the recording into a real photograph.
15
+
16
+ Still missing: **no ML detection** (measured, and it segments the phone body rather than the glass, so it is not shipped), **no occluder handling** — a finger or glare in front of the screen gets painted over — and **no camera motion**: the photo must be a still, so a clip of a moving phone is not this. Also, detection **abstains** rather than guessing when the background is itself neutral (a pale tiled floor, a plain wall); the edges get placed by hand there, which is normal, not a failure. Say so if any of it matters for the photo.
15
17
 
16
18
  **Runs on the user's Mac shell** (Desktop Commander `start_process` or equivalent). The sandboxed Linux shell can't open a browser or reach `~/Desktop`. Paths below are relative to the plugin root — two levels up from this file.
17
19
 
@@ -23,10 +25,22 @@ Still missing: **no ML detection** (M4 — measured, and it segments the phone b
23
25
  python3 scripts/preflight.py
24
26
  ```
25
27
 
26
- Read the JSON. If `ready` is true, note `python` — **use that interpreter for every command below** (it's the venv at `~/.screengraft/venv`, not system Python). If `ready` is false:
28
+ Read the JSON. Note `python` when `ready` is true — **use that interpreter for every command below** (it's the venv at `~/.screengraft/venv`, not system Python).
29
+
30
+ **Both dependency questions are asked HERE, before the browser opens, and never later.** Once the UI is up the user is in a browser tab, not in the chat, and an `AskUserQuestion` there is a prompt they have walked away from. Ask everything you might need in one exchange, then launch.
31
+
32
+ **If `ready` is false — OpenCV is missing:**
33
+
34
+ - Say plainly what's missing and that **OpenCV is the engine: without it nothing runs — not worse results, no results.**
35
+ - Ask with `AskUserQuestion` whether to run `python3 scripts/preflight.py --install` (creates `~/.screengraft/venv`, pip-installs `opencv-python-headless` + `numpy`, ~60 MB, touches nothing else). Run it only after a yes.
27
36
 
28
- - Tell the user plainly what's missing and that **OpenCV is the engine: without it nothing runs — not worse results, no results.**
29
- - Ask with `AskUserQuestion` whether to run the install: `python3 scripts/preflight.py --install` (creates `~/.screengraft/venv`, pip-installs `opencv-python-headless` + `numpy`, ~60 MB, touches nothing else). Run it only after a yes. This is the **only** interview question this skill asks in chat — everything else happens in the UI.
37
+ **Independently of that — check `optional` for ffmpeg every time.** This check runs whether or not `ready` is true, because the usual case is exactly the one that would be skipped otherwise: someone who installed screengraft before video existed has a working venv, `ready` is `true`, and **nothing** tells them ffmpeg is absent until a render fails at the very end of a job. The JSON gives `status`, `install_command` and `install_size`. If it says `missing`:
38
+
39
+ - It is **optional and stills are unaffected** — never describe it as broken. It encodes video renders and nothing else.
40
+ - Offer it with `AskUserQuestion` — combined with the OpenCV question into one card when both are missing: *"Add video support? About 25 MB, into screengraft's own environment; nothing system-wide. Stills work either way."*
41
+ - On a yes, run `python3 scripts/preflight.py --install-ffmpeg` — one wheel, not a reinstall of everything. On a no, launch anyway and mention that video renders are unavailable until it is added; the fitting all works regardless.
42
+
43
+ These two are the **only** interview questions this skill asks in chat, both are install consent, and both belong before the launch. Everything else happens in the UI.
30
44
 
31
45
  ### 1. Launch the UI — with the project folder as the output directory
32
46
 
@@ -48,14 +62,16 @@ It prints one JSON line — `url`, `session`, `job`, `result`, `out_dir` — and
48
62
 
49
63
  > **What this does** — it computes the perspective between your photo and your screenshot, so the screenshot lands on the glass exactly. Geometry, not AI: nothing is invented and your pixels are unchanged.
50
64
  >
51
- > **1 · Choose a photo, then a screenshot.** Recent images from Desktop and Downloads are listed for you — or drag a file in, browse, paste a path, or paste a Figma frame link and I'll export it.
65
+ > **1 · Choose a photo, then a screenshot — or a screen recording.** Recent images from Desktop and Downloads are listed for you — or drag a file in, browse, paste a path, or paste a Figma frame link and I'll export it. A video source (mp4/mov/webm) works the same way; you'll pick which frame to match the edges on.
52
66
  >
53
67
  > **2 · Match the four edges to the screen.** Drag an edge's middle to slide it, or near an end to pivot — only that edge moves. The magnified strip below shows the boundary straightened, so aligned reads as flat. Arrow keys nudge 1px, Shift+arrow 10px, Tab moves to the next edge.
54
68
  >
55
- > **3 · Preview, then Save**, then **Send to Claude** and I'll check the result and show it here. Saves go to `<out-dir>`.
69
+ > **3 · Watch the Result pane** — it re-renders as you drag, so you judge the fit against the composite. Then **Save** — **Render**, for a video — and **Send to Claude**, and I'll check the result and show it here. Saves go to `<out-dir>`.
56
70
  >
57
71
  > A detector proposes a starting quad, but it's only a guess — you confirm all four edges. That's deliberate: a confident-looking wrong result is the one failure this tool won't risk.
58
72
 
73
+ **If ffmpeg was missing and the user declined it, say so in this same message** — one line, that video renders are unavailable until it's added and everything else works. That is the last moment they are still reading the chat.
74
+
59
75
  Adapt it: name the real output folder, and mention the realism pass only if it matters (it is on by default and changes the screenshot's colour, which is worth flagging if they are reviewing brand colour). Say it once, on launch — not again on every re-arm.
60
76
 
61
77
  Do not build a chat *interview* — the page collects every input, and duplicating its questions is what the one-interview-surface rule forbids. Explaining the workflow is not an interview. The page scans `~/Desktop` and `~/Downloads` for recent images itself, every launch.
@@ -94,7 +110,8 @@ The user pressed Save. Read the output image back (you can see images). Check:
94
110
 
95
111
  - The injected screen sits on the bezel edge all the way round — no sliver of the original screen showing, no UI poking past the glass. Zoom a corner if unsure.
96
112
  - Text in the injected area is sharp. Soft means double resampling — that's a bug, not a setting.
97
- - Nothing that was in front of the screen in the photo has been painted over (if it has, say so — M5).
113
+ - Nothing that was in front of the screen in the photo has been painted over (if it has, say so — occluders are not handled yet).
114
+ - **For a video render**, the same checks on a frame, plus: play it and confirm the screen does not pulse or shift, and that the photo around it is perfectly static. The renderer guarantees the second by construction — everything outside the screen mask is the original photo's bytes — so movement there is a bug worth reporting, not a setting.
98
115
 
99
116
  Then `present_files` the output. Report what you checked, not "done".
100
117