screengraft 0.38.0 → 0.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "screengraft",
3
- "version": "0.38.0",
3
+ "version": "0.40.0",
4
4
  "description": "Put a UI screenshot or screen recording onto a photographed device screen with the perspective exactly right \u2014 a homography you confirm by hand, not a generative guess.",
5
5
  "keywords": [
6
6
  "mockup",
package/scripts/detect.py CHANGED
@@ -797,12 +797,31 @@ def detect_edges(gray: np.ndarray, click=None, trace=None):
797
797
  # Sweep the Canny thresholds off the image's own median rather than fixed
798
798
  # numbers, then a few sigmas around it — one exposure doesn't suit both a
799
799
  # bright render and a dim photo.
800
+ #
801
+ # ... and ALSO off Otsu's threshold, because the median anchor has a hole
802
+ # the median cannot see: a dark photograph. A black phone on a black
803
+ # backdrop has a median of 0..4, so every sweep point lands at hi <= 7 and
804
+ # Canny fires on every pixel of noise -- the edge map is a solid sheet and
805
+ # no closed quad survives it. Both photographs in the labelled corpus on
806
+ # which nothing near the screen was EVER proposed (11 Sep 2026) are exactly
807
+ # this, and edge returned zero candidates on them. Otsu splits the two
808
+ # modes that are actually there -- 97 and 127 on those two -- and the same
809
+ # sweep then proposes the screen at 9% and 12% unrefined, which is where
810
+ # every other photograph's nearest candidate sits. On the other seven the
811
+ # nearest candidate is unchanged. The saturation channel already anchors
812
+ # on Otsu for the same reason (one percentile fails when the thing sought
813
+ # is smaller than the percentile).
800
814
  med = float(np.median(blur))
801
- for sigma in (0.20, 0.33, 0.50, 0.66):
802
- lo = int(max(0, (1.0 - sigma) * med))
803
- hi = int(min(255, (1.0 + sigma) * med))
804
- if hi <= lo:
815
+ otsu, _ = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
816
+ sweep = [((1.0 - s) * med, (1.0 + s) * med) for s in (0.20, 0.33, 0.50, 0.66)]
817
+ sweep += [(otsu * f / 2.0, otsu * f) for f in (0.5, 1.0, 1.5)]
818
+ seen = set()
819
+ for lo_f, hi_f in sweep:
820
+ lo = int(max(0, lo_f))
821
+ hi = int(min(255, hi_f))
822
+ if hi <= lo or (lo, hi) in seen:
805
823
  continue
824
+ seen.add((lo, hi))
806
825
  edges = cv2.Canny(blur, lo, hi, L2gradient=True)
807
826
  # Close small gaps so a bezel outline broken by a notch or a glare
808
827
  # spot still forms one closed contour.
@@ -996,22 +1015,63 @@ def arbitrate(results, shape, click=None):
996
1015
  sat = next((r for r in results if r["method"] == "saturation"), None)
997
1016
  why = "it was the only detector left after the rounded-corner filter" \
998
1017
  if len(results) == 1 else "its tone band assumption held, which is itself evidence"
999
- if t and e:
1000
- tq, eq = t["_corners_np"], e["_corners_np"]
1001
- ta = float(cv2.contourArea(tq.astype(np.float32)))
1018
+ # ONE region-vs-edge arbitration, for whichever region channel survived the
1019
+ # rounded-corner filter above. Until 11 Sep 2026 only tone got the nesting
1020
+ # rules; saturation-vs-edge fell through to "saturation wins". On iPhone-4
1021
+ # that returned a saturation quad 18% off, nested around an edge quad 3%
1022
+ # off at 94% of its area -- exactly the "screen inside a body" shape the
1023
+ # tone branch already knew how to read.
1024
+ region = t if t is not None else sat
1025
+ rname = "tone" if t is not None else "saturation"
1026
+ if region is not None and e is not None:
1027
+ rq, eq = region["_corners_np"], e["_corners_np"]
1028
+ ra = float(cv2.contourArea(rq.astype(np.float32)))
1002
1029
  ea = float(cv2.contourArea(eq.astype(np.float32)))
1003
- nested = overlap_frac(tq, eq) >= 0.90 and ta < ea
1004
- ratio = ta / ea if ea > 0 else 0.0
1005
- if nested and ratio < NEST_FLOOR:
1006
- best, why = e, ("the tone quad is only %.0f%% of the edge quad it sits "
1007
- "inside — that's content drawn on the screen, not the "
1008
- "screen" % (ratio * 100))
1030
+ # Nesting is read on whichever quad is inside the other. It used to be
1031
+ # read only with the region quad inside the edge quad, so an edge quad
1032
+ # sitting inside a region quad (iPhone-4: edge 3% off inside a
1033
+ # saturation quad 18% off at 94%) was "not nested" and fell through to
1034
+ # the region winning by default.
1035
+ inner, outer = (region, e) if ra < ea else (e, region)
1036
+ iname, oname = (rname, "edge") if ra < ea else ("edge", rname)
1037
+ nested = overlap_frac(inner["_corners_np"], outer["_corners_np"]) >= 0.90
1038
+ ratio = min(ra, ea) / max(ra, ea) if max(ra, ea) > 0 else 0.0
1039
+ ti, to = shape_tier(inner), shape_tier(outer)
1040
+ tr, te = shape_tier(region), shape_tier(e)
1041
+ # Nested: shape evidence first, the area ratio only to break a tie.
1042
+ #
1043
+ # The ratio alone reads "small inside big" as content-inside-screen and
1044
+ # "large inside big" as screen-inside-body, and it is right on the
1045
+ # gradient-screen fixture (41%: a slab on the screen) and on most
1046
+ # photographs. It is wrong exactly when the quads' own corners say the
1047
+ # opposite: a UI content region at 96% of the screen with loose corners
1048
+ # (two of ten labelled photographs, 15% off) is not a screen inside a
1049
+ # body, and a confident phone screen at 34% of a loosely-rounded table
1050
+ # (the hand-built case in test_detect) is not content on it. Tiers
1051
+ # decide those; at equal tiers the ratio still does, unchanged.
1052
+ if nested and to > ti:
1053
+ best, why = outer, ("the %s quad sits inside the %s quad at %.0f%% of its "
1054
+ "area, but the outer quad's corners agree on a radius "
1055
+ "(tier %d) and the inner's do not (tier %d) — content "
1056
+ "inside a screen, not a screen inside a body"
1057
+ % (iname, oname, ratio * 100, to, ti))
1058
+ elif nested and ti > to:
1059
+ best, why = inner, ("the %s quad sits inside the %s quad at %.0f%% of its "
1060
+ "area, and the inner quad's corners agree on a radius "
1061
+ "(tier %d) where the outer's do not (tier %d) — a "
1062
+ "screen on something larger"
1063
+ % (iname, oname, ratio * 100, ti, to))
1064
+ elif nested and ratio < NEST_FLOOR:
1065
+ best, why = outer, ("the %s quad is only %.0f%% of the %s quad it sits "
1066
+ "inside — that's content drawn on the screen, not the "
1067
+ "screen" % (iname, ratio * 100, oname))
1009
1068
  elif nested:
1010
- best, why = t, ("the tone quad sits inside the edge quad at %.0f%% of "
1011
- "its area — a screen inside a device body" % (ratio * 100))
1012
- elif shape_tier(e) > shape_tier(t):
1069
+ best, why = inner, ("the %s quad sits inside the %s quad at %.0f%% of "
1070
+ "its area — a screen inside a device body"
1071
+ % (iname, oname, ratio * 100))
1072
+ elif te > tr:
1013
1073
  # Not nested, and the edge quad's own shape says "screen" more
1014
- # strongly than tone's does. Before 11 Sep 2026 tone won here
1074
+ # strongly than the region's does. Before 11 Sep 2026 tone won here
1015
1075
  # unconditionally, on the argument that its band assumption holding
1016
1076
  # was itself evidence -- and on two of ten labelled photographs that
1017
1077
  # handed the answer to a tone quad 159% and 249% off, with corner
@@ -1019,22 +1079,12 @@ def arbitrate(results, shape, click=None):
1019
1079
  # radius. A band assumption is weaker evidence than four agreeing
1020
1080
  # corners; the tiers say so and this reads them.
1021
1081
  best, why = e, ("the two quads aren't nested and the edge quad's "
1022
- "corners agree on a radius (tier %d) where tone's do "
1023
- "not (tier %d)" % (shape_tier(e), shape_tier(t)))
1024
- else:
1025
- best, why = t, ("the two quads aren't nested; tone's band assumption "
1026
- "holding is itself evidence that it found a screen")
1027
- elif t is None and sat is not None:
1028
- # tone was dropped for having sharp corners, or never fired. The
1029
- # surviving region detector measured a real corner radius; edge usually
1030
- # cannot -- but when it did, and more confidently, it wins the same way.
1031
- if e is not None and shape_tier(e) > shape_tier(sat):
1032
- best, why = e, ("the edge quad's corners agree on a radius (tier %d) "
1033
- "where saturation's do not (tier %d)"
1034
- % (shape_tier(e), shape_tier(sat)))
1082
+ "corners agree on a radius (tier %d) where %s's do "
1083
+ "not (tier %d)" % (te, rname, tr))
1035
1084
  else:
1036
- best, why = sat, ("the saturation detector found a region with rounded "
1037
- "corners where tone did not")
1085
+ best, why = region, ("the two quads aren't nested; %s's assumption "
1086
+ "holding is itself evidence that it found a screen"
1087
+ % rname)
1038
1088
  else:
1039
1089
  best = t or e or sat
1040
1090
  if best is None:
@@ -5,7 +5,7 @@ description: Injects a UI screenshot OR a screen recording onto a photographed d
5
5
 
6
6
  # Inject a screenshot onto a photographed device
7
7
 
8
- **What ships (v0.38):** a local browser UI (`scripts/ui.py`) that walks the designer through the whole job — pick the photo and the screen source, which may be an image **or a video** (recent Desktop/Downloads images, drag-drop, browse, path, or a **Figma frame link**), auto-detect the screen as a starting position — and when detection cannot tell which region is a screen, **Point at screen**: one click inside it and the detector uses that point — or, if this photograph has been fitted before, **the fit it was saved with comes back** as the starting position instead of a detection, recognised by the photo's own pixels so a rename or a drag-drop still match — and every save also writes a **portable `.fit.json` beside the mockup** that can be dropped back onto the page later, which is how a fit survives a re-export, another machine, or someone else's hands — then **match the four edges** (drag an edge's middle to slide it, near an end to pivot; corners still draggable) with canvas navigation that follows the usual conventions — **hold ⌘ and scroll to zoom to the pointer, hold space and drag to pan** — and a rectified strip loupe. The fit and the composite sit **side by side and always have** — the result pane re-renders as you drag, which is how a corner gets judged, so it is the layout rather than a mode you can switch off. Then an on-by-default realism pass that colour-matches the source to the photo's light, **Save** (or **Render**, for a video) into the project folder (`--out-dir`), and a **Send to Claude** button that reaches you through the plugin's own MCP server. The UI is a hand port of the project's Figma design file — dark only.
8
+ **What ships (v0.40):** a local browser UI (`scripts/ui.py`) that walks the designer through the whole job — pick the photo and the screen source, which may be an image **or a video** (recent Desktop/Downloads images, drag-drop, browse, path, or a **Figma frame link**), auto-detect the screen as a starting position — and when detection cannot tell which region is a screen, **Point at screen**: one click inside it and the detector uses that point — or, if this photograph has been fitted before, **the fit it was saved with comes back** as the starting position instead of a detection, recognised by the photo's own pixels so a rename or a drag-drop still match — and every save also writes a **portable `.fit.json` beside the mockup** that can be dropped back onto the page later, which is how a fit survives a re-export, another machine, or someone else's hands — then **match the four edges** (drag an edge's middle to slide it, near an end to pivot; corners still draggable) with canvas navigation that follows the usual conventions — **hold ⌘ and scroll to zoom to the pointer, hold space and drag to pan** — and a rectified strip loupe. The fit and the composite sit **side by side and always have** — the result pane re-renders as you drag, which is how a corner gets judged, so it is the layout rather than a mode you can switch off. Then an on-by-default realism pass that colour-matches the source to the photo's light, **Save** (or **Render**, for a video) into the project folder (`--out-dir`), and a **Send to Claude** button that reaches you through the plugin's own MCP server. The UI is a hand port of the project's Figma design file — dark only.
9
9
 
10
10
  The geometry is exact (`warp.py`); the detection is advisory (`detect.py`) and the human corrects it. **When a detection is wrong and you want to know why**, ask for the candidate list: `POST /api/detect {"trace": true}` writes `<session>/candidates.json`, or run `python3 scripts/detect.py --photo P --out-corners /tmp/c.json --trace /tmp/t.json` (add `--click X,Y`). Every candidate quad is in there with its score and whether it was accepted, rejected, never reached, or filtered out by the click — which is what separates "the screen was never proposed" from "it was proposed and something else won".
11
11