screengraft 0.25.2 → 0.38.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +35 -0
- package/package.json +1 -1
- package/scripts/detect.py +335 -36
- package/scripts/fitfile.py +140 -0
- package/scripts/fits.py +151 -0
- package/scripts/ui.py +398 -42
- package/scripts/warp.py +155 -7
- package/skills/inject-screenshot/SKILL.md +3 -3
- package/ui/index.html +723 -67
package/README.md
CHANGED
|
@@ -38,6 +38,22 @@ prototype, then put the recording inside a real photograph.
|
|
|
38
38
|
You match the edges on one frame and every frame gets that same geometry — the
|
|
39
39
|
photograph is still, so there is nothing to track and nothing to drift. Output
|
|
40
40
|
is H.264 at CRF 16 or ProRes 422 HQ.
|
|
41
|
+
- **Watch it before you render it.** Press Play and the clip runs on the photo
|
|
42
|
+
immediately — the browser warps it onto the same four corners, with the same
|
|
43
|
+
corner radius, and approximates the emissive blend. Nothing to wait for, and
|
|
44
|
+
it follows the quad while you drag. **Render preview** composites a few
|
|
45
|
+
seconds through the real pipeline for when the grade, the grain and the true
|
|
46
|
+
blend are what you need to judge.
|
|
47
|
+
- **It remembers a photograph.** Fit a photo once and save, and the next run on
|
|
48
|
+
that photograph starts from those exact corners instead of a detection —
|
|
49
|
+
including a second screenshot into the same shot, which is the common case.
|
|
50
|
+
The photo is recognised by its pixels, not its name, so a rename or a
|
|
51
|
+
drag-drop from a different folder still match.
|
|
52
|
+
- **And the fit is a file you can keep.** Every save writes `<mockup>.fit.json`
|
|
53
|
+
beside the output. Drag it back onto the page and the corners come back — on
|
|
54
|
+
another machine, from a colleague, or onto the same scene exported again,
|
|
55
|
+
which the pixel key cannot recognise. It says which photograph it was made
|
|
56
|
+
for, and stretches nothing without telling you.
|
|
41
57
|
- **You confirm every fit.** Detection is advisory and says so; you drag the four
|
|
42
58
|
edges onto the glass with a magnified loupe. A silent misdetection producing a
|
|
43
59
|
confident, wrong result is the one failure this tool refuses to have.
|
|
@@ -154,6 +170,25 @@ What survives either way is the `result.json` sidecar: a few hundred bytes
|
|
|
154
170
|
recording the corners, radius, grade and blend of that fit. It reproduces a
|
|
155
171
|
composite exactly, and it is the first thing a bug report should include.
|
|
156
172
|
|
|
173
|
+
**A detection trace**, when you ask for one, is written to
|
|
174
|
+
`<session>/candidates.json`: every candidate quad the detectors generated, with
|
|
175
|
+
its score and whether it was accepted, rejected, never reached, or filtered out
|
|
176
|
+
by your click. Ask for it with `POST /api/detect {"trace": true}`, or from the
|
|
177
|
+
CLI with `scripts/detect.py --trace FILE`. Nothing produces it unless asked, and
|
|
178
|
+
it is the right thing to attach to a bug report about detection.
|
|
179
|
+
|
|
180
|
+
**The fit file** is written beside every mockup as `<name>.fit.json` — four
|
|
181
|
+
corners, the corner radius, the device, and which photograph it was made for. It
|
|
182
|
+
is yours: keep it with the project, send it to someone, drop it back on the page
|
|
183
|
+
to reuse the fit. Geometry only; the grade and blend of a *composite* live in
|
|
184
|
+
`result.json`.
|
|
185
|
+
|
|
186
|
+
**Remembered fits** live in `~/.screengraft/fits.json`, outside any session so a
|
|
187
|
+
sweep cannot take them. Each entry is the four corners, the radius fraction and
|
|
188
|
+
the device, keyed by a hash of the photograph's decoded pixels — no image data,
|
|
189
|
+
no absolute paths, a couple of hundred bytes each, and the oldest are dropped
|
|
190
|
+
past 500. Delete the file to forget every fit; nothing else depends on it.
|
|
191
|
+
|
|
157
192
|
Sidecars written by v0.23.0–v0.25.1 may name a dragged-in source that release
|
|
158
193
|
deleted. Those cannot be repaired — the bytes are gone — but screengraft now
|
|
159
194
|
marks them `"source_retained": false` rather than leaving them looking valid.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "screengraft",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.38.0",
|
|
4
4
|
"description": "Put a UI screenshot or screen recording onto a photographed device screen with the perspective exactly right \u2014 a homography you confirm by hand, not a generative guess.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"mockup",
|
package/scripts/detect.py
CHANGED
|
@@ -54,6 +54,22 @@ MIN_AREA_FRAC = 0.01
|
|
|
54
54
|
MAX_AREA_FRAC = 0.60
|
|
55
55
|
MIN_FILL = 0.60 # the source blob must fill this much of the final quad
|
|
56
56
|
MIN_SIDE_RATIO = 0.08 # reject slivers: shortest side vs longest, after perspective
|
|
57
|
+
# How unequal two OPPOSITE sides of the quad may be. A screen is a rectangle, and
|
|
58
|
+
# the image of a rectangle under perspective foreshortens one end relative to the
|
|
59
|
+
# other -- but not without limit at the angles anyone photographs a device from.
|
|
60
|
+
#
|
|
61
|
+
# Measured 10 Sep 2026 against seven hand-placed labels (the first real ground
|
|
62
|
+
# truth this project has had): every true screen sits at 1.01-1.05, the two
|
|
63
|
+
# usable detections at 1.04 and 1.11, and a quad returned CONFIDENTLY with one
|
|
64
|
+
# corner collapsed ~800px into the middle of the screen sits at 2.26. Worst
|
|
65
|
+
# correct 1.11 against best wrong 2.26 is a 2.0x margin.
|
|
66
|
+
#
|
|
67
|
+
# Set LOOSE at 1.9, not near the data. All seven are phones at modest angles;
|
|
68
|
+
# a laptop or a monitor shot from the side genuinely foreshortens more, and this
|
|
69
|
+
# check can only ever ADD refusals -- the cost of being wrong here is an honest
|
|
70
|
+
# abstention and a manual fit, never a bad composite. Re-derive it on a wider
|
|
71
|
+
# set of angles and device classes before tightening.
|
|
72
|
+
MAX_OPPOSITE_RATIO = 1.9
|
|
57
73
|
# Size stops being a virtue past this fraction of the frame. score_contour used
|
|
58
74
|
# to reward area linearly, so on a real photo (7 Sep 2026) the sunlit TABLE at
|
|
59
75
|
# 34% of the frame beat the phone screen at 9% by 3.6x on that term alone, and
|
|
@@ -246,7 +262,26 @@ def order_quad(pts: np.ndarray) -> np.ndarray:
|
|
|
246
262
|
return np.roll(pts, -start, axis=0)
|
|
247
263
|
|
|
248
264
|
|
|
249
|
-
|
|
265
|
+
# The Canny path traces a RING -- both sides of one boundary -- and catches
|
|
266
|
+
# content edges drawn inside the screen along with it. Fitting a line to all of
|
|
267
|
+
# that lands it between the glass edge and whatever is drawn near it, which is
|
|
268
|
+
# the 78px failure that banned refinement from this path at v0.13.0.
|
|
269
|
+
#
|
|
270
|
+
# The fix is to pick a rail rather than to ban the fit. Of the points assigned
|
|
271
|
+
# to one edge, keep those within `RAIL_BAND_K` close-kernels of the outermost;
|
|
272
|
+
# the ring's two rails are a kernel apart by construction, and content edges are
|
|
273
|
+
# far deeper. Measured 11 Sep 2026 on the top-scoring edge candidate of five
|
|
274
|
+
# hand-labelled photographs, worst per-edge median residual over the quad's
|
|
275
|
+
# diagonal: 0.0673 fitting everything, 0.0052 after rail selection -- and the
|
|
276
|
+
# quad that refinement had pushed from 12% to 40% lands at 0.4%.
|
|
277
|
+
#
|
|
278
|
+
# 2 is the ring's own thickness rounded up, not a fitted number: one kernel
|
|
279
|
+
# separates the rails, the second is slack for the Canny width itself. A region
|
|
280
|
+
# silhouette has one rail, so this keeps every point and changes nothing.
|
|
281
|
+
RAIL_BAND_K = 2.0
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def refine_corners(contour, quad: np.ndarray, rail_band: float = 0.0):
|
|
250
285
|
"""
|
|
251
286
|
Re-derive corners by intersecting fitted edge lines.
|
|
252
287
|
|
|
@@ -269,7 +304,10 @@ def refine_corners(contour, quad: np.ndarray):
|
|
|
269
304
|
return quad, False
|
|
270
305
|
|
|
271
306
|
# Distance from every point to every edge; nearest edge wins the point.
|
|
272
|
-
|
|
307
|
+
# `outward` is the same distance signed so that positive means away from the
|
|
308
|
+
# quad's own centre, which is what lets the outermost rail be picked below.
|
|
309
|
+
centre = quad.mean(axis=0)
|
|
310
|
+
dists, projections, outward = [], [], []
|
|
273
311
|
for i in range(4):
|
|
274
312
|
a, b = quad[i], quad[(i + 1) % 4]
|
|
275
313
|
ab = b - a
|
|
@@ -278,17 +316,29 @@ def refine_corners(contour, quad: np.ndarray):
|
|
|
278
316
|
return quad, False
|
|
279
317
|
u = ab / length
|
|
280
318
|
rel = pts - a
|
|
281
|
-
|
|
319
|
+
cross = u[0] * rel[:, 1] - u[1] * rel[:, 0]
|
|
320
|
+
rc = centre - a
|
|
321
|
+
sign = -np.sign(u[0] * rc[1] - u[1] * rc[0])
|
|
322
|
+
dists.append(np.abs(cross))
|
|
282
323
|
projections.append((rel @ ab) / (length ** 2))
|
|
324
|
+
outward.append(sign * cross)
|
|
283
325
|
nearest = np.argmin(np.vstack(dists), axis=0)
|
|
284
326
|
|
|
285
327
|
margin = (1.0 - EDGE_MIDDLE) / 2.0
|
|
286
328
|
lines = []
|
|
287
329
|
for i in range(4):
|
|
288
330
|
t = projections[i]
|
|
289
|
-
|
|
331
|
+
chosen = (nearest == i) & (t > margin) & (t < 1.0 - margin)
|
|
332
|
+
sel = pts[chosen]
|
|
290
333
|
if len(sel) < 10:
|
|
291
334
|
return quad, False
|
|
335
|
+
if rail_band > 0:
|
|
336
|
+
# 95th percentile rather than the maximum: one stray point further
|
|
337
|
+
# out than the rail would otherwise drag the window off it.
|
|
338
|
+
far = outward[i][chosen]
|
|
339
|
+
keep = far >= (float(np.percentile(far, 95)) - rail_band)
|
|
340
|
+
if int(keep.sum()) >= 10:
|
|
341
|
+
sel = sel[keep]
|
|
292
342
|
vx, vy, x0, y0 = cv2.fitLine(
|
|
293
343
|
sel.astype(np.float32), cv2.DIST_HUBER, 0, 0.01, 0.01
|
|
294
344
|
).ravel()
|
|
@@ -481,6 +531,20 @@ def validate_quad(corners: np.ndarray, contour, img_area: float, img_shape=None)
|
|
|
481
531
|
if min(sides) < MIN_SIDE_RATIO * max(sides):
|
|
482
532
|
return False, ("final quad is a sliver (shortest side %.0f%% of the longest) "
|
|
483
533
|
% (100 * min(sides) / max(sides)))
|
|
534
|
+
# A collapsed corner passes every check above: the quad stays convex, no side
|
|
535
|
+
# is a sliver, the area is in range and the blob still fills it. What gives it
|
|
536
|
+
# away is that one PAIR of opposite sides is wildly unequal while the other is
|
|
537
|
+
# not -- which is what a rectangle cannot do, at any angle a device is
|
|
538
|
+
# photographed from. Found on a real photograph where three corners sat on the
|
|
539
|
+
# glass and the fourth was in the middle of the screen, returned with no
|
|
540
|
+
# abstention: the one failure this tool says it does not have.
|
|
541
|
+
opposite = max(max(sides[0], sides[2]) / max(min(sides[0], sides[2]), 1e-6),
|
|
542
|
+
max(sides[1], sides[3]) / max(min(sides[1], sides[3]), 1e-6))
|
|
543
|
+
if opposite > MAX_OPPOSITE_RATIO:
|
|
544
|
+
return False, ("opposite sides of the final quad differ by %.1fx (ceiling "
|
|
545
|
+
"%.1fx) — a rectangle in perspective does not do that, so a "
|
|
546
|
+
"corner has been pulled off the screen"
|
|
547
|
+
% (opposite, MAX_OPPOSITE_RATIO))
|
|
484
548
|
if contour is not None:
|
|
485
549
|
fill = float(cv2.contourArea(contour)) / max(area, 1e-6)
|
|
486
550
|
if fill < MIN_FILL:
|
|
@@ -504,8 +568,24 @@ def contains(quad, point) -> bool:
|
|
|
504
568
|
(float(point[0]), float(point[1])), False) >= 0
|
|
505
569
|
|
|
506
570
|
|
|
571
|
+
def _tag_json(tag):
|
|
572
|
+
"""Whatever a detector used to label a candidate: a tone band, a threshold."""
|
|
573
|
+
if isinstance(tag, (tuple, list)):
|
|
574
|
+
return [int(t) for t in tag]
|
|
575
|
+
return int(tag) if isinstance(tag, (int, float, np.integer, np.floating)) else tag
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def _trace_row(method, score, quad, tag, verdict, why=""):
|
|
579
|
+
return {"method": method,
|
|
580
|
+
"score": round(float(score), 5),
|
|
581
|
+
"tag": _tag_json(tag),
|
|
582
|
+
"quad": [[round(float(x), 1), round(float(y), 1)] for x, y in quad],
|
|
583
|
+
"verdict": verdict,
|
|
584
|
+
"why": why}
|
|
585
|
+
|
|
586
|
+
|
|
507
587
|
def _finalize(candidates, img_area: float, refine: bool = True, img_shape=None,
|
|
508
|
-
click=None):
|
|
588
|
+
click=None, trace=None, method=None, rail_band: float = 0.0):
|
|
509
589
|
"""Best candidate that survives refinement AND validation.
|
|
510
590
|
|
|
511
591
|
Walks candidates best-score-first rather than trusting the top one: a
|
|
@@ -513,14 +593,29 @@ def _finalize(candidates, img_area: float, refine: bool = True, img_shape=None,
|
|
|
513
593
|
miss, not a result, and the next candidate deserves a look before the
|
|
514
594
|
detector gives up.
|
|
515
595
|
|
|
516
|
-
`refine`
|
|
517
|
-
is a filled region's
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
596
|
+
`refine` used to be off for the Canny path, and `rail_band` is why it no
|
|
597
|
+
longer is. refine_corners() assumes the contour is a filled region's
|
|
598
|
+
silhouette, where each side has one long straight run to fit; a Canny
|
|
599
|
+
contour is a ring tracing both sides of an edge, with content edges caught
|
|
600
|
+
inside it, so the per-edge line fits picked up the wrong points -- measured
|
|
601
|
+
on the gradient-screen mockup at v0.13.0, 78px off an otherwise correct
|
|
602
|
+
quad.
|
|
603
|
+
|
|
604
|
+
The claim that came with that ban -- "the polygon approximation of a Canny
|
|
605
|
+
boundary is already on the edge, so there is nothing to recover" -- was
|
|
606
|
+
wrong, and it cost this project the largest single bucket of detection
|
|
607
|
+
error for eighteen releases. approxPolyDP puts its vertices ON the rounded
|
|
608
|
+
corner arcs, inside the true corners, and on real photographs that is
|
|
609
|
+
**9-11% of the screen's own width** lost from every side.
|
|
610
|
+
|
|
611
|
+
`rail_band` fixes the fit instead of banning it: of the points assigned to
|
|
612
|
+
one edge, only those within a couple of close-kernels of the outermost are
|
|
613
|
+
fitted, which is the ring's own outer rail and excludes content edges by
|
|
614
|
+
construction. Measured over the top-scoring edge candidate on five
|
|
615
|
+
hand-labelled photographs: 12% -> 0.4%, 11% -> 0.5%, 8% -> 3.4%, 9% -> 6.8%.
|
|
616
|
+
A region silhouette has one rail, so passing 0 there changes nothing.
|
|
523
617
|
"""
|
|
618
|
+
generated = candidates
|
|
524
619
|
if click is not None:
|
|
525
620
|
# Before ranking, not after: the point of the click is to shrink the
|
|
526
621
|
# field to the things the user actually pointed at, and let the existing
|
|
@@ -528,15 +623,59 @@ def _finalize(candidates, img_area: float, refine: bool = True, img_shape=None,
|
|
|
528
623
|
# re-confirm whichever candidate already won.
|
|
529
624
|
candidates = [c for c in candidates if contains(c[1], click)]
|
|
530
625
|
rejected = []
|
|
626
|
+
# The instrument. Every diagnosis this project has made about
|
|
627
|
+
# detection began by printing this list by hand, and TWICE it changed what
|
|
628
|
+
# the fix was: the 7 Sep rescoring that was the obvious answer and was
|
|
629
|
+
# wrong, because the true screen was never generated at all (693px away);
|
|
630
|
+
# and the click feature, where the correct quad turned out to be the top-scoring
|
|
631
|
+
# candidate containing the click, which deleted most of the planned work.
|
|
632
|
+
#
|
|
633
|
+
# What it exists to separate is RECALL from RANKING -- was the screen never
|
|
634
|
+
# proposed, or proposed and beaten? Those have opposite fixes and this
|
|
635
|
+
# project has never had the number. Off unless asked for, and it must not
|
|
636
|
+
# change the answer: it observes the same lists the walk below uses.
|
|
637
|
+
walked = {}
|
|
531
638
|
for cand in sorted(candidates, key=lambda c: c[0], reverse=True):
|
|
532
|
-
|
|
639
|
+
# NOT necessarily `cand`: pick_innermost steps inward from it while a
|
|
640
|
+
# comparably screen-like quad nests inside, so the quad that gets
|
|
641
|
+
# validated -- and returned -- can belong to a different candidate.
|
|
642
|
+
picked = pick_innermost(candidates, cand)
|
|
643
|
+
score, quad, contour, tag = picked
|
|
533
644
|
if refine:
|
|
534
|
-
refined, did_refine = refine_corners(contour, quad)
|
|
645
|
+
refined, did_refine = refine_corners(contour, quad, rail_band)
|
|
535
646
|
else:
|
|
536
647
|
refined, did_refine = quad, False
|
|
537
648
|
corners = order_quad(refined)
|
|
538
649
|
ok, why = validate_quad(corners, contour, img_area, img_shape)
|
|
650
|
+
if trace is not None:
|
|
651
|
+
# "accepted" by its own CHANNEL, which is not the same as winning:
|
|
652
|
+
# each of tone, edge and saturation accepts one, and detect() then
|
|
653
|
+
# arbitrates between them. The file's top-level `chosen` says which
|
|
654
|
+
# channel actually won, so a trace with three accepted rows and one
|
|
655
|
+
# chosen method is right, not a contradiction.
|
|
656
|
+
#
|
|
657
|
+
# And the walked candidate is not always the one whose quad came
|
|
658
|
+
# back. Measured on five real photographs: pick_innermost stepped
|
|
659
|
+
# inward on one of them, so a trace that credited `cand` would show
|
|
660
|
+
# the accepted row holding a quad that never became the answer,
|
|
661
|
+
# while the quad that DID sat in an `unreached` row. An instrument
|
|
662
|
+
# that misattributes the win is worse than no instrument -- a recall
|
|
663
|
+
# analysis reading it would draw the opposite conclusion.
|
|
664
|
+
if not ok:
|
|
665
|
+
walked[id(cand)] = ("rejected", why)
|
|
666
|
+
elif picked is cand:
|
|
667
|
+
walked[id(cand)] = ("accepted", "")
|
|
668
|
+
else:
|
|
669
|
+
walked[id(cand)] = ("accepted", "this candidate won the walk, but the "
|
|
670
|
+
"quad returned came from the nested "
|
|
671
|
+
"candidate marked supplied_the_answer")
|
|
672
|
+
walked[id(picked)] = ("supplied_the_answer",
|
|
673
|
+
"nested inside the accepted candidate; "
|
|
674
|
+
"pick_innermost stepped inward to it")
|
|
539
675
|
if ok:
|
|
676
|
+
if trace is not None:
|
|
677
|
+
trace.extend(_walk_rows(generated, candidates, walked, method, click,
|
|
678
|
+
refine, rail_band))
|
|
540
679
|
return {
|
|
541
680
|
"corners": [[round(float(x), 1), round(float(y), 1)] for x, y in corners],
|
|
542
681
|
"score": round(float(score), 5),
|
|
@@ -547,10 +686,58 @@ def _finalize(candidates, img_area: float, refine: bool = True, img_shape=None,
|
|
|
547
686
|
"rejected": rejected,
|
|
548
687
|
}
|
|
549
688
|
rejected.append({"score": round(float(score), 5), "tag": tag, "why": why})
|
|
689
|
+
if trace is not None:
|
|
690
|
+
trace.extend(_walk_rows(generated, candidates, walked, method, click, refine,
|
|
691
|
+
rail_band))
|
|
550
692
|
return None
|
|
551
693
|
|
|
552
694
|
|
|
553
|
-
def
|
|
695
|
+
def _walk_rows(generated, survived, walked, method, click, refine=True,
|
|
696
|
+
rail_band: float = 0.0):
|
|
697
|
+
"""One row per candidate the detector GENERATED, with what became of it.
|
|
698
|
+
|
|
699
|
+
Four verdicts, and the distinction between the last two is the whole point:
|
|
700
|
+
`filtered_by_click` and `unreached` both mean "never judged", but one is the
|
|
701
|
+
user narrowing the field and the other is a higher-scoring candidate winning
|
|
702
|
+
first. A recall analysis needs to see quads in all four states, so nothing
|
|
703
|
+
is dropped from this list -- it is written to a file, not to a page.
|
|
704
|
+
|
|
705
|
+
Each row carries the REFINED quad, not the polygon approximation the
|
|
706
|
+
candidate was built from. The walk refines before it validates, so a row
|
|
707
|
+
holding the raw quad describes a proposal the detector never actually
|
|
708
|
+
considered -- and the two differ by the whole rounded-corner inset, which on
|
|
709
|
+
real photographs is 9-11% of the screen's own width. That defect cost a day
|
|
710
|
+
on 11 Sep 2026: the recall column said the best proposal was 9% off
|
|
711
|
+
when refining it put it at 0.4%, so a corner-accuracy change looked like it
|
|
712
|
+
had bought nothing. An instrument that understates what a candidate is worth
|
|
713
|
+
is as wrong as one that misattributes the win, which this function already
|
|
714
|
+
goes to some trouble to avoid.
|
|
715
|
+
"""
|
|
716
|
+
# Identity, not equality: two candidates can hold equal numbers and be
|
|
717
|
+
# different proposals. `survived` is a filtered view of `generated`, so the
|
|
718
|
+
# objects are shared and `is` holds -- but only while nobody rebuilds the
|
|
719
|
+
# list, which is why `click is None` is checked explicitly below rather than
|
|
720
|
+
# inferred from the sets matching. A future `[transform(c) for c in ...]`
|
|
721
|
+
# would otherwise relabel every row `filtered_by_click` in silence.
|
|
722
|
+
kept = {id(c) for c in survived}
|
|
723
|
+
rows = []
|
|
724
|
+
for cand in generated:
|
|
725
|
+
score, quad, contour, tag = cand
|
|
726
|
+
if refine:
|
|
727
|
+
quad = order_quad(refine_corners(contour, quad, rail_band)[0])
|
|
728
|
+
if click is not None and id(cand) not in kept:
|
|
729
|
+
verdict, why = "filtered_by_click", "the click was not inside this quad"
|
|
730
|
+
else:
|
|
731
|
+
verdict, why = walked.get(id(cand), ("unreached",
|
|
732
|
+
"a higher-scoring candidate was accepted first"))
|
|
733
|
+
rows.append(_trace_row(method, score, quad, tag, verdict, why))
|
|
734
|
+
if click is not None:
|
|
735
|
+
for r in rows:
|
|
736
|
+
r["contains_click"] = r["verdict"] != "filtered_by_click"
|
|
737
|
+
return rows
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
def detect_tone(gray: np.ndarray, tone=None, click=None, trace=None):
|
|
554
741
|
"""Tone-band segmentation. Assumes the screen sits in a narrow tone band."""
|
|
555
742
|
h, w = gray.shape[:2]
|
|
556
743
|
img_area = float(h * w)
|
|
@@ -581,7 +768,8 @@ def detect_tone(gray: np.ndarray, tone=None, click=None):
|
|
|
581
768
|
score *= (16.0 / (hi - lo + 1)) ** 0.25
|
|
582
769
|
candidates.append((score, order_quad(quad), contour, (lo, hi)))
|
|
583
770
|
|
|
584
|
-
res = _finalize(candidates, img_area, img_shape=gray.shape[:2], click=click
|
|
771
|
+
res = _finalize(candidates, img_area, img_shape=gray.shape[:2], click=click,
|
|
772
|
+
trace=trace, method="tone")
|
|
585
773
|
if res is None:
|
|
586
774
|
return None
|
|
587
775
|
band = res.pop("_tag")
|
|
@@ -590,7 +778,7 @@ def detect_tone(gray: np.ndarray, tone=None, click=None):
|
|
|
590
778
|
return res
|
|
591
779
|
|
|
592
780
|
|
|
593
|
-
def detect_edges(gray: np.ndarray, click=None):
|
|
781
|
+
def detect_edges(gray: np.ndarray, click=None, trace=None):
|
|
594
782
|
"""Canny-and-quad detection — the document-scanner path.
|
|
595
783
|
|
|
596
784
|
Tone banding assumes a near-uniform screen, which breaks the moment the
|
|
@@ -629,7 +817,9 @@ def detect_edges(gray: np.ndarray, click=None):
|
|
|
629
817
|
if score > 0 and quad is not None:
|
|
630
818
|
candidates.append((score, order_quad(quad), c, (lo, hi)))
|
|
631
819
|
|
|
632
|
-
res = _finalize(candidates, img_area,
|
|
820
|
+
res = _finalize(candidates, img_area, img_shape=gray.shape[:2],
|
|
821
|
+
rail_band=RAIL_BAND_K * _odd(0.004 * short),
|
|
822
|
+
click=click, trace=trace, method="edge")
|
|
633
823
|
if res is None:
|
|
634
824
|
return None
|
|
635
825
|
thr = res.pop("_tag")
|
|
@@ -638,7 +828,7 @@ def detect_edges(gray: np.ndarray, click=None):
|
|
|
638
828
|
return res
|
|
639
829
|
|
|
640
830
|
|
|
641
|
-
def detect_saturation(bgr: np.ndarray, click=None):
|
|
831
|
+
def detect_saturation(bgr: np.ndarray, click=None, trace=None):
|
|
642
832
|
"""Neutral-region segmentation — the third detector, and the only one that
|
|
643
833
|
looks at colour.
|
|
644
834
|
|
|
@@ -675,7 +865,8 @@ def detect_saturation(bgr: np.ndarray, click=None):
|
|
|
675
865
|
candidates.append((score * (32.0 / max(thr, 1)) ** 0.25,
|
|
676
866
|
order_quad(quad), contour, thr))
|
|
677
867
|
|
|
678
|
-
res = _finalize(candidates, img_area, img_shape=sat.shape[:2], click=click
|
|
868
|
+
res = _finalize(candidates, img_area, img_shape=sat.shape[:2], click=click,
|
|
869
|
+
trace=trace, method="saturation")
|
|
679
870
|
if res is None:
|
|
680
871
|
return None
|
|
681
872
|
res.pop("_tag")
|
|
@@ -684,6 +875,26 @@ def detect_saturation(bgr: np.ndarray, click=None):
|
|
|
684
875
|
return res
|
|
685
876
|
|
|
686
877
|
|
|
878
|
+
def shape_tier(result) -> int:
|
|
879
|
+
"""How strongly a result's own SHAPE says "this is a screen": 0, 1 or 2.
|
|
880
|
+
|
|
881
|
+
2 confident radius -- four per-corner estimates agree within 50%
|
|
882
|
+
1 rounded -- they agree within MAX_RADIUS_SPREAD (2x)
|
|
883
|
+
0 nothing -- sharp, unmeasurable, or wildly inconsistent
|
|
884
|
+
|
|
885
|
+
Both thresholds already existed (measure_corner_radius's `confident`, and
|
|
886
|
+
has_rounded_corners); this only ranks them. Measured 11 Sep 2026 on ten
|
|
887
|
+
hand-labelled photographs, every channel result: **every tier-2 quad was on
|
|
888
|
+
the screen** (worst spread 0.24) and **every wrong quad was >= 1.43 or
|
|
889
|
+
unmeasurable** -- a 6x margin. Tier 1 alone does not separate them (a
|
|
890
|
+
correct tone quad at 1.42 against a wrong one at 1.43), which is exactly why
|
|
891
|
+
a tier-1 result must not outrank or veto a tier-2 one. It was doing both.
|
|
892
|
+
"""
|
|
893
|
+
if (result.get("corner_radius") or {}).get("confident"):
|
|
894
|
+
return 2
|
|
895
|
+
return 1 if has_rounded_corners(result) else 0
|
|
896
|
+
|
|
897
|
+
|
|
687
898
|
def has_rounded_corners(result) -> bool:
|
|
688
899
|
"""Did this quad's own outline actually curve at the corners?
|
|
689
900
|
|
|
@@ -702,7 +913,8 @@ def has_rounded_corners(result) -> bool:
|
|
|
702
913
|
return spread <= MAX_RADIUS_SPREAD
|
|
703
914
|
|
|
704
915
|
|
|
705
|
-
def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None
|
|
916
|
+
def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None,
|
|
917
|
+
trace=None):
|
|
706
918
|
"""Run both detectors; arbitrate on how the two quads nest.
|
|
707
919
|
|
|
708
920
|
The two fail on opposite things. Tone banding needs a tonally uniform
|
|
@@ -721,28 +933,42 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
721
933
|
which — 0.79 on the fixture (screen in body: take the inner, tone), 0.41
|
|
722
934
|
on a gradient screen (slab on screen: take the outer, edge). When they
|
|
723
935
|
don't nest at all, neither is a subregion of the other and tone wins,
|
|
724
|
-
since its assumptions being met is itself evidence
|
|
936
|
+
since its assumptions being met is itself evidence: a screen that lands
|
|
937
|
+
inside one narrow tone band is a screen.
|
|
725
938
|
|
|
726
939
|
Disagreement is reported, never silently resolved — the human confirms
|
|
727
940
|
the corners either way.
|
|
728
941
|
"""
|
|
942
|
+
# `trace`, when given, is a list this fills with one row per candidate every
|
|
943
|
+
# channel generated. It is an OBSERVER: nothing downstream reads it,
|
|
944
|
+
# and the answer is identical with and without -- which is asserted, because
|
|
945
|
+
# an instrument that perturbs what it measures is worse than none.
|
|
729
946
|
results = []
|
|
730
947
|
if method in ("auto", "tone"):
|
|
731
|
-
r = detect_tone(gray, tone, click=click)
|
|
948
|
+
r = detect_tone(gray, tone, click=click, trace=trace)
|
|
732
949
|
if r:
|
|
733
950
|
results.append(r)
|
|
734
951
|
if method in ("auto", "edge") and tone is None:
|
|
735
|
-
r = detect_edges(gray, click=click)
|
|
952
|
+
r = detect_edges(gray, click=click, trace=trace)
|
|
736
953
|
if r:
|
|
737
954
|
results.append(r)
|
|
738
955
|
if method in ("auto", "saturation") and tone is None and color is not None:
|
|
739
|
-
r = detect_saturation(color, click=click)
|
|
956
|
+
r = detect_saturation(color, click=click, trace=trace)
|
|
740
957
|
if r:
|
|
741
958
|
results.append(r)
|
|
742
959
|
|
|
743
960
|
if not results:
|
|
744
961
|
return None
|
|
962
|
+
return arbitrate(results, gray.shape[:2], click)
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
def arbitrate(results, shape, click=None):
|
|
966
|
+
"""Choose among the channels' accepted results and decide whether to abstain.
|
|
745
967
|
|
|
968
|
+
Split out of detect() on 11 Sep 2026 so the rules here can be tested on
|
|
969
|
+
hand-built results, the way _finalize's nested pick already is. `shape` is
|
|
970
|
+
the photograph's (h, w); the only thing this needs it for is the diagonal.
|
|
971
|
+
"""
|
|
746
972
|
# tone and saturation are the same algorithm on different channels, so they
|
|
747
973
|
# are compared to each other before anything else, on whether the region
|
|
748
974
|
# each found actually has rounded corners. A patch of table cut out of a
|
|
@@ -755,8 +981,9 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
755
981
|
# silhouette, so measure_corner_radius reports an artifact for it — 0.0px
|
|
756
982
|
# even when its quad is the correct one to 1.4px (measured on the
|
|
757
983
|
# gradient-screen fixture, where an earlier version of this filter threw
|
|
758
|
-
# away the right answer).
|
|
759
|
-
#
|
|
984
|
+
# away the right answer). Corner REFINEMENT is no longer skipped there --
|
|
985
|
+
# rail selection made the line fits safe on a ring, see refine_corners() --
|
|
986
|
+
# but measuring a RADIUS from a ring is a separate claim and still is.
|
|
760
987
|
region = [r for r in results if r["method"] in ("tone", "saturation")]
|
|
761
988
|
if len(region) == 2:
|
|
762
989
|
rounded = [r for r in region if has_rounded_corners(r)]
|
|
@@ -768,7 +995,7 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
768
995
|
e = next((r for r in results if r["method"] == "edge"), None)
|
|
769
996
|
sat = next((r for r in results if r["method"] == "saturation"), None)
|
|
770
997
|
why = "it was the only detector left after the rounded-corner filter" \
|
|
771
|
-
if len(results) == 1 else "
|
|
998
|
+
if len(results) == 1 else "its tone band assumption held, which is itself evidence"
|
|
772
999
|
if t and e:
|
|
773
1000
|
tq, eq = t["_corners_np"], e["_corners_np"]
|
|
774
1001
|
ta = float(cv2.contourArea(tq.astype(np.float32)))
|
|
@@ -782,13 +1009,32 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
782
1009
|
elif nested:
|
|
783
1010
|
best, why = t, ("the tone quad sits inside the edge quad at %.0f%% of "
|
|
784
1011
|
"its area — a screen inside a device body" % (ratio * 100))
|
|
1012
|
+
elif shape_tier(e) > shape_tier(t):
|
|
1013
|
+
# Not nested, and the edge quad's own shape says "screen" more
|
|
1014
|
+
# strongly than tone's does. Before 11 Sep 2026 tone won here
|
|
1015
|
+
# unconditionally, on the argument that its band assumption holding
|
|
1016
|
+
# was itself evidence -- and on two of ten labelled photographs that
|
|
1017
|
+
# handed the answer to a tone quad 159% and 249% off, with corner
|
|
1018
|
+
# spreads of 3.4 and 19.9, over an edge quad at 0% with a confident
|
|
1019
|
+
# radius. A band assumption is weaker evidence than four agreeing
|
|
1020
|
+
# corners; the tiers say so and this reads them.
|
|
1021
|
+
best, why = e, ("the two quads aren't nested and the edge quad's "
|
|
1022
|
+
"corners agree on a radius (tier %d) where tone's do "
|
|
1023
|
+
"not (tier %d)" % (shape_tier(e), shape_tier(t)))
|
|
785
1024
|
else:
|
|
786
|
-
best, why = t, "the two quads aren't nested;
|
|
1025
|
+
best, why = t, ("the two quads aren't nested; tone's band assumption "
|
|
1026
|
+
"holding is itself evidence that it found a screen")
|
|
787
1027
|
elif t is None and sat is not None:
|
|
788
1028
|
# tone was dropped for having sharp corners, or never fired. The
|
|
789
|
-
# surviving region detector
|
|
790
|
-
|
|
791
|
-
|
|
1029
|
+
# surviving region detector measured a real corner radius; edge usually
|
|
1030
|
+
# cannot -- but when it did, and more confidently, it wins the same way.
|
|
1031
|
+
if e is not None and shape_tier(e) > shape_tier(sat):
|
|
1032
|
+
best, why = e, ("the edge quad's corners agree on a radius (tier %d) "
|
|
1033
|
+
"where saturation's do not (tier %d)"
|
|
1034
|
+
% (shape_tier(e), shape_tier(sat)))
|
|
1035
|
+
else:
|
|
1036
|
+
best, why = sat, ("the saturation detector found a region with rounded "
|
|
1037
|
+
"corners where tone did not")
|
|
792
1038
|
else:
|
|
793
1039
|
best = t or e or sat
|
|
794
1040
|
if best is None:
|
|
@@ -800,7 +1046,7 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
800
1046
|
# somewhere else entirely should not erase the fact that two of them
|
|
801
1047
|
# landed together. Corroboration by any one independent method is the
|
|
802
1048
|
# evidence worth reporting.
|
|
803
|
-
diag = float(np.hypot(*
|
|
1049
|
+
diag = float(np.hypot(*shape))
|
|
804
1050
|
others = [r for r in results if r is not best]
|
|
805
1051
|
gaps = [(float(np.max(np.linalg.norm(best["_corners_np"]
|
|
806
1052
|
- r["_corners_np"], axis=1))) / diag, r)
|
|
@@ -848,9 +1094,15 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
848
1094
|
# not get a vote on whether the rounded thing is a screen. So the gap is
|
|
849
1095
|
# re-measured against peers that also found something screen-shaped; when
|
|
850
1096
|
# there are none, being alone is not evidence of being wrong.
|
|
851
|
-
|
|
1097
|
+
# ... and a peer may only veto a result whose shape evidence it at least
|
|
1098
|
+
# matches. On iPhone-2 (11 Sep 2026) a tone quad 143% off, "rounded" at a
|
|
1099
|
+
# spread of 1.43, vetoed a confident edge quad 0.3% off -- the bench's one
|
|
1100
|
+
# "good quad refused". A veto from weaker evidence is not a disagreement
|
|
1101
|
+
# between peers; it is noise outvoting a measurement.
|
|
1102
|
+
peers = [r for r in results if r is not best and has_rounded_corners(r)
|
|
1103
|
+
and shape_tier(r) >= shape_tier(best)]
|
|
852
1104
|
if peers:
|
|
853
|
-
diag = float(np.hypot(*
|
|
1105
|
+
diag = float(np.hypot(*shape))
|
|
854
1106
|
peer_gap = min(float(np.max(np.linalg.norm(best["_corners_np"]
|
|
855
1107
|
- r["_corners_np"], axis=1))) / diag
|
|
856
1108
|
for r in peers)
|
|
@@ -892,6 +1144,34 @@ def detect(gray: np.ndarray, tone=None, method="auto", color=None, click=None):
|
|
|
892
1144
|
return best
|
|
893
1145
|
|
|
894
1146
|
|
|
1147
|
+
def write_trace(path, photo_path, shape, click, result, rows):
|
|
1148
|
+
"""The candidate list as a file, which is the form a benchmark can read.
|
|
1149
|
+
|
|
1150
|
+
Deliberately not folded into the sidecar: `result.json` is the recipe for
|
|
1151
|
+
reproducing a composite and is contract-tested against compose()'s
|
|
1152
|
+
signature, so hanging diagnostics off it would couple two things that
|
|
1153
|
+
change for different reasons. This is a separate artefact of a separate
|
|
1154
|
+
run.
|
|
1155
|
+
"""
|
|
1156
|
+
by_verdict = {}
|
|
1157
|
+
for r in rows:
|
|
1158
|
+
by_verdict[r["verdict"]] = by_verdict.get(r["verdict"], 0) + 1
|
|
1159
|
+
with open(path, "w") as f:
|
|
1160
|
+
json.dump({
|
|
1161
|
+
"photo": photo_path,
|
|
1162
|
+
"size": [int(shape[1]), int(shape[0])],
|
|
1163
|
+
"click": list(click) if click else None,
|
|
1164
|
+
"chosen": None if result is None else {
|
|
1165
|
+
"method": result.get("method"),
|
|
1166
|
+
"corners": result.get("corners"),
|
|
1167
|
+
"abstained": bool(result.get("abstained")),
|
|
1168
|
+
},
|
|
1169
|
+
"counts": {"candidates": len(rows), **by_verdict},
|
|
1170
|
+
"candidates": rows,
|
|
1171
|
+
}, f, indent=1)
|
|
1172
|
+
return by_verdict
|
|
1173
|
+
|
|
1174
|
+
|
|
895
1175
|
def main() -> None:
|
|
896
1176
|
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
897
1177
|
ap.add_argument("--photo", required=True)
|
|
@@ -899,6 +1179,9 @@ def main() -> None:
|
|
|
899
1179
|
ap.add_argument("--out-overlay", help="Optional PNG showing the detected quad on the photo")
|
|
900
1180
|
ap.add_argument("--out-zooms", help="Optional directory for 2x corner close-ups")
|
|
901
1181
|
ap.add_argument("--tone", help="Skip the sweep and force one band, e.g. --tone 20,40")
|
|
1182
|
+
ap.add_argument("--click", help="A point inside the screen, X,Y in photo pixels")
|
|
1183
|
+
ap.add_argument("--trace", help="Write every candidate quad, with its score and "
|
|
1184
|
+
"what became of it, to this JSON file")
|
|
902
1185
|
args = ap.parse_args()
|
|
903
1186
|
|
|
904
1187
|
photo = cv2.imread(args.photo, cv2.IMREAD_COLOR)
|
|
@@ -914,7 +1197,23 @@ def main() -> None:
|
|
|
914
1197
|
except ValueError:
|
|
915
1198
|
sys.exit("error: --tone must be LO,HI (e.g. 20,40)")
|
|
916
1199
|
|
|
917
|
-
|
|
1200
|
+
click = None
|
|
1201
|
+
if args.click:
|
|
1202
|
+
try:
|
|
1203
|
+
click = tuple(float(v) for v in args.click.split(","))
|
|
1204
|
+
if len(click) != 2:
|
|
1205
|
+
raise ValueError
|
|
1206
|
+
except ValueError:
|
|
1207
|
+
sys.exit("error: --click must be X,Y (e.g. --click 850,637)")
|
|
1208
|
+
|
|
1209
|
+
trace = [] if args.trace else None
|
|
1210
|
+
result = detect(gray, tone, color=photo, click=click, trace=trace)
|
|
1211
|
+
if trace is not None:
|
|
1212
|
+
# Written whether or not anything was found: a run that found NOTHING is
|
|
1213
|
+
# the most interesting one to inspect, and it is exactly the run that
|
|
1214
|
+
# would otherwise leave no trace at all.
|
|
1215
|
+
write_trace(args.trace, args.photo, photo.shape, click, result, trace)
|
|
1216
|
+
print(f"trace: {len(trace)} candidates -> {args.trace}", file=sys.stderr)
|
|
918
1217
|
if result is None:
|
|
919
1218
|
sys.exit(
|
|
920
1219
|
"error: no screen-like quad found. The screen's tone probably isn't "
|