discface 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- discface/__init__.py +17 -0
- discface/__main__.py +5 -0
- discface/_disc.py +210 -0
- discface/_enhance.py +62 -0
- discface/_imaging.py +362 -0
- discface/_orient.py +95 -0
- discface/api.py +201 -0
- discface/cli.py +96 -0
- discface/py.typed +0 -0
- discface-0.1.0.dist-info/METADATA +163 -0
- discface-0.1.0.dist-info/RECORD +14 -0
- discface-0.1.0.dist-info/WHEEL +5 -0
- discface-0.1.0.dist-info/entry_points.txt +2 -0
- discface-0.1.0.dist-info/top_level.txt +1 -0
discface/__init__.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""discface: from a photo of a CD / DVD to the flat, upright disc.
|
|
2
|
+
|
|
3
|
+
import discface
|
|
4
|
+
|
|
5
|
+
disc = discface.extract('photo.jpg') # locate, rectify, turn upright
|
|
6
|
+
disc.save('disc.png') # 2048 x 2048 RGBA, transparent outside the rim and in the hole
|
|
7
|
+
|
|
8
|
+
geometry = discface.locate('photo.jpg') # the single steps
|
|
9
|
+
flat = discface.rectify('photo.jpg', geometry)
|
|
10
|
+
o = discface.orient(flat, enhance='always')
|
|
11
|
+
upright = flat.rotate(o.rotation)
|
|
12
|
+
"""
|
|
13
|
+
from .api import Disc, DiscGeometry, DiscNotFound, Orientation, extract, load_detector, locate, orient, rectify
|
|
14
|
+
|
|
15
|
+
__version__ = '0.1.0'
|
|
16
|
+
__all__ = ['Disc', 'DiscGeometry', 'DiscNotFound', 'Orientation', 'extract', 'load_detector', 'locate', 'orient',
|
|
17
|
+
'rectify', '__version__']
|
discface/__main__.py
ADDED
discface/_disc.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Finding the disc in a photo and rectifying it (pure geometry, no learning).
|
|
2
|
+
|
|
3
|
+
1. Edges of the image (difference to the background colour + grey-level Canny) at half resolution.
|
|
4
|
+
2. Ellipse candidates from the edge contours; the rim is the large, well-supported one, the hub rings are the
|
|
5
|
+
small ones near its centre.
|
|
6
|
+
3. The true disc centre is extrapolated from the hub ellipses: centres of projected concentric circles drift
|
|
7
|
+
linearly with r^2.
|
|
8
|
+
4. The exact homography maps the rim conic to a circle: the polar of the centre w.r.t. the rim is the vanishing
|
|
9
|
+
line of the disc plane; the remaining affine part is a Cholesky factor of the conic.
|
|
10
|
+
5. The centre hole is the innermost ring covered over at least half its circumference in the rectified plane.
|
|
11
|
+
6. Lanczos4 resampling of the disc annulus into a square image with an anti-aliased alpha mask.
|
|
12
|
+
"""
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
from ._imaging import (resize_linear, normalize_minmax_u8, gaussian_blur, canny, rgb2gray, find_contours,
|
|
16
|
+
fit_ellipse, ellipse_pixels, warp_perspective)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def filter_and_detect_edges(img, scale=0.5):
|
|
20
|
+
small = resize_linear(img, scale)
|
|
21
|
+
bg = np.median(np.concatenate([
|
|
22
|
+
small[:20, :].reshape(-1, 3),
|
|
23
|
+
small[-20:, :].reshape(-1, 3),
|
|
24
|
+
small[:, :20].reshape(-1, 3),
|
|
25
|
+
small[:, -20:].reshape(-1, 3)
|
|
26
|
+
]), axis=0)
|
|
27
|
+
diff = np.linalg.norm(small.astype(np.float32) - bg, axis=2)
|
|
28
|
+
diff_norm = normalize_minmax_u8(diff)
|
|
29
|
+
blurred_diff = gaussian_blur(diff_norm, 9, 2)
|
|
30
|
+
edges_diff = canny(blurred_diff, 30, 80)
|
|
31
|
+
|
|
32
|
+
gray = rgb2gray(small)
|
|
33
|
+
blurred_gray = gaussian_blur(gray, 5, 1.2)
|
|
34
|
+
edges_gray = canny(blurred_gray, 20, 60)
|
|
35
|
+
|
|
36
|
+
combined_edges = np.maximum(edges_diff, edges_gray)
|
|
37
|
+
return combined_edges, small
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def detect_candidate_ellipses(edges):
|
|
41
|
+
sh, sw = edges.shape[:2]
|
|
42
|
+
cnts = find_contours(edges)
|
|
43
|
+
candidates = []
|
|
44
|
+
for c in cnts:
|
|
45
|
+
if len(c) >= 30:
|
|
46
|
+
box = fit_ellipse(c)
|
|
47
|
+
if box is None:
|
|
48
|
+
continue
|
|
49
|
+
(cx, cy), (d1, d2), ang = box
|
|
50
|
+
a, b = max(d1, d2) / 2.0, min(d1, d2) / 2.0
|
|
51
|
+
if a > 15 and b > 15 and (b / a) > 0.7:
|
|
52
|
+
if 0.15 * sw < cx < 0.85 * sw and 0.15 * sh < cy < 0.85 * sh:
|
|
53
|
+
py, px = ellipse_pixels(box, (sh, sw), thickness=2)
|
|
54
|
+
votes = np.count_nonzero(edges[py, px])
|
|
55
|
+
perim = 2 * np.pi * np.sqrt((a**2 + b**2) / 2.0)
|
|
56
|
+
score = votes / (perim + 1e-5)
|
|
57
|
+
if score > 0.25:
|
|
58
|
+
candidates.append({
|
|
59
|
+
'box': box,
|
|
60
|
+
'cx': cx,
|
|
61
|
+
'cy': cy,
|
|
62
|
+
'a': a,
|
|
63
|
+
'b': b,
|
|
64
|
+
'ang': ang,
|
|
65
|
+
'score': score
|
|
66
|
+
})
|
|
67
|
+
return candidates
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def ellipse_to_conic(box):
|
|
71
|
+
(cx, cy), (d1, d2), ang = box
|
|
72
|
+
t = np.radians(ang)
|
|
73
|
+
R = np.array([[np.cos(t), -np.sin(t)], [np.sin(t), np.cos(t)]])
|
|
74
|
+
A = np.eye(3)
|
|
75
|
+
A[:2, :2] = R.T
|
|
76
|
+
A[:2, 2] = -R.T @ np.array([cx, cy])
|
|
77
|
+
return A.T @ np.diag([4.0 / d1**2, 4.0 / d2**2, -1.0]) @ A
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def homography_from_rim_and_center(outer_box, center, size=2048, margin=32):
|
|
81
|
+
(cx, cy), (d1, d2), _ = outer_box
|
|
82
|
+
s = 2.0 / (d1 + d2)
|
|
83
|
+
N = np.array([[s, 0, -s * cx], [0, s, -s * cy], [0, 0, 1.0]])
|
|
84
|
+
Ni = np.linalg.inv(N)
|
|
85
|
+
C = Ni.T @ ellipse_to_conic(outer_box) @ Ni
|
|
86
|
+
C /= np.abs(C).max()
|
|
87
|
+
c = N @ np.array([center[0], center[1], 1.0])
|
|
88
|
+
|
|
89
|
+
# the polar of the disc centre w.r.t. the rim conic is the vanishing line of the disc plane
|
|
90
|
+
l = C @ c
|
|
91
|
+
l /= l[2]
|
|
92
|
+
H_proj = np.array([[1, 0, 0], [0, 1, 0], [l[0], l[1], 1.0]])
|
|
93
|
+
Hpi = np.linalg.inv(H_proj)
|
|
94
|
+
C_aff = Hpi.T @ C @ Hpi
|
|
95
|
+
|
|
96
|
+
# remaining affine ellipse -> unit circle
|
|
97
|
+
M, m, k = C_aff[:2, :2], C_aff[:2, 2], C_aff[2, 2]
|
|
98
|
+
x0 = -np.linalg.solve(M, m)
|
|
99
|
+
L = np.linalg.cholesky(M / (m @ np.linalg.solve(M, m) - k))
|
|
100
|
+
H_aff = np.eye(3)
|
|
101
|
+
H_aff[:2, :2] = L.T
|
|
102
|
+
H_aff[:2, 2] = -L.T @ x0
|
|
103
|
+
H_unit = H_aff @ H_proj @ N
|
|
104
|
+
|
|
105
|
+
# in-plane rotation: keep the image "down" direction pointing down
|
|
106
|
+
c_img = np.linalg.inv(H_unit) @ np.array([0, 0, 1.0])
|
|
107
|
+
c_img /= c_img[2]
|
|
108
|
+
q = H_unit @ np.array([c_img[0], c_img[1] + 0.3 * (d1 + d2) / 2.0, 1.0])
|
|
109
|
+
q /= q[2]
|
|
110
|
+
phi = np.arctan2(q[0], q[1])
|
|
111
|
+
R = np.array([[np.cos(phi), -np.sin(phi), 0], [np.sin(phi), np.cos(phi), 0], [0, 0, 1.0]])
|
|
112
|
+
|
|
113
|
+
tc = size / 2.0
|
|
114
|
+
tr = tc - margin
|
|
115
|
+
S = np.array([[tr, 0, tc], [0, tr, tc], [0, 0, 1.0]])
|
|
116
|
+
H = S @ R @ H_unit
|
|
117
|
+
return H / H[2, 2]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def select_hole_ring(edges, outer_box, center, size=2048, margin=32, min_coverage=0.5):
|
|
121
|
+
# Hough-like radius accumulator around the consensus centre in the rectified plane:
|
|
122
|
+
# a real ring must be covered over at least half of its circumference; the hole is the innermost one
|
|
123
|
+
# inside the search window given by the disc standard (15 mm hole on a 120 mm disc -> 0.125 * r)
|
|
124
|
+
H = homography_from_rim_and_center(outer_box, center, size, margin)
|
|
125
|
+
tc = size / 2.0
|
|
126
|
+
tr = tc - margin
|
|
127
|
+
ys, xs = np.nonzero(edges)
|
|
128
|
+
P = H @ np.stack([xs.astype(np.float64), ys.astype(np.float64), np.ones(len(xs))])
|
|
129
|
+
x, y = P[0] / P[2] - tc, P[1] / P[2] - tc
|
|
130
|
+
r = np.hypot(x, y)
|
|
131
|
+
ang_bin = ((np.arctan2(y, x) + np.pi) / (2 * np.pi) * 360).astype(int) % 360
|
|
132
|
+
|
|
133
|
+
radii = np.arange(0.10 * tr, 0.15 * tr, 0.5)
|
|
134
|
+
coverage = np.array([len(np.unique(ang_bin[np.abs(r - R) < 1.5])) / 360.0 for R in radii])
|
|
135
|
+
for i in range(1, len(radii) - 1):
|
|
136
|
+
if coverage[i] >= min_coverage and coverage[i] >= coverage[i - 1] and coverage[i] >= coverage[i + 1]:
|
|
137
|
+
t = np.linspace(0, 2 * np.pi, 360, endpoint=False)
|
|
138
|
+
Q = np.linalg.inv(H) @ np.stack([tc + radii[i] * np.cos(t), tc + radii[i] * np.sin(t), np.ones_like(t)])
|
|
139
|
+
return fit_ellipse(np.stack([Q[0] / Q[2], Q[1] / Q[2]], 1))
|
|
140
|
+
return None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def select_outer_and_inner_ellipses(candidates, edges):
|
|
144
|
+
sh, sw = edges.shape[:2]
|
|
145
|
+
cd_cands = [c for c in candidates if c['a'] >= 45.0]
|
|
146
|
+
outers = [c for c in cd_cands if c['a'] > 0.25 * min(sw, sh)]
|
|
147
|
+
if not outers:
|
|
148
|
+
outers = [max(cd_cands, key=lambda c: c['a'])]
|
|
149
|
+
best_outer = max(outers, key=lambda c: (c['score'], c['a']))
|
|
150
|
+
|
|
151
|
+
inners = [
|
|
152
|
+
c for c in cd_cands
|
|
153
|
+
if 0.08 * best_outer['a'] <= c['a'] <= 0.40 * best_outer['a']
|
|
154
|
+
and np.hypot(c['cx'] - best_outer['cx'], c['cy'] - best_outer['cy']) < 85.0
|
|
155
|
+
]
|
|
156
|
+
|
|
157
|
+
for c in inners:
|
|
158
|
+
close_members = [o for o in inners if np.hypot(c['cx'] - o['cx'], c['cy'] - o['cy']) < 18.0]
|
|
159
|
+
c['consensus_weight'] = sum(o['score'] for o in close_members)
|
|
160
|
+
c['consensus_members'] = close_members
|
|
161
|
+
|
|
162
|
+
best_inner_anchor = max(inners, key=lambda c: c['consensus_weight'])
|
|
163
|
+
consensus_inners = best_inner_anchor['consensus_members']
|
|
164
|
+
|
|
165
|
+
shared_cx = float(np.median([c['cx'] for c in consensus_inners]))
|
|
166
|
+
shared_cy = float(np.median([c['cy'] for c in consensus_inners]))
|
|
167
|
+
|
|
168
|
+
# projected ellipse centres of concentric circles drift linearly with r^2 -> extrapolate to r = 0
|
|
169
|
+
rho2 = (float(np.median([c['a'] for c in consensus_inners])) / best_outer['a']) ** 2
|
|
170
|
+
center = (
|
|
171
|
+
(shared_cx - rho2 * best_outer['cx']) / (1.0 - rho2),
|
|
172
|
+
(shared_cy - rho2 * best_outer['cy']) / (1.0 - rho2)
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
outer_box = best_outer['box']
|
|
176
|
+
inner_box = select_hole_ring(edges, outer_box, center)
|
|
177
|
+
if inner_box is None:
|
|
178
|
+
hole_cands = [c for c in consensus_inners if c['a'] < 0.16 * best_outer['a']]
|
|
179
|
+
if hole_cands:
|
|
180
|
+
innermost = min(hole_cands, key=lambda c: c['a'])
|
|
181
|
+
else:
|
|
182
|
+
innermost = min(consensus_inners, key=lambda c: c['a'])
|
|
183
|
+
inner_box = innermost['box']
|
|
184
|
+
return outer_box, inner_box, center
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def rectify_and_project_to_circle(img, outer_box, inner_box, center, scale=0.5, size=2048, margin=32):
|
|
188
|
+
up = lambda b: ((b[0][0] / scale, b[0][1] / scale), (b[1][0] / scale, b[1][1] / scale), b[2])
|
|
189
|
+
outer_box, inner_box = up(outer_box), up(inner_box)
|
|
190
|
+
center = (center[0] / scale, center[1] / scale)
|
|
191
|
+
|
|
192
|
+
H = homography_from_rim_and_center(outer_box, center, size, margin)
|
|
193
|
+
|
|
194
|
+
tc = size / 2.0
|
|
195
|
+
tr = tc - margin
|
|
196
|
+
t = np.linspace(0, 2 * np.pi, 360, endpoint=False)
|
|
197
|
+
(icx, icy), (id1, id2), iang = inner_box
|
|
198
|
+
ia = np.radians(iang)
|
|
199
|
+
ex, ey = id1 / 2.0 * np.cos(t), id2 / 2.0 * np.sin(t)
|
|
200
|
+
hole = H @ np.stack([icx + ex * np.cos(ia) - ey * np.sin(ia), icy + ex * np.sin(ia) + ey * np.cos(ia), np.ones_like(t)])
|
|
201
|
+
hole_r = float(np.median(np.hypot(hole[0] / hole[2] - tc, hole[1] / hole[2] - tc)))
|
|
202
|
+
|
|
203
|
+
# analytic anti-aliased annulus (1px ramp) instead of a blurred hard mask
|
|
204
|
+
yy, xx = np.mgrid[0:size, 0:size].astype(np.float32)
|
|
205
|
+
d = np.hypot(xx - tc, yy - tc)
|
|
206
|
+
alpha = np.clip(tr - d + 0.5, 0, 1) * np.clip(d - hole_r + 0.5, 0, 1)
|
|
207
|
+
|
|
208
|
+
# only the visible disc is resampled; fully transparent pixels stay black
|
|
209
|
+
warped = warp_perspective(img, H, size, mask=alpha > 0)
|
|
210
|
+
return np.dstack([warped, (alpha * 255 + 0.5).astype(np.uint8)])
|
discface/_enhance.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Edge enhancement for faint print (grey on silver, print under rainbow reflections).
|
|
2
|
+
|
|
3
|
+
Two complementary views of the disc, both multi-scale band-passes (differences of Gaussians) on the luminance:
|
|
4
|
+
they keep stroke-sized detail with its polarity (dark print stays dark) and remove smooth shading and reflections,
|
|
5
|
+
each band normalised by its local energy. The second view first normalises the local contrast (local mean / local
|
|
6
|
+
standard deviation) and maps the result through a soft S-curve, which pulls very faint print further up. Their
|
|
7
|
+
errors differ, so their words vote together (evaluated on real discs reduced to 8-20 % contrast with reflections,
|
|
8
|
+
blur and JPEG: 18 of 19 correct, against 17 for either view alone)."""
|
|
9
|
+
import numpy as np
|
|
10
|
+
from PIL import Image
|
|
11
|
+
|
|
12
|
+
_BANDS = ((0.8, 3.0), (1.5, 6.0), (3.0, 12.0))
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _box(a, r, axis):
|
|
16
|
+
"""Mean over a window of 2r+1 along axis (edge-replicated), via cumulative sums."""
|
|
17
|
+
if r < 1:
|
|
18
|
+
return a
|
|
19
|
+
pad = [(0, 0)] * a.ndim
|
|
20
|
+
pad[axis] = (r + 1, r)
|
|
21
|
+
c = np.cumsum(np.pad(a, pad, mode='edge'), axis=axis, dtype=np.float64)
|
|
22
|
+
n = a.shape[axis]
|
|
23
|
+
hi = np.take(c, np.arange(2 * r + 1, 2 * r + 1 + n), axis=axis)
|
|
24
|
+
lo = np.take(c, np.arange(0, n), axis=axis)
|
|
25
|
+
return ((hi - lo) / (2 * r + 1)).astype(np.float32)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _blur(a, sigma):
|
|
29
|
+
"""Gaussian of std sigma, approximated by three box filters per axis."""
|
|
30
|
+
r = max(0, int(round((np.sqrt(4.0 * sigma * sigma + 1.0) - 1.0) / 2.0)))
|
|
31
|
+
a = a.astype(np.float32)
|
|
32
|
+
for axis in (0, 1):
|
|
33
|
+
for _ in range(3):
|
|
34
|
+
a = _box(a, r, axis)
|
|
35
|
+
return a
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _bands(lum):
|
|
39
|
+
acc = np.zeros_like(lum)
|
|
40
|
+
for s1, s2 in _BANDS:
|
|
41
|
+
d = _blur(lum, s1) - _blur(lum, s2)
|
|
42
|
+
acc += d / (np.sqrt(_blur(d * d, 4.0 * s2)) + 1.5)
|
|
43
|
+
return acc
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _grey(v, alpha):
|
|
47
|
+
g = np.clip(v, 0, 255).astype(np.uint8)
|
|
48
|
+
g[alpha == 0] = 200
|
|
49
|
+
return Image.fromarray(np.repeat(g[..., None], 3, -1))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def enhanced_views(rgba: Image.Image, size: int = 1024):
|
|
53
|
+
"""The disc at ``size`` px as grey images in which the print stands out (outside the disc: light grey):
|
|
54
|
+
[band-pass, local-contrast-normalised band-pass with S-curve]."""
|
|
55
|
+
a = np.asarray(rgba.convert('RGBA').resize((size, size), Image.LANCZOS)).astype(np.float32)
|
|
56
|
+
alpha = a[..., 3]
|
|
57
|
+
lum = 0.299 * a[..., 0] + 0.587 * a[..., 1] + 0.114 * a[..., 2]
|
|
58
|
+
plain = _grey(128.0 + 40.0 * _bands(lum), alpha)
|
|
59
|
+
m = _blur(lum, 16.0)
|
|
60
|
+
lcn = 40.0 * (lum - m) / (np.sqrt(_blur((lum - m) ** 2, 16.0)) + 2.0)
|
|
61
|
+
curved = _grey(128.0 + 127.0 * np.tanh(0.45 * _bands(lcn)), alpha)
|
|
62
|
+
return [plain, curved]
|
discface/_imaging.py
ADDED
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
"""Image primitives in numpy + Pillow (no OpenCV). Images are RGB uint8 (H x W x 3).
|
|
2
|
+
|
|
3
|
+
Grey conversion, resize, 8-bit Gaussian and Canny reproduce OpenCV's fixed-point arithmetic bit-exactly,
|
|
4
|
+
find_contours is the same Suzuki-Abe border following, fit_ellipse the same estimator as cv.fitEllipse.
|
|
5
|
+
"""
|
|
6
|
+
import os
|
|
7
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
from PIL import Image, ImageOps
|
|
11
|
+
|
|
12
|
+
# ------------------------------------------------------------------ I/O
|
|
13
|
+
def imread_rgb(path):
|
|
14
|
+
"""Any format Pillow can decode -> RGB uint8. EXIF orientation is applied (as cv2.imread does);
|
|
15
|
+
16-bit, greyscale, palette, CMYK and alpha images are converted."""
|
|
16
|
+
im = ImageOps.exif_transpose(Image.open(path))
|
|
17
|
+
if im.mode in ('I;16', 'I;16B', 'I;16L', 'I'):
|
|
18
|
+
a = np.asarray(im, np.float64)
|
|
19
|
+
im = Image.fromarray(np.clip(a / (257.0 if a.max() > 255 else 1.0) + 0.5, 0, 255).astype(np.uint8))
|
|
20
|
+
if im.mode in ('RGBA', 'LA', 'PA') or (im.mode == 'P' and 'transparency' in im.info):
|
|
21
|
+
im = im.convert('RGBA').convert('RGB') # alpha is dropped, like cv2.IMREAD_COLOR
|
|
22
|
+
return np.asarray(im.convert('RGB'))
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# ------------------------------------------------------------------ basic filters
|
|
26
|
+
def rgb2gray(rgb):
|
|
27
|
+
"""OpenCV's 8-bit fixed-point conversion: 0.299 R + 0.587 G + 0.114 B in 15-bit fixed point (bit-exact)."""
|
|
28
|
+
a = rgb.astype(np.int32)
|
|
29
|
+
return ((a[..., 0] * 9798 + a[..., 1] * 19235 + a[..., 2] * 3735 + (1 << 14)) >> 15).astype(np.uint8)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def resize_linear(img, scale):
|
|
33
|
+
"""Bilinear resize with OpenCV's pixel-centre convention (cv.resize(..., fx=scale, fy=scale))."""
|
|
34
|
+
h, w = img.shape[:2]
|
|
35
|
+
nh, nw = int(round(h * scale)), int(round(w * scale))
|
|
36
|
+
if scale == 0.5 and h % 2 == 0 and w % 2 == 0 and img.dtype == np.uint8:
|
|
37
|
+
# exact 2x reduction: OpenCV's fixed-point bilinear equals the rounded mean of each 2x2 block
|
|
38
|
+
a = img.astype(np.uint16)
|
|
39
|
+
return ((a[0::2, 0::2] + a[0::2, 1::2] + a[1::2, 0::2] + a[1::2, 1::2] + 2) >> 2).astype(np.uint8)
|
|
40
|
+
|
|
41
|
+
def axis(n_out, n_in):
|
|
42
|
+
s = (np.arange(n_out) + 0.5) / scale - 0.5
|
|
43
|
+
s = np.clip(s, 0, n_in - 1)
|
|
44
|
+
i0 = np.floor(s).astype(np.int64)
|
|
45
|
+
i1 = np.minimum(i0 + 1, n_in - 1)
|
|
46
|
+
return i0, i1, (s - i0).astype(np.float32)
|
|
47
|
+
|
|
48
|
+
y0, y1, fy = axis(nh, h)
|
|
49
|
+
x0, x1, fx = axis(nw, w)
|
|
50
|
+
a = img.astype(np.float32)
|
|
51
|
+
rows = a[y0] * (1 - fy)[:, None, None] + a[y1] * fy[:, None, None] if a.ndim == 3 else \
|
|
52
|
+
a[y0] * (1 - fy)[:, None] + a[y1] * fy[:, None]
|
|
53
|
+
out = rows[:, x0] * (1 - fx)[None, :, None] + rows[:, x1] * fx[None, :, None] if a.ndim == 3 else \
|
|
54
|
+
rows[:, x0] * (1 - fx)[None, :] + rows[:, x1] * fx[None, :]
|
|
55
|
+
return np.clip(np.floor(out + 0.5), 0, 255).astype(np.uint8) if img.dtype == np.uint8 else out
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _sep_filter(a, k, mode):
|
|
59
|
+
r = len(k) // 2
|
|
60
|
+
p = np.pad(a, r, mode=mode)
|
|
61
|
+
tmp = sum(k[i] * p[i:i + a.shape[0], :] for i in range(len(k)))
|
|
62
|
+
return sum(k[i] * tmp[:, i:i + a.shape[1]] for i in range(len(k)))
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def gaussian_blur(img, ksize, sigma):
|
|
66
|
+
"""cv.GaussianBlur(img, (ksize, ksize), sigma) with BORDER_REFLECT_101. For uint8 input OpenCV's bit-exact
|
|
67
|
+
path is reproduced: kernel in 8.8 fixed point (taps sum to exactly 256), rounding after both passes."""
|
|
68
|
+
x = np.arange(ksize) - (ksize - 1) / 2.0
|
|
69
|
+
k = np.exp(-x ** 2 / (2 * sigma ** 2))
|
|
70
|
+
k /= k.sum()
|
|
71
|
+
if img.dtype != np.uint8:
|
|
72
|
+
return _sep_filter(img.astype(np.float32), k.astype(np.float32), 'reflect')
|
|
73
|
+
kq = np.round(k * 256).astype(np.int32)
|
|
74
|
+
kq[ksize // 2] += 256 - kq.sum()
|
|
75
|
+
r = ksize // 2
|
|
76
|
+
p = np.pad(img.astype(np.int32), r, mode='reflect')
|
|
77
|
+
h = sum(kq[i] * p[:, i:i + img.shape[1]] for i in range(ksize))
|
|
78
|
+
v = sum(kq[i] * h[i:i + img.shape[0], :] for i in range(ksize))
|
|
79
|
+
return ((v + (1 << 15)) >> 16).astype(np.uint8)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def normalize_minmax_u8(a):
|
|
83
|
+
"""cv.normalize(a, None, 0, 255, cv.NORM_MINMAX).astype(np.uint8) (float result, then truncated)."""
|
|
84
|
+
a = a.astype(np.float32)
|
|
85
|
+
lo, hi = float(a.min()), float(a.max())
|
|
86
|
+
scale = np.float32(255.0 / max(hi - lo, 1e-12))
|
|
87
|
+
return np.clip(a * scale + np.float32(-lo * scale), 0, 255).astype(np.uint8)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# ------------------------------------------------------------------ connected components (8-neighbourhood)
|
|
91
|
+
def label_pixels(mask):
|
|
92
|
+
"""Connected components of the True pixels (8-connectivity).
|
|
93
|
+
Returns (ys, xs, labels) with labels in 0..n-1 for every True pixel."""
|
|
94
|
+
ys, xs = np.nonzero(mask)
|
|
95
|
+
n = len(ys)
|
|
96
|
+
if n == 0:
|
|
97
|
+
return ys, xs, np.zeros(0, np.int64)
|
|
98
|
+
h, w = mask.shape
|
|
99
|
+
idx = np.full((h + 2, w + 2), -1, np.int64)
|
|
100
|
+
idx[ys + 1, xs + 1] = np.arange(n)
|
|
101
|
+
a_list, b_list = [], []
|
|
102
|
+
for dy, dx in ((0, 1), (1, -1), (1, 0), (1, 1)):
|
|
103
|
+
nb = idx[ys + 1 + dy, xs + 1 + dx]
|
|
104
|
+
ok = nb >= 0
|
|
105
|
+
a_list.append(np.nonzero(ok)[0])
|
|
106
|
+
b_list.append(nb[ok])
|
|
107
|
+
a = np.concatenate(a_list)
|
|
108
|
+
b = np.concatenate(b_list)
|
|
109
|
+
parent = np.arange(n)
|
|
110
|
+
while True:
|
|
111
|
+
pa, pb = parent[a], parent[b]
|
|
112
|
+
lo, hi = np.minimum(pa, pb), np.maximum(pa, pb)
|
|
113
|
+
changed = lo != hi
|
|
114
|
+
if not changed.any():
|
|
115
|
+
break
|
|
116
|
+
np.minimum.at(parent, hi[changed], lo[changed])
|
|
117
|
+
while True: # pointer jumping
|
|
118
|
+
nxt = parent[parent]
|
|
119
|
+
if np.array_equal(nxt, parent):
|
|
120
|
+
break
|
|
121
|
+
parent = nxt
|
|
122
|
+
_, labels = np.unique(parent, return_inverse=True)
|
|
123
|
+
return ys, xs, labels
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ------------------------------------------------------------------ Canny (OpenCV semantics, L1 gradient, aperture 3)
|
|
127
|
+
def canny(img, low, high):
|
|
128
|
+
a = np.pad(img.astype(np.int32), 1, mode='edge') # Sobel with BORDER_REPLICATE
|
|
129
|
+
dx = (a[:-2, 2:] + 2 * a[1:-1, 2:] + a[2:, 2:]) - (a[:-2, :-2] + 2 * a[1:-1, :-2] + a[2:, :-2])
|
|
130
|
+
dy = (a[2:, :-2] + 2 * a[2:, 1:-1] + a[2:, 2:]) - (a[:-2, :-2] + 2 * a[:-2, 1:-1] + a[:-2, 2:])
|
|
131
|
+
mag = np.abs(dx) + np.abs(dy)
|
|
132
|
+
low, high = int(np.floor(low)), int(np.floor(high))
|
|
133
|
+
h, w = img.shape
|
|
134
|
+
m = np.zeros((h + 2, w + 2), np.int32) # outside the image the magnitude is 0
|
|
135
|
+
m[1:-1, 1:-1] = mag
|
|
136
|
+
ys, xs = np.nonzero(mag > low) # non-maximum suppression only where needed
|
|
137
|
+
c = mag[ys, xs]
|
|
138
|
+
gx, gy = dx[ys, xs], dy[ys, xs]
|
|
139
|
+
ax, ay = np.abs(gx), np.abs(gy) << 15
|
|
140
|
+
tg22 = ax * 13573 # tan(22.5 deg) * 2^15
|
|
141
|
+
tg67 = tg22 + (ax << 16)
|
|
142
|
+
Y, X = ys + 1, xs + 1
|
|
143
|
+
horiz = ay < tg22
|
|
144
|
+
vert = ~horiz & (ay > tg67)
|
|
145
|
+
diag = ~horiz & ~vert
|
|
146
|
+
keep = np.zeros(len(c), bool)
|
|
147
|
+
keep[horiz] = (c[horiz] > m[Y[horiz], X[horiz] - 1]) & (c[horiz] >= m[Y[horiz], X[horiz] + 1])
|
|
148
|
+
keep[vert] = (c[vert] > m[Y[vert] - 1, X[vert]]) & (c[vert] >= m[Y[vert] + 1, X[vert]])
|
|
149
|
+
s_ = np.where((gx[diag] ^ gy[diag]) < 0, -1, 1)
|
|
150
|
+
Yd, Xd = Y[diag], X[diag]
|
|
151
|
+
keep[diag] = (c[diag] > m[Yd - 1, Xd - s_]) & (c[diag] > m[Yd + 1, Xd + s_])
|
|
152
|
+
cand = np.zeros((h, w), bool)
|
|
153
|
+
cand[ys[keep], xs[keep]] = True
|
|
154
|
+
strong = np.zeros((h, w), bool)
|
|
155
|
+
sk = keep & (c > high)
|
|
156
|
+
strong[ys[sk], xs[sk]] = True
|
|
157
|
+
# hysteresis: keep the 8-connected candidate chains that contain a strong pixel
|
|
158
|
+
cy, cx, lab = label_pixels(cand)
|
|
159
|
+
good = np.zeros(lab.max() + 1 if len(lab) else 0, bool)
|
|
160
|
+
good[lab[strong[cy, cx]]] = True
|
|
161
|
+
out = np.zeros((h, w), np.uint8)
|
|
162
|
+
sel = good[lab]
|
|
163
|
+
out[cy[sel], cx[sel]] = 255
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
# ------------------------------------------------------------------ contours / ellipses
|
|
168
|
+
# 8-neighbourhood in counter-clockwise order (image coordinates, y down): E, NE, N, NW, W, SW, S, SE
|
|
169
|
+
_DY = (0, -1, -1, -1, 0, 1, 1, 1)
|
|
170
|
+
_DX = (1, 1, 0, -1, -1, -1, 0, 1)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def find_contours(binary):
|
|
174
|
+
"""Suzuki-Abe border following (RETR_LIST, CHAIN_APPROX_NONE): every outer and hole border of the
|
|
175
|
+
8-connected foreground as an array of (x, y) points, like cv.findContours."""
|
|
176
|
+
h, w = binary.shape
|
|
177
|
+
f = np.zeros((h + 2, w + 2), np.int32)
|
|
178
|
+
f[1:-1, 1:-1] = binary != 0
|
|
179
|
+
W = w + 2
|
|
180
|
+
fl = f.ravel().tolist() # python list: fast scalar access in the tracing loop
|
|
181
|
+
offs = [dy * W + dx for dy, dx in zip(_DY, _DX)]
|
|
182
|
+
contours = []
|
|
183
|
+
nbd = 1
|
|
184
|
+
starts = np.flatnonzero(f.ravel())
|
|
185
|
+
for p in starts.tolist():
|
|
186
|
+
v = fl[p]
|
|
187
|
+
if v == 0:
|
|
188
|
+
continue
|
|
189
|
+
if v == 1 and fl[p - 1] == 0:
|
|
190
|
+
nbd += 1
|
|
191
|
+
from_dir = 4 # came from the west neighbour
|
|
192
|
+
elif v >= 1 and fl[p + 1] == 0:
|
|
193
|
+
nbd += 1
|
|
194
|
+
from_dir = 0 # came from the east neighbour
|
|
195
|
+
else:
|
|
196
|
+
continue
|
|
197
|
+
# 3.1: clockwise from the "from" neighbour, find the first non-zero pixel
|
|
198
|
+
d1 = -1
|
|
199
|
+
for k in range(8):
|
|
200
|
+
d = (from_dir - k) % 8
|
|
201
|
+
if fl[p + offs[d]] != 0:
|
|
202
|
+
d1 = d
|
|
203
|
+
break
|
|
204
|
+
if d1 < 0: # isolated pixel
|
|
205
|
+
fl[p] = -nbd
|
|
206
|
+
contours.append([p])
|
|
207
|
+
continue
|
|
208
|
+
pts = []
|
|
209
|
+
p2 = p + offs[d1]
|
|
210
|
+
p3 = p
|
|
211
|
+
d2 = d1 # direction from p3 to p2
|
|
212
|
+
while True:
|
|
213
|
+
pts.append(p3)
|
|
214
|
+
# 3.3: counter-clockwise from the element after p2, find first non-zero neighbour p4
|
|
215
|
+
east_zero_examined = False
|
|
216
|
+
k = (d2 + 1) % 8
|
|
217
|
+
while True:
|
|
218
|
+
q = p3 + offs[k]
|
|
219
|
+
if fl[q] != 0:
|
|
220
|
+
break
|
|
221
|
+
if k == 0:
|
|
222
|
+
east_zero_examined = True
|
|
223
|
+
k = (k + 1) % 8
|
|
224
|
+
p4, d4 = q, k
|
|
225
|
+
# 3.4
|
|
226
|
+
if east_zero_examined:
|
|
227
|
+
fl[p3] = -nbd
|
|
228
|
+
elif fl[p3] == 1:
|
|
229
|
+
fl[p3] = nbd
|
|
230
|
+
# 3.5
|
|
231
|
+
if p4 == p and p3 == p + offs[d1]:
|
|
232
|
+
break
|
|
233
|
+
# next step: new p2 is old p3, seen from p4
|
|
234
|
+
d2 = (d4 + 4) % 8
|
|
235
|
+
p3 = p4
|
|
236
|
+
contours.append(pts)
|
|
237
|
+
out = []
|
|
238
|
+
for c in contours:
|
|
239
|
+
c = np.asarray(c, np.int64)
|
|
240
|
+
out.append(np.stack([c % W - 1, c // W - 1], 1))
|
|
241
|
+
return out
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def fit_ellipse(points):
|
|
245
|
+
"""Least-squares ellipse fit with the same estimator as cv.fitEllipse (general conic fit for the centre,
|
|
246
|
+
then a centred re-fit of the quadratic form; algorithm by D. Weiss). Axes and angle come from the
|
|
247
|
+
eigen-decomposition of the quadratic form, which is also correct where OpenCV 4.13 reports angle 0.
|
|
248
|
+
Returns ((cx, cy), (w, h), angle) with w <= h and w lying along the angle; None if not an ellipse."""
|
|
249
|
+
p = np.asarray(points, np.float64).reshape(-1, 2)
|
|
250
|
+
n = len(p)
|
|
251
|
+
if n < 5:
|
|
252
|
+
return None
|
|
253
|
+
c = p.mean(0)
|
|
254
|
+
q = p - c
|
|
255
|
+
s = np.abs(q).sum()
|
|
256
|
+
scale = 100.0 / max(s, np.finfo(np.float32).eps)
|
|
257
|
+
px, py = q[:, 0] * scale, q[:, 1] * scale
|
|
258
|
+
A = np.stack([-px * px, -py * py, -px * py, px, py], 1)
|
|
259
|
+
try:
|
|
260
|
+
gfp = np.linalg.solve(A.T @ A, A.T @ np.full(n, 10000.0))
|
|
261
|
+
r0 = np.linalg.solve(np.array([[2 * gfp[0], gfp[2]], [gfp[2], 2 * gfp[1]]]), gfp[3:5])
|
|
262
|
+
dx, dy = px - r0[0], py - r0[1]
|
|
263
|
+
B = np.stack([dx * dx, dy * dy, dx * dy], 1)
|
|
264
|
+
g = np.linalg.solve(B.T @ B, B.sum(0))
|
|
265
|
+
except np.linalg.LinAlgError:
|
|
266
|
+
return None
|
|
267
|
+
lam, vec = np.linalg.eigh(np.array([[g[0], g[2] / 2.0], [g[2] / 2.0, g[1]]]))
|
|
268
|
+
if not np.all(lam > 0):
|
|
269
|
+
return None
|
|
270
|
+
semi = 1.0 / np.sqrt(lam) / scale # lam ascending -> semi descending
|
|
271
|
+
w, h = 2 * semi[1], 2 * semi[0]
|
|
272
|
+
v = vec[:, 1] # direction of the shorter axis
|
|
273
|
+
angle = float(np.degrees(np.arctan2(v[1], v[0])) % 180.0)
|
|
274
|
+
return ((float(r0[0] / scale + c[0]), float(r0[1] / scale + c[1])), (float(w), float(h)), angle)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def ellipse_pixels(box, shape, thickness=2):
|
|
278
|
+
"""Pixels of an ellipse outline drawn with the given line thickness (pixel centres within thickness * 0.75
|
|
279
|
+
of the curve, which matches the coverage of cv.ellipse(..., thickness))."""
|
|
280
|
+
(cx, cy), (d1, d2), ang = box
|
|
281
|
+
a, b = d1 / 2.0, d2 / 2.0
|
|
282
|
+
half = 0.75 * thickness
|
|
283
|
+
t = np.linspace(0, 2 * np.pi, int(np.ceil(2 * np.pi * max(a, b) * 2)) + 16, endpoint=False)
|
|
284
|
+
th = np.radians(ang)
|
|
285
|
+
ex, ey = a * np.cos(t), b * np.sin(t)
|
|
286
|
+
px, py = cx + ex * np.cos(th) - ey * np.sin(th), cy + ex * np.sin(th) + ey * np.cos(th)
|
|
287
|
+
r = int(np.ceil(half)) + 1
|
|
288
|
+
oy, ox = np.mgrid[-r:r + 1, -r:r + 1]
|
|
289
|
+
near = (np.abs(ox) - 1) ** 2 + (np.abs(oy) - 1) ** 2 <= half * half # offsets that can lie within reach
|
|
290
|
+
ox, oy = ox[near], oy[near]
|
|
291
|
+
X = (np.floor(px)[:, None] + ox[None]).astype(np.int64)
|
|
292
|
+
Y = (np.floor(py)[:, None] + oy[None]).astype(np.int64)
|
|
293
|
+
ok = ((X - px[:, None]) ** 2 + (Y - py[:, None]) ** 2 <= half * half) & (X >= 0) & (X < shape[1]) & (Y >= 0) & (Y < shape[0])
|
|
294
|
+
lin = np.unique(Y[ok] * shape[1] + X[ok])
|
|
295
|
+
return lin // shape[1], lin % shape[1]
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
# ------------------------------------------------------------------ perspective warp (Lanczos4)
|
|
299
|
+
def _lanczos4_weights(f):
|
|
300
|
+
"""(N,) fractional offsets -> (N, 8) Lanczos4 weights for taps -3..4 (normalised like OpenCV)."""
|
|
301
|
+
k = np.arange(-3, 5, dtype=np.float32)
|
|
302
|
+
x = (f[:, None] - k[None, :]).astype(np.float32)
|
|
303
|
+
px = np.float32(np.pi) * x
|
|
304
|
+
with np.errstate(divide='ignore', invalid='ignore'):
|
|
305
|
+
w = np.where(np.abs(x) < 1e-6, np.float32(1.0), 4.0 * np.sin(px) * np.sin(px / 4.0) / (px * px))
|
|
306
|
+
return (w / w.sum(1, keepdims=True)).astype(np.float32)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def warp_perspective(img, H, size, mask=None, rows_per_chunk=32, workers=None):
|
|
310
|
+
"""dst(x, y) = src(H^-1 (x, y)), Lanczos4 interpolation, constant 0 outside the source.
|
|
311
|
+
mask (bool size x size, optional): only these destination pixels are computed, the rest stay 0."""
|
|
312
|
+
h, w = img.shape[:2]
|
|
313
|
+
Hi = np.linalg.inv(H)
|
|
314
|
+
# source bounding box actually needed (+ kernel support)
|
|
315
|
+
ys, xs = (np.nonzero(mask) if mask is not None else np.mgrid[0:size:16, 0:size:16].reshape(2, -1))
|
|
316
|
+
if mask is not None and len(ys) > 200000:
|
|
317
|
+
sel = np.linspace(0, len(ys) - 1, 200000).astype(np.int64)
|
|
318
|
+
ys, xs = ys[sel], xs[sel]
|
|
319
|
+
den = Hi[2, 0] * xs + Hi[2, 1] * ys + Hi[2, 2]
|
|
320
|
+
sx = (Hi[0, 0] * xs + Hi[0, 1] * ys + Hi[0, 2]) / den
|
|
321
|
+
sy = (Hi[1, 0] * xs + Hi[1, 1] * ys + Hi[1, 2]) / den
|
|
322
|
+
x0 = int(max(0, np.floor(sx.min()) - 8)); x1 = int(min(w, np.ceil(sx.max()) + 9))
|
|
323
|
+
y0 = int(max(0, np.floor(sy.min()) - 8)); y1 = int(min(h, np.ceil(sy.max()) + 9))
|
|
324
|
+
pad = 4
|
|
325
|
+
cw, ch = x1 - x0 + 2 * pad, y1 - y0 + 2 * pad
|
|
326
|
+
packed = np.zeros((ch, cw, 4), np.uint8)
|
|
327
|
+
packed[pad:pad + y1 - y0, pad:pad + x1 - x0, :3] = img[y0:y1, x0:x1]
|
|
328
|
+
flat = packed.view(np.uint32).reshape(-1)
|
|
329
|
+
out = np.zeros((size, size, 3), np.uint8)
|
|
330
|
+
|
|
331
|
+
def rows(r0):
|
|
332
|
+
r1 = min(size, r0 + rows_per_chunk)
|
|
333
|
+
yy, xx = np.mgrid[r0:r1, 0:size]
|
|
334
|
+
m = mask[r0:r1] if mask is not None else np.ones(yy.shape, bool)
|
|
335
|
+
X, Y = xx[m].astype(np.float64), yy[m].astype(np.float64)
|
|
336
|
+
if len(X) == 0:
|
|
337
|
+
return
|
|
338
|
+
den = Hi[2, 0] * X + Hi[2, 1] * Y + Hi[2, 2]
|
|
339
|
+
sx = (Hi[0, 0] * X + Hi[0, 1] * Y + Hi[0, 2]) / den - x0 + pad
|
|
340
|
+
sy = (Hi[1, 0] * X + Hi[1, 1] * Y + Hi[1, 2]) / den - y0 + pad
|
|
341
|
+
ix, iy = np.floor(sx).astype(np.int64), np.floor(sy).astype(np.int64)
|
|
342
|
+
wx = _lanczos4_weights((sx - ix).astype(np.float32))
|
|
343
|
+
wy = _lanczos4_weights((sy - iy).astype(np.float32))
|
|
344
|
+
ix = np.clip(ix, 3, cw - 5)
|
|
345
|
+
iy = np.clip(iy, 3, ch - 5)
|
|
346
|
+
base = iy * cw + ix
|
|
347
|
+
acc = np.zeros((len(X), 3), np.float32)
|
|
348
|
+
for j in range(8):
|
|
349
|
+
rowacc = np.zeros((len(X), 3), np.float32)
|
|
350
|
+
rb = base + (j - 3) * cw
|
|
351
|
+
for i in range(8):
|
|
352
|
+
v = flat[rb + (i - 3)].view(np.uint8).reshape(-1, 4)[:, :3]
|
|
353
|
+
rowacc += wx[:, i:i + 1] * v
|
|
354
|
+
acc += wy[:, j:j + 1] * rowacc
|
|
355
|
+
blk = np.zeros(yy.shape + (3,), np.uint8)
|
|
356
|
+
blk[m] = np.clip(acc + 0.5, 0, 255).astype(np.uint8)
|
|
357
|
+
out[r0:r1] = blk
|
|
358
|
+
|
|
359
|
+
# numpy releases the GIL in gathers and arithmetic, so row blocks run in parallel in plain threads
|
|
360
|
+
with ThreadPoolExecutor(workers or min(4, os.cpu_count() or 1)) as ex:
|
|
361
|
+
list(ex.map(rows, range(0, size, rows_per_chunk)))
|
|
362
|
+
return out
|
discface/_orient.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Turning a rectified disc upright from the reading direction of its text (doritex)."""
|
|
2
|
+
import math
|
|
3
|
+
from typing import Optional, Sequence
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def is_rim_text(word, center, radius, zone=0.8, max_cos=0.35) -> bool:
|
|
9
|
+
"""Text running round the disc near its rim: in the outer ring (beyond ``zone`` of the radius) and tangential,
|
|
10
|
+
i.e. its reading direction nearly perpendicular to the radius. Such text occurs at every angle, so it says
|
|
11
|
+
nothing about which way the design is up."""
|
|
12
|
+
r = np.asarray(word.center, np.float64) - center
|
|
13
|
+
dist = float(np.linalg.norm(r))
|
|
14
|
+
if dist <= zone * radius:
|
|
15
|
+
return False
|
|
16
|
+
return abs(float(np.dot(word.direction, r / max(dist, 1e-9)))) < max_cos
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def disc_vote(words: Sequence, center=None, radius=None):
|
|
20
|
+
"""(angle, support): rotation of the disc's printing (counter-clockwise degrees, like ``doritex.Word.angle``) and
|
|
21
|
+
the summed weight of the words that agree with it; (None, 0.0) without words.
|
|
22
|
+
|
|
23
|
+
Text running round the rim is left out (given ``center`` and ``radius``) unless there is nothing else, because it
|
|
24
|
+
points in every direction. Of the rest, the most common reading direction wins, not the mean: each word weighs
|
|
25
|
+
by its size in px (titles define "up" more than small print), detection score and direction confidence; the
|
|
26
|
+
peak of the angle distribution is refined by the weighted mean of the words within 20 degrees of it."""
|
|
27
|
+
if not words:
|
|
28
|
+
return None, 0.0
|
|
29
|
+
if center is not None and radius is not None:
|
|
30
|
+
inner = [wd for wd in words if not is_rim_text(wd, np.asarray(center, np.float64), radius)]
|
|
31
|
+
words = inner or words
|
|
32
|
+
w = np.array([wd.score * max(2.0 * wd.direction_confidence - 1.0, 1e-3) * wd.height for wd in words])
|
|
33
|
+
a = np.radians([wd.angle for wd in words])
|
|
34
|
+
grid = np.radians(np.arange(0.0, 360.0, 1.0))
|
|
35
|
+
density = (w[None] * np.exp(8.0 * (np.cos(grid[:, None] - a[None]) - 1.0))).sum(1) # von Mises kernel
|
|
36
|
+
peak = grid[int(np.argmax(density))]
|
|
37
|
+
near = np.cos(a - peak) > math.cos(math.radians(20.0))
|
|
38
|
+
angle = math.degrees(math.atan2(float((w[near] * np.sin(a[near])).sum()), float((w[near] * np.cos(a[near])).sum())))
|
|
39
|
+
return float(angle), float(w[near].sum())
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def disc_angle(words: Sequence, center=None, radius=None) -> Optional[float]:
|
|
43
|
+
"""The angle of ``disc_vote``, or None without words."""
|
|
44
|
+
return disc_vote(words, center, radius)[0]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def rotate_points(points: np.ndarray, angle: float, center) -> np.ndarray:
|
|
48
|
+
"""Points (x right, y down) rotated counter-clockwise on screen by ``angle`` degrees about ``center``,
|
|
49
|
+
the same motion as ``PIL.Image.rotate(angle)``."""
|
|
50
|
+
t = math.radians(angle)
|
|
51
|
+
c, s = math.cos(t), math.sin(t)
|
|
52
|
+
d = np.asarray(points, np.float64) - center
|
|
53
|
+
return np.stack([center[0] + c * d[..., 0] + s * d[..., 1], center[1] - s * d[..., 0] + c * d[..., 1]], -1)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _polygon_area(p):
|
|
57
|
+
x, y = p[:, 0], p[:, 1]
|
|
58
|
+
return 0.5 * float(np.dot(x, np.roll(y, -1)) - np.dot(y, np.roll(x, -1)))
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _clip(subject, a, b):
|
|
62
|
+
out = []
|
|
63
|
+
side = lambda p: (b[0] - a[0]) * (p[1] - a[1]) - (b[1] - a[1]) * (p[0] - a[0])
|
|
64
|
+
for i in range(len(subject)):
|
|
65
|
+
cur, nxt = subject[i], subject[(i + 1) % len(subject)]
|
|
66
|
+
sc, sn = side(cur), side(nxt)
|
|
67
|
+
if sc >= 0:
|
|
68
|
+
out.append(cur)
|
|
69
|
+
if (sc >= 0) != (sn >= 0):
|
|
70
|
+
out.append(cur + sc / (sc - sn) * (nxt - cur))
|
|
71
|
+
return out
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _iou(a, b) -> float:
|
|
75
|
+
a = np.asarray(a, np.float64)
|
|
76
|
+
b = np.asarray(b, np.float64)
|
|
77
|
+
a = a if _polygon_area(a) >= 0 else a[::-1]
|
|
78
|
+
b = b if _polygon_area(b) >= 0 else b[::-1]
|
|
79
|
+
inter = list(a)
|
|
80
|
+
for i in range(len(b)):
|
|
81
|
+
if not inter:
|
|
82
|
+
break
|
|
83
|
+
inter = _clip(inter, b[i], b[(i + 1) % len(b)])
|
|
84
|
+
ia = abs(_polygon_area(np.array(inter))) if len(inter) >= 3 else 0.0
|
|
85
|
+
union = abs(_polygon_area(a)) + abs(_polygon_area(b)) - ia
|
|
86
|
+
return ia / union if union > 1e-9 else 0.0
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def dedupe(words, iou=0.4):
|
|
90
|
+
"""Merge two readings of the same disc: of overlapping boxes (IoU >= iou) keep the more confident one."""
|
|
91
|
+
kept = []
|
|
92
|
+
for w in sorted(words, key=lambda w: -w.score):
|
|
93
|
+
if all(_iou(w.quad, k.quad) < iou for k in kept):
|
|
94
|
+
kept.append(w)
|
|
95
|
+
return kept
|
discface/api.py
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""Public API: from a photo of a CD / DVD to the flat, upright disc.
|
|
2
|
+
|
|
3
|
+
The pipeline has three steps, each callable on its own:
|
|
4
|
+
locate(photo) -> DiscGeometry (rim, hole, true centre in the photo)
|
|
5
|
+
rectify(photo, geometry) -> the flat disc (square RGBA, transparent outside the rim and in the hole)
|
|
6
|
+
orient(flat_disc) -> Orientation (how far to turn it so that its printing reads upright)
|
|
7
|
+
``extract`` runs all three.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from typing import List, Optional, Tuple, Union
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
from PIL import Image
|
|
17
|
+
|
|
18
|
+
from ._imaging import imread_rgb
|
|
19
|
+
from ._disc import filter_and_detect_edges, detect_candidate_ellipses, select_outer_and_inner_ellipses, \
|
|
20
|
+
rectify_and_project_to_circle
|
|
21
|
+
from ._orient import disc_vote, rotate_points, dedupe
|
|
22
|
+
from ._enhance import enhanced_views
|
|
23
|
+
|
|
24
|
+
ImageLike = Union[str, os.PathLike, Image.Image, np.ndarray]
|
|
25
|
+
Ellipse = Tuple[Tuple[float, float], Tuple[float, float], float] # ((cx, cy), (w, h), angle in degrees)
|
|
26
|
+
ENHANCE_MODES = ('auto', 'always', 'off')
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class DiscNotFound(ValueError):
|
|
30
|
+
"""No disc rim / centre hole could be found in the image."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class DiscGeometry:
|
|
35
|
+
"""Where the disc lies in the photo (pixels): rim and centre hole as ellipses ``((cx, cy), (w, h), angle)``
|
|
36
|
+
and the true disc centre, which under perspective is not the centre of the rim ellipse."""
|
|
37
|
+
|
|
38
|
+
rim: Ellipse
|
|
39
|
+
hole: Ellipse
|
|
40
|
+
center: Tuple[float, float]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class Orientation:
|
|
45
|
+
"""How far to turn a flat disc so that its printing reads upright.
|
|
46
|
+
|
|
47
|
+
``rotation``: counter-clockwise degrees (the convention of ``PIL.Image.rotate``); 0 if the text was too weak.
|
|
48
|
+
``support``: weight of the text that agreed on the direction (sum of word size x confidence).
|
|
49
|
+
``turned``: whether the support reached the threshold. ``enhanced``: whether faint print was also read from
|
|
50
|
+
edge-enhanced views. ``words``: the words read (``doritex.Word``), in coordinates of the image passed in."""
|
|
51
|
+
|
|
52
|
+
rotation: float
|
|
53
|
+
support: float
|
|
54
|
+
turned: bool
|
|
55
|
+
enhanced: bool
|
|
56
|
+
words: List = field(default_factory=list)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class Disc:
|
|
61
|
+
"""A disc cut out of a photo.
|
|
62
|
+
|
|
63
|
+
``image``: the disc seen from straight above, square RGBA, transparent outside the rim and in the centre hole.
|
|
64
|
+
``rotation``: counter-clockwise degrees by which it was turned to make its printing upright (0 if not turned).
|
|
65
|
+
``orientation_support`` / ``enhanced``: see ``Orientation``. ``words``: words read on the disc, in ``image``
|
|
66
|
+
coordinates. ``rim``, ``hole``, ``center``: where the disc lies in the photo."""
|
|
67
|
+
|
|
68
|
+
image: Image.Image
|
|
69
|
+
rotation: float
|
|
70
|
+
words: List = field(default_factory=list)
|
|
71
|
+
rim: Optional[Ellipse] = None
|
|
72
|
+
hole: Optional[Ellipse] = None
|
|
73
|
+
center: Optional[Tuple[float, float]] = None
|
|
74
|
+
orientation_support: float = 0.0
|
|
75
|
+
enhanced: bool = False
|
|
76
|
+
|
|
77
|
+
def to_image(self, background=None) -> Image.Image:
|
|
78
|
+
"""``image``; with ``background`` (a colour such as ``"white"`` or ``(32, 32, 32)``) the transparent parts
|
|
79
|
+
are filled and an RGB image is returned."""
|
|
80
|
+
if background is None:
|
|
81
|
+
return self.image
|
|
82
|
+
bg = Image.new('RGBA', self.image.size, background)
|
|
83
|
+
return Image.alpha_composite(bg, self.image.convert('RGBA')).convert('RGB')
|
|
84
|
+
|
|
85
|
+
def save(self, path, background=None, **kwargs):
|
|
86
|
+
"""Save the disc. PNG / WebP keep the transparency; formats without alpha (JPEG) get a white background
|
|
87
|
+
unless ``background`` is given."""
|
|
88
|
+
if background is None and os.path.splitext(str(path))[1].lower() in ('.jpg', '.jpeg', '.bmp'):
|
|
89
|
+
background = 'white'
|
|
90
|
+
self.to_image(background).save(path, **kwargs)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# ----------------------------------------------------------------------------------------------- steps
|
|
94
|
+
|
|
95
|
+
def _photo_rgb(image: ImageLike) -> np.ndarray:
|
|
96
|
+
if isinstance(image, (str, os.PathLike)):
|
|
97
|
+
return imread_rgb(image) # applies the EXIF orientation
|
|
98
|
+
if isinstance(image, Image.Image):
|
|
99
|
+
return np.asarray(image.convert('RGB'))
|
|
100
|
+
a = np.asarray(image)
|
|
101
|
+
if a.dtype != np.uint8:
|
|
102
|
+
a = np.clip(a * 255.0 if a.max() <= 1.0 else a, 0, 255).astype(np.uint8)
|
|
103
|
+
if a.ndim == 2:
|
|
104
|
+
a = np.repeat(a[..., None], 3, -1)
|
|
105
|
+
return np.ascontiguousarray(a[..., :3])
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _scale(box, s):
|
|
109
|
+
(cx, cy), (w, h), ang = box
|
|
110
|
+
return ((cx * s, cy * s), (w * s, h * s), ang)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def locate(image: ImageLike) -> DiscGeometry:
|
|
114
|
+
"""Find the disc in a photo. Raises ``DiscNotFound`` if no disc is visible."""
|
|
115
|
+
return _locate(_photo_rgb(image))
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _locate(rgb: np.ndarray) -> DiscGeometry:
|
|
119
|
+
edges, _ = filter_and_detect_edges(rgb) # at half resolution
|
|
120
|
+
candidates = detect_candidate_ellipses(edges)
|
|
121
|
+
try:
|
|
122
|
+
outer, inner, center = select_outer_and_inner_ellipses(candidates, edges)
|
|
123
|
+
except (ValueError, IndexError) as exc:
|
|
124
|
+
raise DiscNotFound('no disc found in the image') from exc
|
|
125
|
+
return DiscGeometry(_scale(outer, 2.0), _scale(inner, 2.0), (center[0] * 2.0, center[1] * 2.0))
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def rectify(image: ImageLike, geometry: Optional[DiscGeometry] = None, size: int = 2048, margin: int = 32) -> Image.Image:
|
|
129
|
+
"""The disc as seen from straight above: ``size`` x ``size`` RGBA, the rim at ``margin`` px from the border,
|
|
130
|
+
transparent outside the rim and in the hole. Locates the disc first if no ``geometry`` is given."""
|
|
131
|
+
rgb = _photo_rgb(image)
|
|
132
|
+
g = geometry or _locate(rgb)
|
|
133
|
+
return Image.fromarray(rectify_and_project_to_circle(rgb, g.rim, g.hole, g.center, scale=1.0, size=size,
|
|
134
|
+
margin=margin), 'RGBA')
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def orient(disc: Image.Image, enhance: str = 'auto', min_support: float = 40.0, margin: int = 32,
|
|
138
|
+
detector=None, device: str = 'auto') -> Orientation:
|
|
139
|
+
"""How far to turn a flat disc (as from ``rectify``) so that its printing reads upright.
|
|
140
|
+
|
|
141
|
+
Every word votes for its reading direction, weighted by size and confidence; text running round the rim is
|
|
142
|
+
left out; the most common direction wins. The disc counts as decided when the agreeing words weigh at least
|
|
143
|
+
``min_support`` (for a 2048 px disc, scaled with its size).
|
|
144
|
+
``enhance``: 'auto' reads faint print a second time from edge-enhanced views, only when the plain reading is
|
|
145
|
+
too weak (discs that read well are not affected); 'always' always adds those views; 'off' never.
|
|
146
|
+
``detector``: a ``doritex.Detector`` to reuse; otherwise the cached one for ``device`` ('auto': an OpenCL GPU
|
|
147
|
+
if present, else the CPU; 'cpu': always the CPU)."""
|
|
148
|
+
if enhance not in ENHANCE_MODES:
|
|
149
|
+
raise ValueError(f'enhance must be one of {ENHANCE_MODES}, not {enhance!r}')
|
|
150
|
+
det = detector or load_detector(device)
|
|
151
|
+
size = disc.size[0]
|
|
152
|
+
work = min(size, 1024)
|
|
153
|
+
words = det.detect(disc, input_size=work)
|
|
154
|
+
vote = lambda ws: disc_vote(ws, center=(size / 2.0, size / 2.0), radius=size / 2.0 - margin)
|
|
155
|
+
angle, support = vote(words)
|
|
156
|
+
threshold = min_support * size / 2048.0
|
|
157
|
+
enhanced = False
|
|
158
|
+
if enhance == 'always' or (enhance == 'auto' and (angle is None or support < threshold)):
|
|
159
|
+
k = size / float(work)
|
|
160
|
+
for view in enhanced_views(disc, work):
|
|
161
|
+
words += [type(w)(quad=w.quad * k, score=w.score, direction_confidence=w.direction_confidence)
|
|
162
|
+
for w in det.detect(view, input_size=work)]
|
|
163
|
+
words, enhanced = dedupe(words), True
|
|
164
|
+
angle, support = vote(words)
|
|
165
|
+
turned = angle is not None and support >= threshold
|
|
166
|
+
return Orientation(rotation=-angle if turned else 0.0, support=float(support), turned=turned, enhanced=enhanced,
|
|
167
|
+
words=words)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def extract(image: ImageLike, size: int = 2048, margin: int = 32, autorotate: bool = True, enhance: str = 'auto',
|
|
171
|
+
min_support: float = 40.0, detector=None, device: str = 'auto') -> Disc:
|
|
172
|
+
"""Find the disc in a photo, rectify it and (``autorotate=True``) turn it so that its printing reads upright.
|
|
173
|
+
See ``rectify`` (``size``, ``margin``) and ``orient`` (``enhance``, ``min_support``, ``detector``, ``device``).
|
|
174
|
+
Raises ``DiscNotFound`` if no disc is visible."""
|
|
175
|
+
rgb = _photo_rgb(image)
|
|
176
|
+
geometry = _locate(rgb)
|
|
177
|
+
flat = rectify(rgb, geometry, size=size, margin=margin)
|
|
178
|
+
if not autorotate:
|
|
179
|
+
return Disc(flat, 0.0, [], geometry.rim, geometry.hole, geometry.center)
|
|
180
|
+
o = orient(flat, enhance=enhance, min_support=min_support, margin=margin, detector=detector, device=device)
|
|
181
|
+
if not o.turned:
|
|
182
|
+
return Disc(flat, 0.0, o.words, geometry.rim, geometry.hole, geometry.center, o.support, o.enhanced)
|
|
183
|
+
c = np.array([size / 2.0, size / 2.0])
|
|
184
|
+
words = [type(w)(quad=rotate_points(w.quad, o.rotation, c), score=w.score, direction_confidence=w.direction_confidence)
|
|
185
|
+
for w in o.words]
|
|
186
|
+
return Disc(flat.rotate(o.rotation, resample=Image.BICUBIC), o.rotation, words, geometry.rim, geometry.hole,
|
|
187
|
+
geometry.center, o.support, o.enhanced)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
_DETECTORS = {}
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def load_detector(device: str = 'auto'):
|
|
194
|
+
"""The doritex detector used for reading the text (cached per device): 'auto' (OpenCL GPU if present) or 'cpu'."""
|
|
195
|
+
if device not in ('auto', 'cpu'):
|
|
196
|
+
raise ValueError(f"device must be 'auto' or 'cpu', not {device!r}")
|
|
197
|
+
if device not in _DETECTORS:
|
|
198
|
+
import doritex
|
|
199
|
+
_DETECTORS[device] = doritex.load(device=device)
|
|
200
|
+
return _DETECTORS[device]
|
|
201
|
+
|
discface/cli.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Command line: discface PHOTO_OR_DIR... [-o OUT_DIR] [options]"""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from typing import List
|
|
9
|
+
|
|
10
|
+
from .api import ENHANCE_MODES, DiscNotFound, extract, load_detector
|
|
11
|
+
|
|
12
|
+
IMAGE_EXTENSIONS = {'.jpg', '.jpeg', '.png', '.tif', '.tiff', '.bmp', '.webp', '.heic'}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _inputs(paths: List[str]) -> List[str]:
|
|
16
|
+
files = []
|
|
17
|
+
for p in paths:
|
|
18
|
+
if os.path.isdir(p):
|
|
19
|
+
files += sorted(os.path.join(p, f) for f in os.listdir(p) if os.path.splitext(f)[1].lower() in IMAGE_EXTENSIONS)
|
|
20
|
+
else:
|
|
21
|
+
files.append(p)
|
|
22
|
+
return files
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _r2(v):
|
|
26
|
+
return [_r2(x) for x in v] if isinstance(v, (tuple, list)) else round(float(v), 2)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main(argv=None) -> int:
|
|
30
|
+
ap = argparse.ArgumentParser(prog='discface', description='Cut CDs / DVDs out of photos: flat, upright, '
|
|
31
|
+
'transparent outside the rim and in the centre hole.')
|
|
32
|
+
ap.add_argument('inputs', nargs='+', help='photos or folders of photos')
|
|
33
|
+
ap.add_argument('-o', '--output', default='.', help='output folder (default: current folder)')
|
|
34
|
+
ap.add_argument('--size', type=int, default=2048, help='side of the output image in px (default: 2048)')
|
|
35
|
+
ap.add_argument('--margin', type=int, default=32, help='px between the rim and the image border (default: 32)')
|
|
36
|
+
ap.add_argument('--no-autorotate', action='store_true', help='only rectify, do not turn the printing upright')
|
|
37
|
+
ap.add_argument('--enhance', choices=ENHANCE_MODES, default='auto',
|
|
38
|
+
help="faint print: 'auto' reads it from edge-enhanced views only when the plain reading is too weak "
|
|
39
|
+
"(default), 'always' always adds them, 'off' never")
|
|
40
|
+
ap.add_argument('--min-support', type=float, default=40.0,
|
|
41
|
+
help='how much agreeing text is needed to turn a disc (default: 40, for 2048 px)')
|
|
42
|
+
ap.add_argument('--background', default=None,
|
|
43
|
+
help="fill the transparent parts with a colour, e.g. white or '#202020' (default: transparent)")
|
|
44
|
+
ap.add_argument('--format', choices=('png', 'jpg', 'webp'), default='png',
|
|
45
|
+
help='output format (default: png; jpg gets a white background unless --background is given)')
|
|
46
|
+
ap.add_argument('--suffix', default='', help="appended to the output file names, e.g. '_disc'")
|
|
47
|
+
ap.add_argument('--device', choices=('auto', 'cpu'), default='auto', help="text reading on an OpenCL GPU ('auto') or the CPU")
|
|
48
|
+
ap.add_argument('--draw', action='store_true', help='also save <name>_words.png with the words read on the disc')
|
|
49
|
+
ap.add_argument('--json', action='store_true', help='print one JSON object per photo instead of a summary line')
|
|
50
|
+
ap.add_argument('-q', '--quiet', action='store_true', help='print nothing except errors')
|
|
51
|
+
args = ap.parse_args(argv)
|
|
52
|
+
|
|
53
|
+
files = _inputs(args.inputs)
|
|
54
|
+
if not files:
|
|
55
|
+
ap.error('no input images')
|
|
56
|
+
os.makedirs(args.output, exist_ok=True)
|
|
57
|
+
detector = None if args.no_autorotate else load_detector(args.device)
|
|
58
|
+
|
|
59
|
+
failed = 0
|
|
60
|
+
for path in files:
|
|
61
|
+
name = os.path.splitext(os.path.basename(path))[0] + args.suffix
|
|
62
|
+
out = os.path.join(args.output, f'{name}.{args.format}')
|
|
63
|
+
try:
|
|
64
|
+
disc = extract(path, size=args.size, margin=args.margin, autorotate=not args.no_autorotate,
|
|
65
|
+
enhance=args.enhance, min_support=args.min_support, detector=detector)
|
|
66
|
+
except (DiscNotFound, OSError) as exc:
|
|
67
|
+
failed += 1
|
|
68
|
+
err = 'no disc found' if isinstance(exc, DiscNotFound) else str(exc)
|
|
69
|
+
if args.json:
|
|
70
|
+
print(json.dumps({'input': path, 'error': err}))
|
|
71
|
+
else:
|
|
72
|
+
print(f'{path}: {err}', file=sys.stderr)
|
|
73
|
+
continue
|
|
74
|
+
disc.save(out, background=args.background)
|
|
75
|
+
if args.draw:
|
|
76
|
+
import doritex
|
|
77
|
+
doritex.draw(disc.to_image(args.background or (200, 200, 200)), disc.words).save(
|
|
78
|
+
os.path.join(args.output, f'{name}_words.png'))
|
|
79
|
+
if args.json:
|
|
80
|
+
print(json.dumps({'input': path, 'output': out, 'rotation': round(disc.rotation, 2),
|
|
81
|
+
'oriented': disc.rotation != 0.0, 'orientation_support': round(disc.orientation_support, 1),
|
|
82
|
+
'enhanced': disc.enhanced, 'words': len(disc.words), 'rim': _r2(disc.rim),
|
|
83
|
+
'hole': _r2(disc.hole), 'center': _r2(disc.center)}))
|
|
84
|
+
elif not args.quiet:
|
|
85
|
+
if args.no_autorotate:
|
|
86
|
+
note = 'rectified'
|
|
87
|
+
elif disc.rotation != 0.0:
|
|
88
|
+
note = f'turned {disc.rotation:.1f}° {len(disc.words)} words' + (' (faint print, edge-enhanced)' if disc.enhanced else '')
|
|
89
|
+
else:
|
|
90
|
+
note = 'not turned: too little readable text, kept as photographed'
|
|
91
|
+
print(f'{path} -> {out} {note}')
|
|
92
|
+
return 1 if failed else 0
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
if __name__ == '__main__':
|
|
96
|
+
sys.exit(main())
|
discface/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: discface
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: From a photo of a CD or DVD to the flat, upright disc image
|
|
5
|
+
Author: discface contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/GodOfLores/discface
|
|
8
|
+
Project-URL: Source, https://github.com/GodOfLores/discface
|
|
9
|
+
Project-URL: Issues, https://github.com/GodOfLores/discface/issues
|
|
10
|
+
Keywords: CD,DVD,disc,rectification,perspective,orientation,doritex
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: doritex>=0.1.0
|
|
18
|
+
Requires-Dist: numpy>=1.23
|
|
19
|
+
Requires-Dist: Pillow>=9.0
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest; extra == "test"
|
|
22
|
+
|
|
23
|
+
# discface
|
|
24
|
+
|
|
25
|
+
Transforms photos of CDs and DVDs into flat, top-down disc images. It removes perspective distortion, cuts out the disc with a transparent outer rim and centre hole, and rotates it so printed text reads upright. The disc boundary is detected geometrically, while text is read with [doritex](https://pypi.org/project/doritex/).
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install discface
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Example
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
import discface
|
|
35
|
+
|
|
36
|
+
disc = discface.extract("blank_dvd.jpg") # photo of a blank DVD+R on a light table
|
|
37
|
+
print(disc.image.size, disc.image.mode)
|
|
38
|
+
print(f"turned by {disc.rotation:.1f}°, faint print enhanced: {disc.enhanced}")
|
|
39
|
+
disc.save("blank_dvd_disc.png")
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Output:
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
(2048, 2048) RGBA
|
|
46
|
+
turned by -17.7°, faint print enhanced: True
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
<table>
|
|
50
|
+
<tr>
|
|
51
|
+
<td align="center"><img src="https://raw.githubusercontent.com/GodOfLores/discface/main/blank_dvd.jpg" alt="photo of a blank DVD lying on a table" width="420"><br><code>blank_dvd.jpg</code></td>
|
|
52
|
+
<td align="center"><img src="https://raw.githubusercontent.com/GodOfLores/discface/main/blank_dvd.png" alt="the same disc, flat, cut out and upright" width="300"><br><code>blank_dvd_disc.png</code></td>
|
|
53
|
+
</tr>
|
|
54
|
+
</table>
|
|
55
|
+
|
|
56
|
+
Perspective distortion is removed, leaving the disc neatly cut out with a transparent outer rim and centre hole. The image was rotated by −17.7° so that "Verbatim" reads horizontally. Because grey print on silver is often faint, it was also detected via edge-enhanced views (`enhanced: True`).
|
|
57
|
+
|
|
58
|
+
## Taking the photo
|
|
59
|
+
|
|
60
|
+
discface is designed for discs placed on a **plain, uniform background** and photographed from roughly above. A sheet of white paper or a light grey tabletop works best, which is how all reference test photos were taken.
|
|
61
|
+
|
|
62
|
+
- Keep the entire disc in frame with some surrounding margin, ensuring the centre hole is visible.
|
|
63
|
+
- Shoot from roughly above. Moderate tilt is corrected automatically, and perspective is rectified precisely.
|
|
64
|
+
- Ensure even lighting. Moderate reflections are tolerated, and faint print benefits from enabling `enhance`.
|
|
65
|
+
- Capture a single disc per photo, keeping other circular objects out of frame.
|
|
66
|
+
|
|
67
|
+
In a benchmark where backgrounds around real discs were replaced while keeping the discs intact, rim detection rates were as follows:
|
|
68
|
+
|
|
69
|
+
| background | rim found |
|
|
70
|
+
|---|---|
|
|
71
|
+
| light, uniform (all real photos) | 7/7 |
|
|
72
|
+
| dark, uniform | 6/7 |
|
|
73
|
+
| mid grey, uniform | 6/7 |
|
|
74
|
+
| busy (photos, printed paper, noise) | 5/35 |
|
|
75
|
+
|
|
76
|
+
On dark backgrounds, centre hole detection is less reliable (4/7). Patterned or cluttered backgrounds are not currently supported.
|
|
77
|
+
|
|
78
|
+
## Command line
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
discface photos/ -o discs/
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
photos/IMG_4097.JPG -> discs/IMG_4097.png turned 39.2° 59 words
|
|
86
|
+
photos/IMG_4100.JPG -> discs/IMG_4100.png turned -1.7° 80 words
|
|
87
|
+
blank_dvd.jpg -> discs/blank_dvd.png turned -17.7° 71 words (faint print, edge-enhanced)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Inputs can be individual photos, directories, or a mix of both. Each disc is saved as `<photo name><suffix>.<format>` in the output folder.
|
|
91
|
+
|
|
92
|
+
| option | meaning |
|
|
93
|
+
|---|---|
|
|
94
|
+
| `-o DIR` | output folder (default: current folder) |
|
|
95
|
+
| `--size N` | output image dimension in px (default 2048) |
|
|
96
|
+
| `--margin N` | margin between the rim and the image border in px (default 32) |
|
|
97
|
+
| `--no-autorotate` | rectify perspective only, without rotating text upright |
|
|
98
|
+
| `--enhance auto\|always\|off` | edge enhancement for faint print, detailed below (default `auto`) |
|
|
99
|
+
| `--min-support X` | minimum text agreement score required to rotate a disc (default 40) |
|
|
100
|
+
| `--background COLOR` | fill transparent areas with a solid colour, e.g. `white` or `'#202020'` |
|
|
101
|
+
| `--format png\|jpg\|webp` | output format; `jpg` defaults to a white background unless `--background` is specified |
|
|
102
|
+
| `--suffix TEXT` | string appended to output filenames |
|
|
103
|
+
| `--device auto\|cpu` | device used for text recognition (OpenCL GPU or CPU) |
|
|
104
|
+
| `--draw` | also save `<name>_words.png` visualizing detected words |
|
|
105
|
+
| `--json` | print one JSON line per photo containing detection metadata |
|
|
106
|
+
| `-q` | quiet mode, suppressing all output except errors |
|
|
107
|
+
|
|
108
|
+
Photos without a detectable disc are reported and skipped, returning exit code 1. You can also run the tool via `python -m discface`.
|
|
109
|
+
|
|
110
|
+
## API
|
|
111
|
+
|
|
112
|
+
**`discface.extract(image, size=2048, margin=32, autorotate=True, enhance="auto", min_support=40, device="auto", detector=None) -> Disc`**
|
|
113
|
+
Runs the three steps below in sequence. `image` can be a file path (with automatic EXIF orientation handling), a `PIL.Image`, or an RGB NumPy array. `device` selects the text recognition backend: `"auto"` uses an OpenCL GPU if present and falls back to CPU, while `"cpu"` forces CPU execution. Raises `discface.DiscNotFound` if no disc is detected.
|
|
114
|
+
|
|
115
|
+
The individual steps can also be called directly:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
geometry = discface.locate("photo.jpg") # DiscGeometry: rim, hole, center in the photo
|
|
119
|
+
flat = discface.rectify("photo.jpg", geometry, size=2048) # flat disc, RGBA, not turned
|
|
120
|
+
o = discface.orient(flat, enhance="always") # Orientation: rotation, support, turned, enhanced, words
|
|
121
|
+
upright = flat.rotate(o.rotation)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
| | |
|
|
125
|
+
|---|---|
|
|
126
|
+
| `Disc.image` | Rectified disc as a `PIL.Image` (RGBA, `size × size`) with transparency outside the rim and inside the centre hole |
|
|
127
|
+
| `Disc.rotation` | Counter-clockwise rotation in degrees applied to align text upright (`0` if unrotated) |
|
|
128
|
+
| `Disc.orientation_support` | Cumulative score of agreeing text; below `min_support`, the disc is left unrotated |
|
|
129
|
+
| `Disc.enhanced` | `True` if faint text was detected using edge-enhanced views |
|
|
130
|
+
| `Disc.words` | Words detected on the disc (`doritex.Word`: `quad`, `angle360`, `score`, …) in `image` coordinates |
|
|
131
|
+
| `Disc.rim`, `.hole`, `.center` | Disc geometry in the input photo: fitted ellipses `((cx, cy), (width, height), angle)` and true centre coordinates |
|
|
132
|
+
| `Disc.save(path, background=None)` | Saves the image to disk. PNG and WebP preserve transparency, while JPEG defaults to a white background |
|
|
133
|
+
| `Disc.to_image(background=None)` | Returns the `PIL.Image`, optionally composited over a solid background colour |
|
|
134
|
+
| `discface.load_detector(device="auto")` | Returns a cached doritex detector instance; pass via `detector=` to reuse across calls |
|
|
135
|
+
|
|
136
|
+
## How upright orientation is decided
|
|
137
|
+
|
|
138
|
+
Every word detected on the disc votes for its reading direction. Large words carry more weight than fine print, and high-confidence detections outweigh uncertain ones. The dominant direction wins, so a few words printed sideways do not skew the result. Small text running circular along the rim is excluded, as it points in every direction. If too little text agrees (`min_support`), the disc retains its original orientation from the photo.
|
|
139
|
+
|
|
140
|
+
Faint print, such as grey text on a silver disc or lettering obscured by rainbow reflections, can also be read from two edge-enhanced views. These views emphasize character strokes, boost local contrast, and suppress smooth shading and reflections. The `enhance` option controls this behavior:
|
|
141
|
+
|
|
142
|
+
- `auto` (default): applies edge enhancement only when standard detection yields insufficient confidence. Discs that read well are processed without extra overhead.
|
|
143
|
+
- `always`: processes edge-enhanced views for every disc, adding roughly 1 to 2 seconds per image.
|
|
144
|
+
- `off`: never applies edge enhancement.
|
|
145
|
+
|
|
146
|
+
## Performance
|
|
147
|
+
|
|
148
|
+
Benchmark times for a 24-megapixel photo after the initial warmup run (which compiles GPU kernels on first execution):
|
|
149
|
+
|
|
150
|
+
| step | GPU (`device="auto"`) | CPU (`device="cpu"`) |
|
|
151
|
+
|---|---|---|
|
|
152
|
+
| read the photo | 0.2 s | 0.2 s |
|
|
153
|
+
| `locate` | 1.1 s | 1.1 s |
|
|
154
|
+
| `rectify` | 0.8 s | 0.8 s |
|
|
155
|
+
| `orient` | 0.3 s | 1.9 s |
|
|
156
|
+
| `orient`, `enhance="always"` | 3.1 s | 8.6 s |
|
|
157
|
+
| **total per photo** | **≈ 2.4 s** | **≈ 4.1 s** |
|
|
158
|
+
|
|
159
|
+
Both devices produce identical results. Only text recognition runs on the GPU; locating and rectifying the disc always run on the CPU.
|
|
160
|
+
|
|
161
|
+
## License
|
|
162
|
+
|
|
163
|
+
MIT
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
discface/__init__.py,sha256=cKfP3KI6iLjqmc32nWRPHZStnd9kogsbbQG28XgtA-0,803
|
|
2
|
+
discface/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
|
|
3
|
+
discface/_disc.py,sha256=2ar9fwJE9f92P3EU82yCjjblW0ZDk1QHGHpC5RDMuHQ,9077
|
|
4
|
+
discface/_enhance.py,sha256=vkNeVCUoy1jOy4QQ90bAAZAyW8NFe4eYE6mjsrJCjkk,2599
|
|
5
|
+
discface/_imaging.py,sha256=S_fLKOePr_IxkT-nsa1GHBcDjc8r6MlFwb1riwWbC1Y,16331
|
|
6
|
+
discface/_orient.py,sha256=WtvW-w93t-Monnz6l9mUIFhtEaijiJGsTj3cfDiobZI,4264
|
|
7
|
+
discface/api.py,sha256=E6JlDDc0zJGVp-iNj6OYFszdaoT251y13f9OdA_hjnI,9502
|
|
8
|
+
discface/cli.py,sha256=yvvmh5eIRgM2qRmklTGB4enM365s9geaqEgc5SPRe6w,4999
|
|
9
|
+
discface/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
10
|
+
discface-0.1.0.dist-info/METADATA,sha256=WmRBNW5EMaiYjnk5Vd8DnKhMHLtUwSnJguTvluEYUSY,8758
|
|
11
|
+
discface-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
12
|
+
discface-0.1.0.dist-info/entry_points.txt,sha256=sk-Pquw4vGTaOtV9CKd8Rno701VOn1DwNAhV092pU2c,47
|
|
13
|
+
discface-0.1.0.dist-info/top_level.txt,sha256=knxNzh5NWQwZcU2_Y0FC73gCIG8g0YP1FGqN6-WQV_E,9
|
|
14
|
+
discface-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
discface
|