discface 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- discface-0.1.0/LICENSE +21 -0
- discface-0.1.0/MANIFEST.in +4 -0
- discface-0.1.0/PKG-INFO +163 -0
- discface-0.1.0/README.md +141 -0
- discface-0.1.0/discface/__init__.py +17 -0
- discface-0.1.0/discface/__main__.py +5 -0
- discface-0.1.0/discface/_disc.py +210 -0
- discface-0.1.0/discface/_enhance.py +62 -0
- discface-0.1.0/discface/_imaging.py +362 -0
- discface-0.1.0/discface/_orient.py +95 -0
- discface-0.1.0/discface/api.py +201 -0
- discface-0.1.0/discface/cli.py +96 -0
- discface-0.1.0/discface/py.typed +0 -0
- discface-0.1.0/discface.egg-info/PKG-INFO +163 -0
- discface-0.1.0/discface.egg-info/SOURCES.txt +19 -0
- discface-0.1.0/discface.egg-info/dependency_links.txt +1 -0
- discface-0.1.0/discface.egg-info/entry_points.txt +2 -0
- discface-0.1.0/discface.egg-info/requires.txt +6 -0
- discface-0.1.0/discface.egg-info/top_level.txt +1 -0
- discface-0.1.0/pyproject.toml +43 -0
- discface-0.1.0/setup.cfg +4 -0
discface-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 discface contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
discface-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: discface
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: From a photo of a CD or DVD to the flat, upright disc image
|
|
5
|
+
Author: discface contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/GodOfLores/discface
|
|
8
|
+
Project-URL: Source, https://github.com/GodOfLores/discface
|
|
9
|
+
Project-URL: Issues, https://github.com/GodOfLores/discface/issues
|
|
10
|
+
Keywords: CD,DVD,disc,rectification,perspective,orientation,doritex
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: doritex>=0.1.0
|
|
18
|
+
Requires-Dist: numpy>=1.23
|
|
19
|
+
Requires-Dist: Pillow>=9.0
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest; extra == "test"
|
|
22
|
+
|
|
23
|
+
# discface
|
|
24
|
+
|
|
25
|
+
Transforms photos of CDs and DVDs into flat, top-down disc images. It removes perspective distortion, cuts out the disc with a transparent outer rim and centre hole, and rotates it so printed text reads upright. The disc boundary is detected geometrically, while text is read with [doritex](https://pypi.org/project/doritex/).
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install discface
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Example
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
import discface
|
|
35
|
+
|
|
36
|
+
disc = discface.extract("blank_dvd.jpg") # photo of a blank DVD+R on a light table
|
|
37
|
+
print(disc.image.size, disc.image.mode)
|
|
38
|
+
print(f"turned by {disc.rotation:.1f}°, faint print enhanced: {disc.enhanced}")
|
|
39
|
+
disc.save("blank_dvd_disc.png")
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Output:
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
(2048, 2048) RGBA
|
|
46
|
+
turned by -17.7°, faint print enhanced: True
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
<table>
|
|
50
|
+
<tr>
|
|
51
|
+
<td align="center"><img src="https://raw.githubusercontent.com/GodOfLores/discface/main/blank_dvd.jpg" alt="photo of a blank DVD lying on a table" width="420"><br><code>blank_dvd.jpg</code></td>
|
|
52
|
+
<td align="center"><img src="https://raw.githubusercontent.com/GodOfLores/discface/main/blank_dvd.png" alt="the same disc, flat, cut out and upright" width="300"><br><code>blank_dvd_disc.png</code></td>
|
|
53
|
+
</tr>
|
|
54
|
+
</table>
|
|
55
|
+
|
|
56
|
+
Perspective distortion is removed, leaving the disc neatly cut out with a transparent outer rim and centre hole. The image was rotated by −17.7° so that "Verbatim" reads horizontally. Because grey print on silver is often faint, it was also detected via edge-enhanced views (`enhanced: True`).
|
|
57
|
+
|
|
58
|
+
## Taking the photo
|
|
59
|
+
|
|
60
|
+
discface is designed for discs placed on a **plain, uniform background** and photographed from roughly above. A sheet of white paper or a light grey tabletop works best, which is how all reference test photos were taken.
|
|
61
|
+
|
|
62
|
+
- Keep the entire disc in frame with some surrounding margin, ensuring the centre hole is visible.
|
|
63
|
+
- Shoot from roughly above. Moderate tilt is corrected automatically, and perspective is rectified precisely.
|
|
64
|
+
- Ensure even lighting. Moderate reflections are tolerated, and faint print benefits from enabling `enhance`.
|
|
65
|
+
- Capture a single disc per photo, keeping other circular objects out of frame.
|
|
66
|
+
|
|
67
|
+
In a benchmark where backgrounds around real discs were replaced while keeping the discs intact, rim detection rates were as follows:
|
|
68
|
+
|
|
69
|
+
| background | rim found |
|
|
70
|
+
|---|---|
|
|
71
|
+
| light, uniform (all real photos) | 7/7 |
|
|
72
|
+
| dark, uniform | 6/7 |
|
|
73
|
+
| mid grey, uniform | 6/7 |
|
|
74
|
+
| busy (photos, printed paper, noise) | 5/35 |
|
|
75
|
+
|
|
76
|
+
On dark backgrounds, centre hole detection is less reliable (4/7). Patterned or cluttered backgrounds are not currently supported.
|
|
77
|
+
|
|
78
|
+
## Command line
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
discface photos/ -o discs/
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
photos/IMG_4097.JPG -> discs/IMG_4097.png turned 39.2° 59 words
|
|
86
|
+
photos/IMG_4100.JPG -> discs/IMG_4100.png turned -1.7° 80 words
|
|
87
|
+
blank_dvd.jpg -> discs/blank_dvd.png turned -17.7° 71 words (faint print, edge-enhanced)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Inputs can be individual photos, directories, or a mix of both. Each disc is saved as `<photo name><suffix>.<format>` in the output folder.
|
|
91
|
+
|
|
92
|
+
| option | meaning |
|
|
93
|
+
|---|---|
|
|
94
|
+
| `-o DIR` | output folder (default: current folder) |
|
|
95
|
+
| `--size N` | output image dimension in px (default 2048) |
|
|
96
|
+
| `--margin N` | margin between the rim and the image border in px (default 32) |
|
|
97
|
+
| `--no-autorotate` | rectify perspective only, without rotating text upright |
|
|
98
|
+
| `--enhance auto\|always\|off` | edge enhancement for faint print, detailed below (default `auto`) |
|
|
99
|
+
| `--min-support X` | minimum text agreement score required to rotate a disc (default 40) |
|
|
100
|
+
| `--background COLOR` | fill transparent areas with a solid colour, e.g. `white` or `'#202020'` |
|
|
101
|
+
| `--format png\|jpg\|webp` | output format; `jpg` defaults to a white background unless `--background` is specified |
|
|
102
|
+
| `--suffix TEXT` | string appended to output filenames |
|
|
103
|
+
| `--device auto\|cpu` | device used for text recognition (OpenCL GPU or CPU) |
|
|
104
|
+
| `--draw` | also save `<name>_words.png` visualizing detected words |
|
|
105
|
+
| `--json` | print one JSON line per photo containing detection metadata |
|
|
106
|
+
| `-q` | quiet mode, suppressing all output except errors |
|
|
107
|
+
|
|
108
|
+
Photos without a detectable disc are reported and skipped, returning exit code 1. You can also run the tool via `python -m discface`.
|
|
109
|
+
|
|
110
|
+
## API
|
|
111
|
+
|
|
112
|
+
**`discface.extract(image, size=2048, margin=32, autorotate=True, enhance="auto", min_support=40, device="auto", detector=None) -> Disc`**
|
|
113
|
+
Runs the three steps below in sequence. `image` can be a file path (with automatic EXIF orientation handling), a `PIL.Image`, or an RGB NumPy array. `device` selects the text recognition backend: `"auto"` uses an OpenCL GPU if present and falls back to CPU, while `"cpu"` forces CPU execution. Raises `discface.DiscNotFound` if no disc is detected.
|
|
114
|
+
|
|
115
|
+
The individual steps can also be called directly:
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
geometry = discface.locate("photo.jpg") # DiscGeometry: rim, hole, center in the photo
|
|
119
|
+
flat = discface.rectify("photo.jpg", geometry, size=2048) # flat disc, RGBA, not turned
|
|
120
|
+
o = discface.orient(flat, enhance="always") # Orientation: rotation, support, turned, enhanced, words
|
|
121
|
+
upright = flat.rotate(o.rotation)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
| | |
|
|
125
|
+
|---|---|
|
|
126
|
+
| `Disc.image` | Rectified disc as a `PIL.Image` (RGBA, `size × size`) with transparency outside the rim and inside the centre hole |
|
|
127
|
+
| `Disc.rotation` | Counter-clockwise rotation in degrees applied to align text upright (`0` if unrotated) |
|
|
128
|
+
| `Disc.orientation_support` | Cumulative score of agreeing text; below `min_support`, the disc is left unrotated |
|
|
129
|
+
| `Disc.enhanced` | `True` if faint text was detected using edge-enhanced views |
|
|
130
|
+
| `Disc.words` | Words detected on the disc (`doritex.Word`: `quad`, `angle360`, `score`, …) in `image` coordinates |
|
|
131
|
+
| `Disc.rim`, `.hole`, `.center` | Disc geometry in the input photo: fitted ellipses `((cx, cy), (width, height), angle)` and true centre coordinates |
|
|
132
|
+
| `Disc.save(path, background=None)` | Saves the image to disk. PNG and WebP preserve transparency, while JPEG defaults to a white background |
|
|
133
|
+
| `Disc.to_image(background=None)` | Returns the `PIL.Image`, optionally composited over a solid background colour |
|
|
134
|
+
| `discface.load_detector(device="auto")` | Returns a cached doritex detector instance; pass via `detector=` to reuse across calls |
|
|
135
|
+
|
|
136
|
+
## How upright orientation is decided
|
|
137
|
+
|
|
138
|
+
Every word detected on the disc votes for its reading direction. Large words carry more weight than fine print, and high-confidence detections outweigh uncertain ones. The dominant direction wins, so a few words printed sideways do not skew the result. Small text running circular along the rim is excluded, as it points in every direction. If too little text agrees (`min_support`), the disc retains its original orientation from the photo.
|
|
139
|
+
|
|
140
|
+
Faint print, such as grey text on a silver disc or lettering obscured by rainbow reflections, can also be read from two edge-enhanced views. These views emphasize character strokes, boost local contrast, and suppress smooth shading and reflections. The `enhance` option controls this behavior:
|
|
141
|
+
|
|
142
|
+
- `auto` (default): applies edge enhancement only when standard detection yields insufficient confidence. Discs that read well are processed without extra overhead.
|
|
143
|
+
- `always`: processes edge-enhanced views for every disc, adding roughly 1 to 2 seconds per image.
|
|
144
|
+
- `off`: never applies edge enhancement.
|
|
145
|
+
|
|
146
|
+
## Performance
|
|
147
|
+
|
|
148
|
+
Benchmark times for a 24-megapixel photo after the initial warmup run (which compiles GPU kernels on first execution):
|
|
149
|
+
|
|
150
|
+
| step | GPU (`device="auto"`) | CPU (`device="cpu"`) |
|
|
151
|
+
|---|---|---|
|
|
152
|
+
| read the photo | 0.2 s | 0.2 s |
|
|
153
|
+
| `locate` | 1.1 s | 1.1 s |
|
|
154
|
+
| `rectify` | 0.8 s | 0.8 s |
|
|
155
|
+
| `orient` | 0.3 s | 1.9 s |
|
|
156
|
+
| `orient`, `enhance="always"` | 3.1 s | 8.6 s |
|
|
157
|
+
| **total per photo** | **≈ 2.4 s** | **≈ 4.1 s** |
|
|
158
|
+
|
|
159
|
+
Both devices produce identical results. Only text recognition runs on the GPU; locating and rectifying the disc always run on the CPU.
|
|
160
|
+
|
|
161
|
+
## License
|
|
162
|
+
|
|
163
|
+
MIT
|
discface-0.1.0/README.md
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# discface
|
|
2
|
+
|
|
3
|
+
Transforms photos of CDs and DVDs into flat, top-down disc images. It removes perspective distortion, cuts out the disc with a transparent outer rim and centre hole, and rotates it so printed text reads upright. The disc boundary is detected geometrically, while text is read with [doritex](https://pypi.org/project/doritex/).
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install discface
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
## Example
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
import discface
|
|
13
|
+
|
|
14
|
+
disc = discface.extract("blank_dvd.jpg") # photo of a blank DVD+R on a light table
|
|
15
|
+
print(disc.image.size, disc.image.mode)
|
|
16
|
+
print(f"turned by {disc.rotation:.1f}°, faint print enhanced: {disc.enhanced}")
|
|
17
|
+
disc.save("blank_dvd_disc.png")
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
Output:
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
(2048, 2048) RGBA
|
|
24
|
+
turned by -17.7°, faint print enhanced: True
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
<table>
|
|
28
|
+
<tr>
|
|
29
|
+
<td align="center"><img src="https://raw.githubusercontent.com/GodOfLores/discface/main/blank_dvd.jpg" alt="photo of a blank DVD lying on a table" width="420"><br><code>blank_dvd.jpg</code></td>
|
|
30
|
+
<td align="center"><img src="https://raw.githubusercontent.com/GodOfLores/discface/main/blank_dvd.png" alt="the same disc, flat, cut out and upright" width="300"><br><code>blank_dvd_disc.png</code></td>
|
|
31
|
+
</tr>
|
|
32
|
+
</table>
|
|
33
|
+
|
|
34
|
+
Perspective distortion is removed, leaving the disc neatly cut out with a transparent outer rim and centre hole. The image was rotated by −17.7° so that "Verbatim" reads horizontally. Because grey print on silver is often faint, it was also detected via edge-enhanced views (`enhanced: True`).
|
|
35
|
+
|
|
36
|
+
## Taking the photo
|
|
37
|
+
|
|
38
|
+
discface is designed for discs placed on a **plain, uniform background** and photographed from roughly above. A sheet of white paper or a light grey tabletop works best, which is how all reference test photos were taken.
|
|
39
|
+
|
|
40
|
+
- Keep the entire disc in frame with some surrounding margin, ensuring the centre hole is visible.
|
|
41
|
+
- Shoot from roughly above. Moderate tilt is corrected automatically, and perspective is rectified precisely.
|
|
42
|
+
- Ensure even lighting. Moderate reflections are tolerated, and faint print benefits from enabling `enhance`.
|
|
43
|
+
- Capture a single disc per photo, keeping other circular objects out of frame.
|
|
44
|
+
|
|
45
|
+
In a benchmark where backgrounds around real discs were replaced while keeping the discs intact, rim detection rates were as follows:
|
|
46
|
+
|
|
47
|
+
| background | rim found |
|
|
48
|
+
|---|---|
|
|
49
|
+
| light, uniform (all real photos) | 7/7 |
|
|
50
|
+
| dark, uniform | 6/7 |
|
|
51
|
+
| mid grey, uniform | 6/7 |
|
|
52
|
+
| busy (photos, printed paper, noise) | 5/35 |
|
|
53
|
+
|
|
54
|
+
On dark backgrounds, centre hole detection is less reliable (4/7). Patterned or cluttered backgrounds are not currently supported.
|
|
55
|
+
|
|
56
|
+
## Command line
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
discface photos/ -o discs/
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
photos/IMG_4097.JPG -> discs/IMG_4097.png turned 39.2° 59 words
|
|
64
|
+
photos/IMG_4100.JPG -> discs/IMG_4100.png turned -1.7° 80 words
|
|
65
|
+
blank_dvd.jpg -> discs/blank_dvd.png turned -17.7° 71 words (faint print, edge-enhanced)
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Inputs can be individual photos, directories, or a mix of both. Each disc is saved as `<photo name><suffix>.<format>` in the output folder.
|
|
69
|
+
|
|
70
|
+
| option | meaning |
|
|
71
|
+
|---|---|
|
|
72
|
+
| `-o DIR` | output folder (default: current folder) |
|
|
73
|
+
| `--size N` | output image dimension in px (default 2048) |
|
|
74
|
+
| `--margin N` | margin between the rim and the image border in px (default 32) |
|
|
75
|
+
| `--no-autorotate` | rectify perspective only, without rotating text upright |
|
|
76
|
+
| `--enhance auto\|always\|off` | edge enhancement for faint print, detailed below (default `auto`) |
|
|
77
|
+
| `--min-support X` | minimum text agreement score required to rotate a disc (default 40) |
|
|
78
|
+
| `--background COLOR` | fill transparent areas with a solid colour, e.g. `white` or `'#202020'` |
|
|
79
|
+
| `--format png\|jpg\|webp` | output format; `jpg` defaults to a white background unless `--background` is specified |
|
|
80
|
+
| `--suffix TEXT` | string appended to output filenames |
|
|
81
|
+
| `--device auto\|cpu` | device used for text recognition (OpenCL GPU or CPU) |
|
|
82
|
+
| `--draw` | also save `<name>_words.png` visualizing detected words |
|
|
83
|
+
| `--json` | print one JSON line per photo containing detection metadata |
|
|
84
|
+
| `-q` | quiet mode, suppressing all output except errors |
|
|
85
|
+
|
|
86
|
+
Photos without a detectable disc are reported and skipped, returning exit code 1. You can also run the tool via `python -m discface`.
|
|
87
|
+
|
|
88
|
+
## API
|
|
89
|
+
|
|
90
|
+
**`discface.extract(image, size=2048, margin=32, autorotate=True, enhance="auto", min_support=40, device="auto", detector=None) -> Disc`**
|
|
91
|
+
Runs the three steps below in sequence. `image` can be a file path (with automatic EXIF orientation handling), a `PIL.Image`, or an RGB NumPy array. `device` selects the text recognition backend: `"auto"` uses an OpenCL GPU if present and falls back to CPU, while `"cpu"` forces CPU execution. Raises `discface.DiscNotFound` if no disc is detected.
|
|
92
|
+
|
|
93
|
+
The individual steps can also be called directly:
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
geometry = discface.locate("photo.jpg") # DiscGeometry: rim, hole, center in the photo
|
|
97
|
+
flat = discface.rectify("photo.jpg", geometry, size=2048) # flat disc, RGBA, not turned
|
|
98
|
+
o = discface.orient(flat, enhance="always") # Orientation: rotation, support, turned, enhanced, words
|
|
99
|
+
upright = flat.rotate(o.rotation)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
| | |
|
|
103
|
+
|---|---|
|
|
104
|
+
| `Disc.image` | Rectified disc as a `PIL.Image` (RGBA, `size × size`) with transparency outside the rim and inside the centre hole |
|
|
105
|
+
| `Disc.rotation` | Counter-clockwise rotation in degrees applied to align text upright (`0` if unrotated) |
|
|
106
|
+
| `Disc.orientation_support` | Cumulative score of agreeing text; below `min_support`, the disc is left unrotated |
|
|
107
|
+
| `Disc.enhanced` | `True` if faint text was detected using edge-enhanced views |
|
|
108
|
+
| `Disc.words` | Words detected on the disc (`doritex.Word`: `quad`, `angle360`, `score`, …) in `image` coordinates |
|
|
109
|
+
| `Disc.rim`, `.hole`, `.center` | Disc geometry in the input photo: fitted ellipses `((cx, cy), (width, height), angle)` and true centre coordinates |
|
|
110
|
+
| `Disc.save(path, background=None)` | Saves the image to disk. PNG and WebP preserve transparency, while JPEG defaults to a white background |
|
|
111
|
+
| `Disc.to_image(background=None)` | Returns the `PIL.Image`, optionally composited over a solid background colour |
|
|
112
|
+
| `discface.load_detector(device="auto")` | Returns a cached doritex detector instance; pass via `detector=` to reuse across calls |
|
|
113
|
+
|
|
114
|
+
## How upright orientation is decided
|
|
115
|
+
|
|
116
|
+
Every word detected on the disc votes for its reading direction. Large words carry more weight than fine print, and high-confidence detections outweigh uncertain ones. The dominant direction wins, so a few words printed sideways do not skew the result. Small text running circular along the rim is excluded, as it points in every direction. If too little text agrees (`min_support`), the disc retains its original orientation from the photo.
|
|
117
|
+
|
|
118
|
+
Faint print, such as grey text on a silver disc or lettering obscured by rainbow reflections, can also be read from two edge-enhanced views. These views emphasize character strokes, boost local contrast, and suppress smooth shading and reflections. The `enhance` option controls this behavior:
|
|
119
|
+
|
|
120
|
+
- `auto` (default): applies edge enhancement only when standard detection yields insufficient confidence. Discs that read well are processed without extra overhead.
|
|
121
|
+
- `always`: processes edge-enhanced views for every disc, adding roughly 1 to 2 seconds per image.
|
|
122
|
+
- `off`: never applies edge enhancement.
|
|
123
|
+
|
|
124
|
+
## Performance
|
|
125
|
+
|
|
126
|
+
Benchmark times for a 24-megapixel photo after the initial warmup run (which compiles GPU kernels on first execution):
|
|
127
|
+
|
|
128
|
+
| step | GPU (`device="auto"`) | CPU (`device="cpu"`) |
|
|
129
|
+
|---|---|---|
|
|
130
|
+
| read the photo | 0.2 s | 0.2 s |
|
|
131
|
+
| `locate` | 1.1 s | 1.1 s |
|
|
132
|
+
| `rectify` | 0.8 s | 0.8 s |
|
|
133
|
+
| `orient` | 0.3 s | 1.9 s |
|
|
134
|
+
| `orient`, `enhance="always"` | 3.1 s | 8.6 s |
|
|
135
|
+
| **total per photo** | **≈ 2.4 s** | **≈ 4.1 s** |
|
|
136
|
+
|
|
137
|
+
Both devices produce identical results. Only text recognition runs on the GPU; locating and rectifying the disc always run on the CPU.
|
|
138
|
+
|
|
139
|
+
## License
|
|
140
|
+
|
|
141
|
+
MIT
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""discface: from a photo of a CD / DVD to the flat, upright disc.
|
|
2
|
+
|
|
3
|
+
import discface
|
|
4
|
+
|
|
5
|
+
disc = discface.extract('photo.jpg') # locate, rectify, turn upright
|
|
6
|
+
disc.save('disc.png') # 2048 x 2048 RGBA, transparent outside the rim and in the hole
|
|
7
|
+
|
|
8
|
+
geometry = discface.locate('photo.jpg') # the single steps
|
|
9
|
+
flat = discface.rectify('photo.jpg', geometry)
|
|
10
|
+
o = discface.orient(flat, enhance='always')
|
|
11
|
+
upright = flat.rotate(o.rotation)
|
|
12
|
+
"""
|
|
13
|
+
from .api import Disc, DiscGeometry, DiscNotFound, Orientation, extract, load_detector, locate, orient, rectify
|
|
14
|
+
|
|
15
|
+
__version__ = '0.1.0'
|
|
16
|
+
__all__ = ['Disc', 'DiscGeometry', 'DiscNotFound', 'Orientation', 'extract', 'load_detector', 'locate', 'orient',
|
|
17
|
+
'rectify', '__version__']
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Finding the disc in a photo and rectifying it (pure geometry, no learning).
|
|
2
|
+
|
|
3
|
+
1. Edges of the image (difference to the background colour + grey-level Canny) at half resolution.
|
|
4
|
+
2. Ellipse candidates from the edge contours; the rim is the large, well-supported one, the hub rings are the
|
|
5
|
+
small ones near its centre.
|
|
6
|
+
3. The true disc centre is extrapolated from the hub ellipses: centres of projected concentric circles drift
|
|
7
|
+
linearly with r^2.
|
|
8
|
+
4. The exact homography maps the rim conic to a circle: the polar of the centre w.r.t. the rim is the vanishing
|
|
9
|
+
line of the disc plane; the remaining affine part is a Cholesky factor of the conic.
|
|
10
|
+
5. The centre hole is the innermost ring covered over at least half its circumference in the rectified plane.
|
|
11
|
+
6. Lanczos4 resampling of the disc annulus into a square image with an anti-aliased alpha mask.
|
|
12
|
+
"""
|
|
13
|
+
import numpy as np
|
|
14
|
+
|
|
15
|
+
from ._imaging import (resize_linear, normalize_minmax_u8, gaussian_blur, canny, rgb2gray, find_contours,
|
|
16
|
+
fit_ellipse, ellipse_pixels, warp_perspective)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def filter_and_detect_edges(img, scale=0.5):
|
|
20
|
+
small = resize_linear(img, scale)
|
|
21
|
+
bg = np.median(np.concatenate([
|
|
22
|
+
small[:20, :].reshape(-1, 3),
|
|
23
|
+
small[-20:, :].reshape(-1, 3),
|
|
24
|
+
small[:, :20].reshape(-1, 3),
|
|
25
|
+
small[:, -20:].reshape(-1, 3)
|
|
26
|
+
]), axis=0)
|
|
27
|
+
diff = np.linalg.norm(small.astype(np.float32) - bg, axis=2)
|
|
28
|
+
diff_norm = normalize_minmax_u8(diff)
|
|
29
|
+
blurred_diff = gaussian_blur(diff_norm, 9, 2)
|
|
30
|
+
edges_diff = canny(blurred_diff, 30, 80)
|
|
31
|
+
|
|
32
|
+
gray = rgb2gray(small)
|
|
33
|
+
blurred_gray = gaussian_blur(gray, 5, 1.2)
|
|
34
|
+
edges_gray = canny(blurred_gray, 20, 60)
|
|
35
|
+
|
|
36
|
+
combined_edges = np.maximum(edges_diff, edges_gray)
|
|
37
|
+
return combined_edges, small
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def detect_candidate_ellipses(edges):
|
|
41
|
+
sh, sw = edges.shape[:2]
|
|
42
|
+
cnts = find_contours(edges)
|
|
43
|
+
candidates = []
|
|
44
|
+
for c in cnts:
|
|
45
|
+
if len(c) >= 30:
|
|
46
|
+
box = fit_ellipse(c)
|
|
47
|
+
if box is None:
|
|
48
|
+
continue
|
|
49
|
+
(cx, cy), (d1, d2), ang = box
|
|
50
|
+
a, b = max(d1, d2) / 2.0, min(d1, d2) / 2.0
|
|
51
|
+
if a > 15 and b > 15 and (b / a) > 0.7:
|
|
52
|
+
if 0.15 * sw < cx < 0.85 * sw and 0.15 * sh < cy < 0.85 * sh:
|
|
53
|
+
py, px = ellipse_pixels(box, (sh, sw), thickness=2)
|
|
54
|
+
votes = np.count_nonzero(edges[py, px])
|
|
55
|
+
perim = 2 * np.pi * np.sqrt((a**2 + b**2) / 2.0)
|
|
56
|
+
score = votes / (perim + 1e-5)
|
|
57
|
+
if score > 0.25:
|
|
58
|
+
candidates.append({
|
|
59
|
+
'box': box,
|
|
60
|
+
'cx': cx,
|
|
61
|
+
'cy': cy,
|
|
62
|
+
'a': a,
|
|
63
|
+
'b': b,
|
|
64
|
+
'ang': ang,
|
|
65
|
+
'score': score
|
|
66
|
+
})
|
|
67
|
+
return candidates
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def ellipse_to_conic(box):
|
|
71
|
+
(cx, cy), (d1, d2), ang = box
|
|
72
|
+
t = np.radians(ang)
|
|
73
|
+
R = np.array([[np.cos(t), -np.sin(t)], [np.sin(t), np.cos(t)]])
|
|
74
|
+
A = np.eye(3)
|
|
75
|
+
A[:2, :2] = R.T
|
|
76
|
+
A[:2, 2] = -R.T @ np.array([cx, cy])
|
|
77
|
+
return A.T @ np.diag([4.0 / d1**2, 4.0 / d2**2, -1.0]) @ A
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def homography_from_rim_and_center(outer_box, center, size=2048, margin=32):
|
|
81
|
+
(cx, cy), (d1, d2), _ = outer_box
|
|
82
|
+
s = 2.0 / (d1 + d2)
|
|
83
|
+
N = np.array([[s, 0, -s * cx], [0, s, -s * cy], [0, 0, 1.0]])
|
|
84
|
+
Ni = np.linalg.inv(N)
|
|
85
|
+
C = Ni.T @ ellipse_to_conic(outer_box) @ Ni
|
|
86
|
+
C /= np.abs(C).max()
|
|
87
|
+
c = N @ np.array([center[0], center[1], 1.0])
|
|
88
|
+
|
|
89
|
+
# the polar of the disc centre w.r.t. the rim conic is the vanishing line of the disc plane
|
|
90
|
+
l = C @ c
|
|
91
|
+
l /= l[2]
|
|
92
|
+
H_proj = np.array([[1, 0, 0], [0, 1, 0], [l[0], l[1], 1.0]])
|
|
93
|
+
Hpi = np.linalg.inv(H_proj)
|
|
94
|
+
C_aff = Hpi.T @ C @ Hpi
|
|
95
|
+
|
|
96
|
+
# remaining affine ellipse -> unit circle
|
|
97
|
+
M, m, k = C_aff[:2, :2], C_aff[:2, 2], C_aff[2, 2]
|
|
98
|
+
x0 = -np.linalg.solve(M, m)
|
|
99
|
+
L = np.linalg.cholesky(M / (m @ np.linalg.solve(M, m) - k))
|
|
100
|
+
H_aff = np.eye(3)
|
|
101
|
+
H_aff[:2, :2] = L.T
|
|
102
|
+
H_aff[:2, 2] = -L.T @ x0
|
|
103
|
+
H_unit = H_aff @ H_proj @ N
|
|
104
|
+
|
|
105
|
+
# in-plane rotation: keep the image "down" direction pointing down
|
|
106
|
+
c_img = np.linalg.inv(H_unit) @ np.array([0, 0, 1.0])
|
|
107
|
+
c_img /= c_img[2]
|
|
108
|
+
q = H_unit @ np.array([c_img[0], c_img[1] + 0.3 * (d1 + d2) / 2.0, 1.0])
|
|
109
|
+
q /= q[2]
|
|
110
|
+
phi = np.arctan2(q[0], q[1])
|
|
111
|
+
R = np.array([[np.cos(phi), -np.sin(phi), 0], [np.sin(phi), np.cos(phi), 0], [0, 0, 1.0]])
|
|
112
|
+
|
|
113
|
+
tc = size / 2.0
|
|
114
|
+
tr = tc - margin
|
|
115
|
+
S = np.array([[tr, 0, tc], [0, tr, tc], [0, 0, 1.0]])
|
|
116
|
+
H = S @ R @ H_unit
|
|
117
|
+
return H / H[2, 2]
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def select_hole_ring(edges, outer_box, center, size=2048, margin=32, min_coverage=0.5):
|
|
121
|
+
# Hough-like radius accumulator around the consensus centre in the rectified plane:
|
|
122
|
+
# a real ring must be covered over at least half of its circumference; the hole is the innermost one
|
|
123
|
+
# inside the search window given by the disc standard (15 mm hole on a 120 mm disc -> 0.125 * r)
|
|
124
|
+
H = homography_from_rim_and_center(outer_box, center, size, margin)
|
|
125
|
+
tc = size / 2.0
|
|
126
|
+
tr = tc - margin
|
|
127
|
+
ys, xs = np.nonzero(edges)
|
|
128
|
+
P = H @ np.stack([xs.astype(np.float64), ys.astype(np.float64), np.ones(len(xs))])
|
|
129
|
+
x, y = P[0] / P[2] - tc, P[1] / P[2] - tc
|
|
130
|
+
r = np.hypot(x, y)
|
|
131
|
+
ang_bin = ((np.arctan2(y, x) + np.pi) / (2 * np.pi) * 360).astype(int) % 360
|
|
132
|
+
|
|
133
|
+
radii = np.arange(0.10 * tr, 0.15 * tr, 0.5)
|
|
134
|
+
coverage = np.array([len(np.unique(ang_bin[np.abs(r - R) < 1.5])) / 360.0 for R in radii])
|
|
135
|
+
for i in range(1, len(radii) - 1):
|
|
136
|
+
if coverage[i] >= min_coverage and coverage[i] >= coverage[i - 1] and coverage[i] >= coverage[i + 1]:
|
|
137
|
+
t = np.linspace(0, 2 * np.pi, 360, endpoint=False)
|
|
138
|
+
Q = np.linalg.inv(H) @ np.stack([tc + radii[i] * np.cos(t), tc + radii[i] * np.sin(t), np.ones_like(t)])
|
|
139
|
+
return fit_ellipse(np.stack([Q[0] / Q[2], Q[1] / Q[2]], 1))
|
|
140
|
+
return None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def select_outer_and_inner_ellipses(candidates, edges):
|
|
144
|
+
sh, sw = edges.shape[:2]
|
|
145
|
+
cd_cands = [c for c in candidates if c['a'] >= 45.0]
|
|
146
|
+
outers = [c for c in cd_cands if c['a'] > 0.25 * min(sw, sh)]
|
|
147
|
+
if not outers:
|
|
148
|
+
outers = [max(cd_cands, key=lambda c: c['a'])]
|
|
149
|
+
best_outer = max(outers, key=lambda c: (c['score'], c['a']))
|
|
150
|
+
|
|
151
|
+
inners = [
|
|
152
|
+
c for c in cd_cands
|
|
153
|
+
if 0.08 * best_outer['a'] <= c['a'] <= 0.40 * best_outer['a']
|
|
154
|
+
and np.hypot(c['cx'] - best_outer['cx'], c['cy'] - best_outer['cy']) < 85.0
|
|
155
|
+
]
|
|
156
|
+
|
|
157
|
+
for c in inners:
|
|
158
|
+
close_members = [o for o in inners if np.hypot(c['cx'] - o['cx'], c['cy'] - o['cy']) < 18.0]
|
|
159
|
+
c['consensus_weight'] = sum(o['score'] for o in close_members)
|
|
160
|
+
c['consensus_members'] = close_members
|
|
161
|
+
|
|
162
|
+
best_inner_anchor = max(inners, key=lambda c: c['consensus_weight'])
|
|
163
|
+
consensus_inners = best_inner_anchor['consensus_members']
|
|
164
|
+
|
|
165
|
+
shared_cx = float(np.median([c['cx'] for c in consensus_inners]))
|
|
166
|
+
shared_cy = float(np.median([c['cy'] for c in consensus_inners]))
|
|
167
|
+
|
|
168
|
+
# projected ellipse centres of concentric circles drift linearly with r^2 -> extrapolate to r = 0
|
|
169
|
+
rho2 = (float(np.median([c['a'] for c in consensus_inners])) / best_outer['a']) ** 2
|
|
170
|
+
center = (
|
|
171
|
+
(shared_cx - rho2 * best_outer['cx']) / (1.0 - rho2),
|
|
172
|
+
(shared_cy - rho2 * best_outer['cy']) / (1.0 - rho2)
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
outer_box = best_outer['box']
|
|
176
|
+
inner_box = select_hole_ring(edges, outer_box, center)
|
|
177
|
+
if inner_box is None:
|
|
178
|
+
hole_cands = [c for c in consensus_inners if c['a'] < 0.16 * best_outer['a']]
|
|
179
|
+
if hole_cands:
|
|
180
|
+
innermost = min(hole_cands, key=lambda c: c['a'])
|
|
181
|
+
else:
|
|
182
|
+
innermost = min(consensus_inners, key=lambda c: c['a'])
|
|
183
|
+
inner_box = innermost['box']
|
|
184
|
+
return outer_box, inner_box, center
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def rectify_and_project_to_circle(img, outer_box, inner_box, center, scale=0.5, size=2048, margin=32):
|
|
188
|
+
up = lambda b: ((b[0][0] / scale, b[0][1] / scale), (b[1][0] / scale, b[1][1] / scale), b[2])
|
|
189
|
+
outer_box, inner_box = up(outer_box), up(inner_box)
|
|
190
|
+
center = (center[0] / scale, center[1] / scale)
|
|
191
|
+
|
|
192
|
+
H = homography_from_rim_and_center(outer_box, center, size, margin)
|
|
193
|
+
|
|
194
|
+
tc = size / 2.0
|
|
195
|
+
tr = tc - margin
|
|
196
|
+
t = np.linspace(0, 2 * np.pi, 360, endpoint=False)
|
|
197
|
+
(icx, icy), (id1, id2), iang = inner_box
|
|
198
|
+
ia = np.radians(iang)
|
|
199
|
+
ex, ey = id1 / 2.0 * np.cos(t), id2 / 2.0 * np.sin(t)
|
|
200
|
+
hole = H @ np.stack([icx + ex * np.cos(ia) - ey * np.sin(ia), icy + ex * np.sin(ia) + ey * np.cos(ia), np.ones_like(t)])
|
|
201
|
+
hole_r = float(np.median(np.hypot(hole[0] / hole[2] - tc, hole[1] / hole[2] - tc)))
|
|
202
|
+
|
|
203
|
+
# analytic anti-aliased annulus (1px ramp) instead of a blurred hard mask
|
|
204
|
+
yy, xx = np.mgrid[0:size, 0:size].astype(np.float32)
|
|
205
|
+
d = np.hypot(xx - tc, yy - tc)
|
|
206
|
+
alpha = np.clip(tr - d + 0.5, 0, 1) * np.clip(d - hole_r + 0.5, 0, 1)
|
|
207
|
+
|
|
208
|
+
# only the visible disc is resampled; fully transparent pixels stay black
|
|
209
|
+
warped = warp_perspective(img, H, size, mask=alpha > 0)
|
|
210
|
+
return np.dstack([warped, (alpha * 255 + 0.5).astype(np.uint8)])
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Edge enhancement for faint print (grey on silver, print under rainbow reflections).
|
|
2
|
+
|
|
3
|
+
Two complementary views of the disc, both multi-scale band-passes (differences of Gaussians) on the luminance:
|
|
4
|
+
they keep stroke-sized detail with its polarity (dark print stays dark) and remove smooth shading and reflections,
|
|
5
|
+
each band normalised by its local energy. The second view first normalises the local contrast (local mean / local
|
|
6
|
+
standard deviation) and maps the result through a soft S-curve, which pulls very faint print further up. Their
|
|
7
|
+
errors differ, so their words vote together (evaluated on real discs reduced to 8-20 % contrast with reflections,
|
|
8
|
+
blur and JPEG: 18 of 19 correct, against 17 for either view alone)."""
|
|
9
|
+
import numpy as np
|
|
10
|
+
from PIL import Image
|
|
11
|
+
|
|
12
|
+
_BANDS = ((0.8, 3.0), (1.5, 6.0), (3.0, 12.0))
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _box(a, r, axis):
|
|
16
|
+
"""Mean over a window of 2r+1 along axis (edge-replicated), via cumulative sums."""
|
|
17
|
+
if r < 1:
|
|
18
|
+
return a
|
|
19
|
+
pad = [(0, 0)] * a.ndim
|
|
20
|
+
pad[axis] = (r + 1, r)
|
|
21
|
+
c = np.cumsum(np.pad(a, pad, mode='edge'), axis=axis, dtype=np.float64)
|
|
22
|
+
n = a.shape[axis]
|
|
23
|
+
hi = np.take(c, np.arange(2 * r + 1, 2 * r + 1 + n), axis=axis)
|
|
24
|
+
lo = np.take(c, np.arange(0, n), axis=axis)
|
|
25
|
+
return ((hi - lo) / (2 * r + 1)).astype(np.float32)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _blur(a, sigma):
|
|
29
|
+
"""Gaussian of std sigma, approximated by three box filters per axis."""
|
|
30
|
+
r = max(0, int(round((np.sqrt(4.0 * sigma * sigma + 1.0) - 1.0) / 2.0)))
|
|
31
|
+
a = a.astype(np.float32)
|
|
32
|
+
for axis in (0, 1):
|
|
33
|
+
for _ in range(3):
|
|
34
|
+
a = _box(a, r, axis)
|
|
35
|
+
return a
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _bands(lum):
|
|
39
|
+
acc = np.zeros_like(lum)
|
|
40
|
+
for s1, s2 in _BANDS:
|
|
41
|
+
d = _blur(lum, s1) - _blur(lum, s2)
|
|
42
|
+
acc += d / (np.sqrt(_blur(d * d, 4.0 * s2)) + 1.5)
|
|
43
|
+
return acc
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _grey(v, alpha):
|
|
47
|
+
g = np.clip(v, 0, 255).astype(np.uint8)
|
|
48
|
+
g[alpha == 0] = 200
|
|
49
|
+
return Image.fromarray(np.repeat(g[..., None], 3, -1))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def enhanced_views(rgba: Image.Image, size: int = 1024):
|
|
53
|
+
"""The disc at ``size`` px as grey images in which the print stands out (outside the disc: light grey):
|
|
54
|
+
[band-pass, local-contrast-normalised band-pass with S-curve]."""
|
|
55
|
+
a = np.asarray(rgba.convert('RGBA').resize((size, size), Image.LANCZOS)).astype(np.float32)
|
|
56
|
+
alpha = a[..., 3]
|
|
57
|
+
lum = 0.299 * a[..., 0] + 0.587 * a[..., 1] + 0.114 * a[..., 2]
|
|
58
|
+
plain = _grey(128.0 + 40.0 * _bands(lum), alpha)
|
|
59
|
+
m = _blur(lum, 16.0)
|
|
60
|
+
lcn = 40.0 * (lum - m) / (np.sqrt(_blur((lum - m) ** 2, 16.0)) + 2.0)
|
|
61
|
+
curved = _grey(128.0 + 127.0 * np.tanh(0.45 * _bands(lcn)), alpha)
|
|
62
|
+
return [plain, curved]
|