ovkit 0.1.2__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ovkit-0.1.2 → ovkit-0.2.0}/PKG-INFO +66 -4
- {ovkit-0.1.2 → ovkit-0.2.0}/README.md +65 -3
- {ovkit-0.1.2 → ovkit-0.2.0}/examples/README.md +1 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/pyproject.toml +1 -1
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/__init__.py +16 -4
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/__main__.py +30 -15
- ovkit-0.2.0/src/ovkit/audio/__init__.py +13 -0
- ovkit-0.2.0/src/ovkit/audio/ops.py +108 -0
- ovkit-0.2.0/src/ovkit/audio/plot.py +47 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/backend.py +46 -1
- ovkit-0.2.0/src/ovkit/core/constants.py +335 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/download.py +24 -1
- ovkit-0.2.0/src/ovkit/core/i18n.py +309 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/model.py +205 -19
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/registry.py +21 -1
- ovkit-0.2.0/src/ovkit/core/results.py +598 -0
- ovkit-0.2.0/src/ovkit/data/imagenet1000.txt +1000 -0
- ovkit-0.2.0/src/ovkit/face/__init__.py +8 -0
- ovkit-0.2.0/src/ovkit/face/analyzer.py +7 -0
- ovkit-0.2.0/src/ovkit/genai/__init__.py +40 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/genai/pipelines.py +105 -1
- ovkit-0.2.0/src/ovkit/gui/__init__.py +20 -0
- ovkit-0.2.0/src/ovkit/gui/app.py +286 -0
- ovkit-0.2.0/src/ovkit/gui/controller.py +291 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/manifests/aliases.yaml +11 -4
- ovkit-0.2.0/src/ovkit/manifests/genai.yaml +156 -0
- ovkit-0.2.0/src/ovkit/manifests/labels.yaml +77 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/manifests/omz.yaml +215 -73
- ovkit-0.2.0/src/ovkit/pipelines/__init__.py +165 -0
- ovkit-0.2.0/src/ovkit/pipelines/analyze.py +165 -0
- ovkit-0.2.0/src/ovkit/pipelines/attention.py +114 -0
- ovkit-0.2.0/src/ovkit/pipelines/base.py +97 -0
- ovkit-0.2.0/src/ovkit/pipelines/gaze.py +227 -0
- ovkit-0.2.0/src/ovkit/pipelines/plates.py +137 -0
- ovkit-0.2.0/src/ovkit/pipelines/privacy.py +128 -0
- ovkit-0.2.0/src/ovkit/pipelines/reid.py +127 -0
- ovkit-0.2.0/src/ovkit/pipelines/scene.py +129 -0
- ovkit-0.2.0/src/ovkit/pipelines/temporal.py +226 -0
- ovkit-0.2.0/src/ovkit/pipelines/text.py +81 -0
- ovkit-0.2.0/src/ovkit/pipelines/tracking.py +120 -0
- ovkit-0.2.0/src/ovkit/recognize/audio.py +117 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/base.py +24 -3
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/classify.py +57 -8
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/detect.py +61 -1
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/generic.py +30 -2
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/segment.py +9 -3
- ovkit-0.2.0/src/ovkit/solutions/__init__.py +14 -0
- ovkit-0.2.0/src/ovkit/solutions/anomaly.py +93 -0
- ovkit-0.2.0/src/ovkit/solutions/ocr.py +7 -0
- ovkit-0.2.0/src/ovkit/solutions/reid.py +7 -0
- ovkit-0.2.0/src/ovkit/solutions/tracking.py +7 -0
- ovkit-0.1.2/src/ovkit/core/constants.py +0 -158
- ovkit-0.1.2/src/ovkit/core/results.py +0 -322
- ovkit-0.1.2/src/ovkit/face/__init__.py +0 -19
- ovkit-0.1.2/src/ovkit/face/analyzer.py +0 -26
- ovkit-0.1.2/src/ovkit/genai/__init__.py +0 -25
- ovkit-0.1.2/src/ovkit/manifests/genai.yaml +0 -38
- ovkit-0.1.2/src/ovkit/solutions/__init__.py +0 -9
- ovkit-0.1.2/src/ovkit/solutions/anomaly.py +0 -19
- ovkit-0.1.2/src/ovkit/solutions/ocr.py +0 -24
- ovkit-0.1.2/src/ovkit/solutions/reid.py +0 -26
- ovkit-0.1.2/src/ovkit/solutions/tracking.py +0 -23
- {ovkit-0.1.2 → ovkit-0.2.0}/.gitignore +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/LICENSE +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/__init__.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/convert.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/errors.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/tasks.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/image/__init__.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/image/ops.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/manifests/models.yaml +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/py.typed +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/__init__.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/face.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/ocr.py +0 -0
- {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/pose.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: ovkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A simple Python inference API for OpenVINO: one import, one Model class, clean Results — with AUTO/NPU, async, and INT8.
|
|
5
5
|
Project-URL: Homepage, https://github.com/leeyunjai82/ovkit
|
|
6
6
|
Project-URL: Documentation, https://leeyunjai82.github.io/ovkit/
|
|
@@ -69,14 +69,36 @@ ready models — with `AUTO`/`NPU`/`GPU` devices, async throughput, and INT8.
|
|
|
69
69
|
```python
|
|
70
70
|
from ovkit import Model
|
|
71
71
|
|
|
72
|
-
r = Model("detect"
|
|
73
|
-
r
|
|
72
|
+
r = Model("detect", "image.jpg") # download -> convert -> cache -> run
|
|
73
|
+
print(r) # 2x person, car
|
|
74
|
+
r.save("out.jpg") # the boxes drawn on the photo
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
One call fits every input — and Korean names work too:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
r = Model("얼굴분석", "group.jpg") # == Model("face_analyze", ...)
|
|
81
|
+
for row in r.found: # [{'name': '사람', 'name_en': 'person',
|
|
82
|
+
print(row["name"], row["pos"]) # 'score': 0.93, 'box': [...], 'pos': '왼쪽 위'}]
|
|
83
|
+
|
|
84
|
+
for r in Model("track", 0): # webcam / video / "mic" -> a stream
|
|
85
|
+
print(r, r.elapsed_ms, r.device) # 2x person (#1, #4) 14.2 GPU
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
The same call also runs whole **capabilities** — several models chained into one
|
|
89
|
+
answer, so you never have to crop, chain and stitch by hand:
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
r = Model("face_analyze", "group.jpg")
|
|
93
|
+
print(r) # 2 faces: age 31 · male 98% · happy 92%, age 27 · female 95% · neutral 88%
|
|
74
94
|
```
|
|
75
95
|
|
|
76
96
|
Or without writing Python at all:
|
|
77
97
|
|
|
78
98
|
```bash
|
|
99
|
+
ovkit gui # a window: pick a capability, point it at your webcam
|
|
79
100
|
ovkit run detect image.jpg # prints results, saves image_out.jpg
|
|
101
|
+
ovkit capabilities # what Model(name) can answer
|
|
80
102
|
```
|
|
81
103
|
|
|
82
104
|
## Install
|
|
@@ -94,9 +116,45 @@ python -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\act
|
|
|
94
116
|
pip install -e ".[dev]"
|
|
95
117
|
```
|
|
96
118
|
|
|
119
|
+
## Capabilities
|
|
120
|
+
|
|
121
|
+
A capability name gives you the answer, not the plumbing. Each one chains a
|
|
122
|
+
detector with the models that describe what it found — same `Model(...)` call,
|
|
123
|
+
same `Results`, and it takes the same sources (path, ndarray, folder, video,
|
|
124
|
+
camera index).
|
|
125
|
+
|
|
126
|
+
| `Model(...)` | Answers | Chains |
|
|
127
|
+
| ------------ | ------- | ------ |
|
|
128
|
+
| `scene` | `2 people (1 happy) · a laptop and a cup · floor 47%` | detection + segmentation + faces |
|
|
129
|
+
| `face_analyze` | `2 faces: age 31 · male 98% · happy 92%, ...` | face detection + age/gender + emotion (+ head pose, landmarks) |
|
|
130
|
+
| `person_analyze` | `3 people: male 0.98 · long pants 0.95 · bag 0.71, ...` | person detection + attributes |
|
|
131
|
+
| `vehicle_analyze` | `2 vehicles: type: car (0.98) · color: black (0.83), ...` | vehicle detection + type/colour |
|
|
132
|
+
| `read_text` | `'STOP AHEAD'` — every word, in reading order | text detection + text recognition |
|
|
133
|
+
| `read_plate` | `2 vehicles: black car — 12GA3456, ...` | plate detection + text recognition + vehicle attributes |
|
|
134
|
+
| `track` | `2x person (#1, #4)` — ids stable across frames | detection + IoU association |
|
|
135
|
+
| `drowsiness` | `EYES CLOSED 1.4s — drowsy` | face + landmarks + eye state + head pose, **over time** |
|
|
136
|
+
| `gesture` | `thumb up 0.94` | sign-language model over a rolling 8-frame clip |
|
|
137
|
+
| `gaze` | `1 face: looking right and slightly up` | face detection + landmarks + head pose + gaze |
|
|
138
|
+
| `attention` | `1 person looking at: laptop` | gaze + object detection (ray-cast into the boxes) |
|
|
139
|
+
| `anonymize` | the picture with every face pixelated | face (and plate) detection + redaction |
|
|
140
|
+
| `face_match` | `('yunjai', 0.81)` — who this is | embedding + cosine matching against your gallery |
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
from ovkit import Model, list_pipelines
|
|
144
|
+
|
|
145
|
+
list_pipelines() # every capability, described
|
|
146
|
+
Model("read_text")("sign.jpg")[0].text # 'STOP AHEAD'
|
|
147
|
+
Model("track")(0) # webcam, ids kept across frames
|
|
148
|
+
Model("face_analyze", attributes=("age_gender",)) # configure what runs
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Aliases: `ocr`, `anpr`, `blur`, `driver`, `describe`, `faces`, `people`, `vehicle`, `tracking`, `reid`.
|
|
152
|
+
`ovkit capabilities` prints the list; **`ovkit gui` opens a window** where you can
|
|
153
|
+
click through them against your webcam or a picture.
|
|
154
|
+
|
|
97
155
|
## Supported tasks
|
|
98
156
|
|
|
99
|
-
Every task below runs end-to-end through the same 3 lines — swap the alias:
|
|
157
|
+
Every single-model task below runs end-to-end through the same 3 lines — swap the alias:
|
|
100
158
|
|
|
101
159
|
| Task | Alias | Output | Example |
|
|
102
160
|
| ---- | ----- | ------ | ------- |
|
|
@@ -106,6 +164,7 @@ Every task below runs end-to-end through the same 3 lines — swap the alias:
|
|
|
106
164
|
| Pose / landmarks | `pose`, `face_landmarks` | `r.keypoints` | [pose.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/pose.py) |
|
|
107
165
|
| Face analysis | `age_gender`, `emotion`, `head_pose`, `face_reid` | `r.text` (e.g. `"age 31 · male 98%"`) | [face_analysis.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_analysis.py) |
|
|
108
166
|
| OCR | `text_recognition` | `r.text` | [ocr.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/ocr.py) |
|
|
167
|
+
| Tracking / matching | `track`, `face_match` | `r.track_ids` · `(label, score)` | [track.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/track.py) / [face_match.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_match.py) |
|
|
109
168
|
| Super-resolution | `super_resolution` | upscaled image via `r.plot()` | [super_resolution.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/super_resolution.py) |
|
|
110
169
|
| LLM / STT (GenAI) | `llm`, `stt` | generated text | [llm.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/llm.py) / [stt.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/stt.py) |
|
|
111
170
|
| NLP / audio / time series | `qa`, `translation`, `noise_suppression`, `time_series` | tensors via `model.infer()` | [denoise_audio.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/denoise_audio.py) |
|
|
@@ -174,7 +233,10 @@ control for any model: `model.infer({name: tensor})` with `model.inputs`.
|
|
|
174
233
|
| `r.boxes` | `xyxy`, `xywh`, `conf`, `cls` |
|
|
175
234
|
| `r.masks` / `r.keypoints` / `r.probs` | masks · `[x,y,conf]` · `top1`/`top5` |
|
|
176
235
|
| `r.text` | decoded text (OCR, face attributes) |
|
|
236
|
+
| `r.labels` / `r.track_ids` | per-box label a pipeline wrote · per-box track id |
|
|
177
237
|
| `r.tensors` | raw `{name: ndarray}` |
|
|
238
|
+
| `r.summary()` | the whole result as one readable line |
|
|
239
|
+
| `r.crop(i)` / `r.to_dict()` / `r.to_json()` | cut a box out · plain Python · JSON |
|
|
178
240
|
| `r.plot()` / `r.save(path)` | annotated image (or the model's output image) |
|
|
179
241
|
|
|
180
242
|
</details>
|
|
@@ -24,14 +24,36 @@ ready models — with `AUTO`/`NPU`/`GPU` devices, async throughput, and INT8.
|
|
|
24
24
|
```python
|
|
25
25
|
from ovkit import Model
|
|
26
26
|
|
|
27
|
-
r = Model("detect"
|
|
28
|
-
r
|
|
27
|
+
r = Model("detect", "image.jpg") # download -> convert -> cache -> run
|
|
28
|
+
print(r) # 2x person, car
|
|
29
|
+
r.save("out.jpg") # the boxes drawn on the photo
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
One call fits every input — and Korean names work too:
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
r = Model("얼굴분석", "group.jpg") # == Model("face_analyze", ...)
|
|
36
|
+
for row in r.found: # [{'name': '사람', 'name_en': 'person',
|
|
37
|
+
print(row["name"], row["pos"]) # 'score': 0.93, 'box': [...], 'pos': '왼쪽 위'}]
|
|
38
|
+
|
|
39
|
+
for r in Model("track", 0): # webcam / video / "mic" -> a stream
|
|
40
|
+
print(r, r.elapsed_ms, r.device) # 2x person (#1, #4) 14.2 GPU
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
The same call also runs whole **capabilities** — several models chained into one
|
|
44
|
+
answer, so you never have to crop, chain and stitch by hand:
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
r = Model("face_analyze", "group.jpg")
|
|
48
|
+
print(r) # 2 faces: age 31 · male 98% · happy 92%, age 27 · female 95% · neutral 88%
|
|
29
49
|
```
|
|
30
50
|
|
|
31
51
|
Or without writing Python at all:
|
|
32
52
|
|
|
33
53
|
```bash
|
|
54
|
+
ovkit gui # a window: pick a capability, point it at your webcam
|
|
34
55
|
ovkit run detect image.jpg # prints results, saves image_out.jpg
|
|
56
|
+
ovkit capabilities # what Model(name) can answer
|
|
35
57
|
```
|
|
36
58
|
|
|
37
59
|
## Install
|
|
@@ -49,9 +71,45 @@ python -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\act
|
|
|
49
71
|
pip install -e ".[dev]"
|
|
50
72
|
```
|
|
51
73
|
|
|
74
|
+
## Capabilities
|
|
75
|
+
|
|
76
|
+
A capability name gives you the answer, not the plumbing. Each one chains a
|
|
77
|
+
detector with the models that describe what it found — same `Model(...)` call,
|
|
78
|
+
same `Results`, and it takes the same sources (path, ndarray, folder, video,
|
|
79
|
+
camera index).
|
|
80
|
+
|
|
81
|
+
| `Model(...)` | Answers | Chains |
|
|
82
|
+
| ------------ | ------- | ------ |
|
|
83
|
+
| `scene` | `2 people (1 happy) · a laptop and a cup · floor 47%` | detection + segmentation + faces |
|
|
84
|
+
| `face_analyze` | `2 faces: age 31 · male 98% · happy 92%, ...` | face detection + age/gender + emotion (+ head pose, landmarks) |
|
|
85
|
+
| `person_analyze` | `3 people: male 0.98 · long pants 0.95 · bag 0.71, ...` | person detection + attributes |
|
|
86
|
+
| `vehicle_analyze` | `2 vehicles: type: car (0.98) · color: black (0.83), ...` | vehicle detection + type/colour |
|
|
87
|
+
| `read_text` | `'STOP AHEAD'` — every word, in reading order | text detection + text recognition |
|
|
88
|
+
| `read_plate` | `2 vehicles: black car — 12GA3456, ...` | plate detection + text recognition + vehicle attributes |
|
|
89
|
+
| `track` | `2x person (#1, #4)` — ids stable across frames | detection + IoU association |
|
|
90
|
+
| `drowsiness` | `EYES CLOSED 1.4s — drowsy` | face + landmarks + eye state + head pose, **over time** |
|
|
91
|
+
| `gesture` | `thumb up 0.94` | sign-language model over a rolling 8-frame clip |
|
|
92
|
+
| `gaze` | `1 face: looking right and slightly up` | face detection + landmarks + head pose + gaze |
|
|
93
|
+
| `attention` | `1 person looking at: laptop` | gaze + object detection (ray-cast into the boxes) |
|
|
94
|
+
| `anonymize` | the picture with every face pixelated | face (and plate) detection + redaction |
|
|
95
|
+
| `face_match` | `('yunjai', 0.81)` — who this is | embedding + cosine matching against your gallery |
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from ovkit import Model, list_pipelines
|
|
99
|
+
|
|
100
|
+
list_pipelines() # every capability, described
|
|
101
|
+
Model("read_text")("sign.jpg")[0].text # 'STOP AHEAD'
|
|
102
|
+
Model("track")(0) # webcam, ids kept across frames
|
|
103
|
+
Model("face_analyze", attributes=("age_gender",)) # configure what runs
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Aliases: `ocr`, `anpr`, `blur`, `driver`, `describe`, `faces`, `people`, `vehicle`, `tracking`, `reid`.
|
|
107
|
+
`ovkit capabilities` prints the list; **`ovkit gui` opens a window** where you can
|
|
108
|
+
click through them against your webcam or a picture.
|
|
109
|
+
|
|
52
110
|
## Supported tasks
|
|
53
111
|
|
|
54
|
-
Every task below runs end-to-end through the same 3 lines — swap the alias:
|
|
112
|
+
Every single-model task below runs end-to-end through the same 3 lines — swap the alias:
|
|
55
113
|
|
|
56
114
|
| Task | Alias | Output | Example |
|
|
57
115
|
| ---- | ----- | ------ | ------- |
|
|
@@ -61,6 +119,7 @@ Every task below runs end-to-end through the same 3 lines — swap the alias:
|
|
|
61
119
|
| Pose / landmarks | `pose`, `face_landmarks` | `r.keypoints` | [pose.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/pose.py) |
|
|
62
120
|
| Face analysis | `age_gender`, `emotion`, `head_pose`, `face_reid` | `r.text` (e.g. `"age 31 · male 98%"`) | [face_analysis.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_analysis.py) |
|
|
63
121
|
| OCR | `text_recognition` | `r.text` | [ocr.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/ocr.py) |
|
|
122
|
+
| Tracking / matching | `track`, `face_match` | `r.track_ids` · `(label, score)` | [track.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/track.py) / [face_match.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_match.py) |
|
|
64
123
|
| Super-resolution | `super_resolution` | upscaled image via `r.plot()` | [super_resolution.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/super_resolution.py) |
|
|
65
124
|
| LLM / STT (GenAI) | `llm`, `stt` | generated text | [llm.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/llm.py) / [stt.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/stt.py) |
|
|
66
125
|
| NLP / audio / time series | `qa`, `translation`, `noise_suppression`, `time_series` | tensors via `model.infer()` | [denoise_audio.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/denoise_audio.py) |
|
|
@@ -129,7 +188,10 @@ control for any model: `model.infer({name: tensor})` with `model.inputs`.
|
|
|
129
188
|
| `r.boxes` | `xyxy`, `xywh`, `conf`, `cls` |
|
|
130
189
|
| `r.masks` / `r.keypoints` / `r.probs` | masks · `[x,y,conf]` · `top1`/`top5` |
|
|
131
190
|
| `r.text` | decoded text (OCR, face attributes) |
|
|
191
|
+
| `r.labels` / `r.track_ids` | per-box label a pipeline wrote · per-box track id |
|
|
132
192
|
| `r.tensors` | raw `{name: ndarray}` |
|
|
193
|
+
| `r.summary()` | the whole result as one readable line |
|
|
194
|
+
| `r.crop(i)` / `r.to_dict()` / `r.to_json()` | cut a box out · plain Python · JSON |
|
|
133
195
|
| `r.plot()` / `r.save(path)` | annotated image (or the model's output image) |
|
|
134
196
|
|
|
135
197
|
</details>
|
|
@@ -16,6 +16,7 @@ pip install -r examples/requirements.txt # fastapi / uvicorn (web demos only)
|
|
|
16
16
|
| [`classify.py`](classify.py) | `classify` | top-5 classes |
|
|
17
17
|
| [`ocr.py`](ocr.py) | `text_recognition` | decoded text |
|
|
18
18
|
| [`super_resolution.py`](super_resolution.py) | `super_resolution` | upscaled image |
|
|
19
|
+
| [`background_matting.py`](background_matting.py) | `background_matting_mobilenetv2` | cut the subject out (needs frame + empty-scene photo) |
|
|
19
20
|
|
|
20
21
|
```bash
|
|
21
22
|
python examples/detect.py photo.jpg # each is ~15 lines — read the source
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "ovkit"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "A simple Python inference API for OpenVINO: one import, one Model class, clean Results — with AUTO/NPU, async, and INT8."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -3,14 +3,23 @@
|
|
|
3
3
|
One import, one :class:`Model` class, a callable object, and clean
|
|
4
4
|
:class:`Results` — plus OpenVINO's strengths (AUTO/NPU devices, async, INT8).
|
|
5
5
|
|
|
6
|
-
|
|
7
|
-
|
|
6
|
+
One model
|
|
7
|
+
---------
|
|
8
8
|
>>> from ovkit import Model
|
|
9
9
|
>>> model = Model("rtdetr_r50") # name -> auto download/convert/cache
|
|
10
10
|
>>> results = model("img.jpg", conf=0.25)
|
|
11
11
|
>>> for r in results:
|
|
12
|
-
... print(r.
|
|
12
|
+
... print(r.summary()) # '2x person, car'
|
|
13
13
|
... r.save("out.jpg")
|
|
14
|
+
|
|
15
|
+
Several models, one answer
|
|
16
|
+
--------------------------
|
|
17
|
+
>>> from ovkit import Model
|
|
18
|
+
>>> for r in Model("face_analyze")("group.jpg"):
|
|
19
|
+
... print(r.summary()) # '2 faces: age 31 · male 98% · happy 92%, ...'
|
|
20
|
+
|
|
21
|
+
A capability name composes the models the answer needs — detection plus
|
|
22
|
+
whatever describes what was found. :func:`list_pipelines` shows them all.
|
|
14
23
|
"""
|
|
15
24
|
|
|
16
25
|
from __future__ import annotations
|
|
@@ -29,11 +38,14 @@ from .core.errors import (
|
|
|
29
38
|
from .core.model import Model
|
|
30
39
|
from .core.registry import list_models
|
|
31
40
|
from .core.results import Boxes, Keypoints, Masks, Probs, Results
|
|
41
|
+
from .pipelines import Pipeline, list_pipelines
|
|
32
42
|
|
|
33
|
-
__version__ = "0.
|
|
43
|
+
__version__ = "0.2.0"
|
|
34
44
|
|
|
35
45
|
__all__ = [
|
|
36
46
|
"Model",
|
|
47
|
+
"Pipeline",
|
|
48
|
+
"list_pipelines",
|
|
37
49
|
"Results",
|
|
38
50
|
"Boxes",
|
|
39
51
|
"Masks",
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""``ovkit`` command-line interface: ``run``, ``list``, ``info``, ``download``, ``devices``."""
|
|
1
|
+
"""``ovkit`` command-line interface: ``gui``, ``run``, ``list``, ``info``, ``download``, ``devices``."""
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
@@ -41,6 +41,19 @@ def _cmd_list(_: argparse.Namespace) -> int:
|
|
|
41
41
|
return 0
|
|
42
42
|
|
|
43
43
|
|
|
44
|
+
def _cmd_capabilities(_: argparse.Namespace) -> int:
|
|
45
|
+
"""What ``Model(name)`` can answer beyond a single network."""
|
|
46
|
+
from .pipelines import ALIASES, list_pipelines
|
|
47
|
+
|
|
48
|
+
print("capabilities — Model(name) chains the models the answer needs:\n")
|
|
49
|
+
for name, description in list_pipelines().items():
|
|
50
|
+
print(f" {name:16s} {description}")
|
|
51
|
+
print("\naliases:")
|
|
52
|
+
for alias, target in sorted(ALIASES.items()):
|
|
53
|
+
print(f" {alias:16s} -> {target}")
|
|
54
|
+
return 0
|
|
55
|
+
|
|
56
|
+
|
|
44
57
|
def _cmd_info(args: argparse.Namespace) -> int:
|
|
45
58
|
entry = resolve(args.name)
|
|
46
59
|
if entry is None:
|
|
@@ -96,25 +109,12 @@ def _cmd_run(args: argparse.Namespace) -> int:
|
|
|
96
109
|
return 0
|
|
97
110
|
|
|
98
111
|
for r in results:
|
|
99
|
-
|
|
100
|
-
if r.text:
|
|
101
|
-
parts.append(f'text="{r.text}"')
|
|
112
|
+
print(r.summary())
|
|
102
113
|
if r.boxes is not None:
|
|
103
|
-
parts.append(f"{len(r.boxes)} boxes")
|
|
104
114
|
for x1, y1, x2, y2, c, cl in r.boxes.data[:20]:
|
|
105
115
|
print(
|
|
106
116
|
f" {r.name_for(int(cl)):16s} {c:.2f} [{int(x1)},{int(y1)},{int(x2)},{int(y2)}]"
|
|
107
117
|
)
|
|
108
|
-
if r.probs is not None:
|
|
109
|
-
top = ", ".join(
|
|
110
|
-
f"{r.name_for(int(i))} {r.probs.data[int(i)]:.2f}" for i in r.probs.top5
|
|
111
|
-
)
|
|
112
|
-
parts.append(f"top-5: {top}")
|
|
113
|
-
if r.masks is not None:
|
|
114
|
-
parts.append(f"masks {tuple(r.masks.data.shape)}")
|
|
115
|
-
if r.keypoints is not None:
|
|
116
|
-
parts.append(f"keypoints {tuple(r.keypoints.data.shape)}")
|
|
117
|
-
print(" | ".join(parts))
|
|
118
118
|
|
|
119
119
|
save = args.save
|
|
120
120
|
if save is None and results and Path(str(args.source)).is_file():
|
|
@@ -125,15 +125,30 @@ def _cmd_run(args: argparse.Namespace) -> int:
|
|
|
125
125
|
return 0
|
|
126
126
|
|
|
127
127
|
|
|
128
|
+
def _cmd_gui(args: argparse.Namespace) -> int:
|
|
129
|
+
"""Open the desktop window: ``ovkit gui``."""
|
|
130
|
+
from .gui import main as gui_main
|
|
131
|
+
|
|
132
|
+
return gui_main(device=args.device, camera=args.camera)
|
|
133
|
+
|
|
134
|
+
|
|
128
135
|
def main(argv: list[str] | None = None) -> int:
|
|
129
136
|
"""CLI entry point. Returns a process exit code."""
|
|
130
137
|
parser = argparse.ArgumentParser(prog="ovkit", description="ovkit model utilities")
|
|
131
138
|
parser.add_argument("--version", action="version", version=f"ovkit {__version__}")
|
|
132
139
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
133
140
|
|
|
141
|
+
p_gui = sub.add_parser("gui", help="open the desktop window (easiest way to try ovkit)")
|
|
142
|
+
p_gui.add_argument("--device", default="AUTO", help="AUTO | CPU | GPU | NPU")
|
|
143
|
+
p_gui.add_argument("--camera", type=int, default=0, help="camera index for the webcam button")
|
|
144
|
+
p_gui.set_defaults(func=_cmd_gui)
|
|
145
|
+
|
|
134
146
|
p_list = sub.add_parser("list", help="list registered models")
|
|
135
147
|
p_list.set_defaults(func=_cmd_list)
|
|
136
148
|
|
|
149
|
+
p_caps = sub.add_parser("capabilities", help="list composed capabilities (Model(name))")
|
|
150
|
+
p_caps.set_defaults(func=_cmd_capabilities)
|
|
151
|
+
|
|
137
152
|
p_info = sub.add_parser("info", help="show details for a model")
|
|
138
153
|
p_info.add_argument("name")
|
|
139
154
|
p_info.set_defaults(func=_cmd_info)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Audio I/O helpers: read a file into the array a model actually wants.
|
|
2
|
+
|
|
3
|
+
Models want float32 mono at a fixed sample rate; files are int16 stereo at
|
|
4
|
+
whatever rate the recorder used. These helpers close that gap so callers never
|
|
5
|
+
have to touch :mod:`wave` or write their own resampler.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from .ops import read_audio, read_wav, resample, to_mono, write_wav
|
|
11
|
+
from .plot import waveform
|
|
12
|
+
|
|
13
|
+
__all__ = ["read_audio", "read_wav", "resample", "to_mono", "waveform", "write_wav"]
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Read, convert and write audio as plain float32 numpy arrays."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import io
|
|
6
|
+
import wave
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import numpy as np
|
|
11
|
+
|
|
12
|
+
from ..core.errors import OVKitError
|
|
13
|
+
|
|
14
|
+
#: Sample formats ``wave`` reports, mapped to (numpy dtype, full-scale value).
|
|
15
|
+
_PCM = {1: (np.int8, 128.0), 2: (np.int16, 32768.0), 4: (np.int32, 2147483648.0)}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def to_mono(audio: np.ndarray) -> np.ndarray:
|
|
19
|
+
"""Average an ``(N, channels)`` array down to a 1-D mono signal."""
|
|
20
|
+
arr = np.asarray(audio, dtype=np.float32)
|
|
21
|
+
return arr.mean(axis=1) if arr.ndim == 2 and arr.shape[1] > 1 else arr.reshape(-1)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def resample(audio: np.ndarray, sr: int, target_sr: int) -> np.ndarray:
|
|
25
|
+
"""Resample a 1-D signal to ``target_sr`` by linear interpolation.
|
|
26
|
+
|
|
27
|
+
Linear interpolation is not a studio-grade resampler, but it is exact when
|
|
28
|
+
the rates match, needs no extra dependency, and is well within the tolerance
|
|
29
|
+
of the classifiers and ASR models ovkit serves.
|
|
30
|
+
"""
|
|
31
|
+
arr = np.asarray(audio, dtype=np.float32).reshape(-1)
|
|
32
|
+
if sr == target_sr or arr.size == 0:
|
|
33
|
+
return arr
|
|
34
|
+
n_out = int(round(arr.size * target_sr / float(sr)))
|
|
35
|
+
if n_out <= 1:
|
|
36
|
+
return arr[:1].copy()
|
|
37
|
+
src = np.linspace(0.0, arr.size - 1, num=n_out, dtype=np.float32)
|
|
38
|
+
return np.interp(src, np.arange(arr.size, dtype=np.float32), arr).astype(np.float32)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def read_audio(path: str | Path, target_sr: int | None = None) -> tuple[np.ndarray, int]:
|
|
42
|
+
"""Read an audio file as ``(float32 mono in [-1, 1], sample_rate)``.
|
|
43
|
+
|
|
44
|
+
``.wav`` is read with the standard library. Other formats are read through
|
|
45
|
+
``soundfile`` when it is installed; otherwise the error says what to do
|
|
46
|
+
rather than failing deep inside a decoder. ``target_sr`` resamples the
|
|
47
|
+
result (and is what most models need, e.g. 16 kHz for speech).
|
|
48
|
+
"""
|
|
49
|
+
src = Path(path)
|
|
50
|
+
if src.suffix.lower() == ".wav":
|
|
51
|
+
audio, sr = read_wav(src)
|
|
52
|
+
else:
|
|
53
|
+
audio, sr = _read_soundfile(src)
|
|
54
|
+
if target_sr:
|
|
55
|
+
audio = resample(audio, sr, target_sr)
|
|
56
|
+
sr = target_sr
|
|
57
|
+
return audio, sr
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def read_wav(source: Any, target_sr: int | None = None) -> tuple[np.ndarray, int]:
|
|
61
|
+
"""Read WAV from a path or an open/byte stream — any depth, any channels.
|
|
62
|
+
|
|
63
|
+
Uploads arrive as bytes and recordings arrive as stereo; both go through
|
|
64
|
+
here so callers never re-implement PCM decoding.
|
|
65
|
+
"""
|
|
66
|
+
if isinstance(source, bytes):
|
|
67
|
+
source = io.BytesIO(source)
|
|
68
|
+
with wave.open(source if hasattr(source, "read") else str(source), "rb") as wf:
|
|
69
|
+
sr, width, channels = wf.getframerate(), wf.getsampwidth(), wf.getnchannels()
|
|
70
|
+
raw = wf.readframes(wf.getnframes())
|
|
71
|
+
if width not in _PCM:
|
|
72
|
+
raise OVKitError(
|
|
73
|
+
f"{width * 8}-bit WAV is not supported (8/16/32-bit PCM is). "
|
|
74
|
+
f"Re-encode it, e.g. ffmpeg -i in.wav -acodec pcm_s16le out.wav"
|
|
75
|
+
)
|
|
76
|
+
dtype, full_scale = _PCM[width]
|
|
77
|
+
samples = np.frombuffer(raw, dtype=dtype).astype(np.float32) / full_scale
|
|
78
|
+
if channels > 1:
|
|
79
|
+
samples = samples[: samples.size // channels * channels].reshape(-1, channels)
|
|
80
|
+
audio = to_mono(samples)
|
|
81
|
+
if target_sr:
|
|
82
|
+
return resample(audio, sr, target_sr), target_sr
|
|
83
|
+
return audio, sr
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _read_soundfile(path: Path) -> tuple[np.ndarray, int]:
|
|
87
|
+
try:
|
|
88
|
+
import soundfile as sf
|
|
89
|
+
except ImportError as exc:
|
|
90
|
+
raise OVKitError(
|
|
91
|
+
f"Reading '{path.suffix}' needs an extra decoder. Either convert to WAV "
|
|
92
|
+
f"(ffmpeg -i {path.name} -ar 16000 -ac 1 out.wav) or install one: "
|
|
93
|
+
f"pip install soundfile"
|
|
94
|
+
) from exc
|
|
95
|
+
audio, sr = sf.read(str(path), dtype="float32", always_2d=True)
|
|
96
|
+
return to_mono(audio), int(sr)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def write_wav(path: str | Path, audio: np.ndarray, sr: int) -> Path:
|
|
100
|
+
"""Write a float32 mono signal to a 16-bit PCM ``.wav`` and return the path."""
|
|
101
|
+
arr = np.clip(np.asarray(audio, dtype=np.float32).reshape(-1), -1.0, 1.0)
|
|
102
|
+
pcm = (arr * 32767.0).astype(np.int16)
|
|
103
|
+
with wave.open(str(path), "wb") as wf:
|
|
104
|
+
wf.setnchannels(1)
|
|
105
|
+
wf.setsampwidth(2)
|
|
106
|
+
wf.setframerate(int(sr))
|
|
107
|
+
wf.writeframes(pcm.tobytes())
|
|
108
|
+
return Path(path)
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Draw a waveform, so an audio result has something to show."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def waveform(
|
|
9
|
+
audio: np.ndarray,
|
|
10
|
+
sr: int = 16_000,
|
|
11
|
+
width: int = 720,
|
|
12
|
+
height: int = 200,
|
|
13
|
+
color: tuple[int, int, int] = (200, 190, 120),
|
|
14
|
+
) -> np.ndarray:
|
|
15
|
+
"""Render a mono signal as a BGR image.
|
|
16
|
+
|
|
17
|
+
Audio results carry this as their image so ``plot()``, ``save()`` and the
|
|
18
|
+
web demo work the same way they do for a picture.
|
|
19
|
+
"""
|
|
20
|
+
import cv2
|
|
21
|
+
|
|
22
|
+
img = np.full((height, width, 3), 24, np.uint8)
|
|
23
|
+
samples = np.asarray(audio, np.float32).reshape(-1)
|
|
24
|
+
mid = height // 2
|
|
25
|
+
cv2.line(img, (0, mid), (width, mid), (60, 60, 60), 1)
|
|
26
|
+
if samples.size:
|
|
27
|
+
# One vertical bar per column: the min/max of the samples it covers.
|
|
28
|
+
edges = np.linspace(0, samples.size, width + 1).astype(int)
|
|
29
|
+
peak = max(float(np.abs(samples).max()), 1e-6)
|
|
30
|
+
for x in range(width):
|
|
31
|
+
chunk = samples[edges[x] : max(edges[x + 1], edges[x] + 1)]
|
|
32
|
+
if not chunk.size:
|
|
33
|
+
continue
|
|
34
|
+
lo, hi = float(chunk.min()) / peak, float(chunk.max()) / peak
|
|
35
|
+
cv2.line(img, (x, mid - int(hi * mid * 0.9)), (x, mid - int(lo * mid * 0.9)), color, 1)
|
|
36
|
+
seconds = samples.size / float(sr or 1)
|
|
37
|
+
cv2.putText(
|
|
38
|
+
img,
|
|
39
|
+
f"{seconds:.1f}s @ {sr} Hz",
|
|
40
|
+
(8, height - 8),
|
|
41
|
+
cv2.FONT_HERSHEY_SIMPLEX,
|
|
42
|
+
0.45,
|
|
43
|
+
(140, 140, 140),
|
|
44
|
+
1,
|
|
45
|
+
cv2.LINE_AA,
|
|
46
|
+
)
|
|
47
|
+
return img
|
|
@@ -32,6 +32,39 @@ def available_devices() -> list[str]:
|
|
|
32
32
|
return list(core().available_devices)
|
|
33
33
|
|
|
34
34
|
|
|
35
|
+
def _friendlier_compile_error(exc: Exception, model: str | Path | Any) -> Exception:
|
|
36
|
+
"""Turn OpenVINO's nested compile errors into something actionable.
|
|
37
|
+
|
|
38
|
+
The most common one by far is IR whose ``.bin`` never arrived (a mirror
|
|
39
|
+
upload that dropped the weights, or an interrupted download). OpenVINO
|
|
40
|
+
reports it as "Empty weights data in bin file", buried under two layers of
|
|
41
|
+
"Exception from ...", which tells a user nothing about what to do.
|
|
42
|
+
"""
|
|
43
|
+
from .errors import OVKitError
|
|
44
|
+
|
|
45
|
+
text = str(exc)
|
|
46
|
+
if "Empty weights data" not in text and "bin file" not in text:
|
|
47
|
+
return exc
|
|
48
|
+
if not isinstance(model, (str, Path)):
|
|
49
|
+
return exc
|
|
50
|
+
xml = Path(str(model))
|
|
51
|
+
if xml.suffix != ".xml":
|
|
52
|
+
return exc
|
|
53
|
+
bin_path = xml.with_suffix(".bin")
|
|
54
|
+
if not bin_path.exists():
|
|
55
|
+
state = "is missing"
|
|
56
|
+
elif bin_path.stat().st_size < 1024:
|
|
57
|
+
state = f"is only {bin_path.stat().st_size} bytes"
|
|
58
|
+
else:
|
|
59
|
+
return exc # weights look fine — a different problem, keep the original
|
|
60
|
+
return OVKitError(
|
|
61
|
+
f"The weights file for this model {state}: {bin_path}\n"
|
|
62
|
+
f"An OpenVINO IR needs both model.xml and model.bin. Delete the cached "
|
|
63
|
+
f"copy and download it again; if it keeps happening, the mirrored model "
|
|
64
|
+
f"itself is incomplete and needs re-uploading."
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
35
68
|
class Backend:
|
|
36
69
|
"""A compiled model bound to a device, with sync and async inference.
|
|
37
70
|
|
|
@@ -47,7 +80,10 @@ class Backend:
|
|
|
47
80
|
self.device = device
|
|
48
81
|
c = core()
|
|
49
82
|
src = str(model) if isinstance(model, (str, Path)) else model
|
|
50
|
-
|
|
83
|
+
try:
|
|
84
|
+
self.compiled = c.compile_model(src, device)
|
|
85
|
+
except Exception as exc:
|
|
86
|
+
raise _friendlier_compile_error(exc, model) from exc
|
|
51
87
|
self.inputs = self.compiled.inputs
|
|
52
88
|
self.outputs = self.compiled.outputs
|
|
53
89
|
|
|
@@ -97,6 +133,15 @@ class Backend:
|
|
|
97
133
|
return np.repeat(arr, 3, axis=1)
|
|
98
134
|
return arr
|
|
99
135
|
|
|
136
|
+
@property
|
|
137
|
+
def actual_device(self) -> str:
|
|
138
|
+
"""The device inference actually runs on (AUTO resolves to a real one)."""
|
|
139
|
+
try:
|
|
140
|
+
devices = self.compiled.get_property("EXECUTION_DEVICES")
|
|
141
|
+
return ",".join(devices) if devices else self.device
|
|
142
|
+
except Exception:
|
|
143
|
+
return self.device
|
|
144
|
+
|
|
100
145
|
def output_signatures(self) -> list[tuple[str, tuple[int, ...]]]:
|
|
101
146
|
"""Return ``(name, shape)`` for each output (``-1`` for dynamic dims)."""
|
|
102
147
|
sigs: list[tuple[str, tuple[int, ...]]] = []
|