ovkit 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. {ovkit-0.1.2 → ovkit-0.2.0}/PKG-INFO +66 -4
  2. {ovkit-0.1.2 → ovkit-0.2.0}/README.md +65 -3
  3. {ovkit-0.1.2 → ovkit-0.2.0}/examples/README.md +1 -0
  4. {ovkit-0.1.2 → ovkit-0.2.0}/pyproject.toml +1 -1
  5. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/__init__.py +16 -4
  6. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/__main__.py +30 -15
  7. ovkit-0.2.0/src/ovkit/audio/__init__.py +13 -0
  8. ovkit-0.2.0/src/ovkit/audio/ops.py +108 -0
  9. ovkit-0.2.0/src/ovkit/audio/plot.py +47 -0
  10. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/backend.py +46 -1
  11. ovkit-0.2.0/src/ovkit/core/constants.py +335 -0
  12. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/download.py +24 -1
  13. ovkit-0.2.0/src/ovkit/core/i18n.py +309 -0
  14. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/model.py +205 -19
  15. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/registry.py +21 -1
  16. ovkit-0.2.0/src/ovkit/core/results.py +598 -0
  17. ovkit-0.2.0/src/ovkit/data/imagenet1000.txt +1000 -0
  18. ovkit-0.2.0/src/ovkit/face/__init__.py +8 -0
  19. ovkit-0.2.0/src/ovkit/face/analyzer.py +7 -0
  20. ovkit-0.2.0/src/ovkit/genai/__init__.py +40 -0
  21. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/genai/pipelines.py +105 -1
  22. ovkit-0.2.0/src/ovkit/gui/__init__.py +20 -0
  23. ovkit-0.2.0/src/ovkit/gui/app.py +286 -0
  24. ovkit-0.2.0/src/ovkit/gui/controller.py +291 -0
  25. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/manifests/aliases.yaml +11 -4
  26. ovkit-0.2.0/src/ovkit/manifests/genai.yaml +156 -0
  27. ovkit-0.2.0/src/ovkit/manifests/labels.yaml +77 -0
  28. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/manifests/omz.yaml +215 -73
  29. ovkit-0.2.0/src/ovkit/pipelines/__init__.py +165 -0
  30. ovkit-0.2.0/src/ovkit/pipelines/analyze.py +165 -0
  31. ovkit-0.2.0/src/ovkit/pipelines/attention.py +114 -0
  32. ovkit-0.2.0/src/ovkit/pipelines/base.py +97 -0
  33. ovkit-0.2.0/src/ovkit/pipelines/gaze.py +227 -0
  34. ovkit-0.2.0/src/ovkit/pipelines/plates.py +137 -0
  35. ovkit-0.2.0/src/ovkit/pipelines/privacy.py +128 -0
  36. ovkit-0.2.0/src/ovkit/pipelines/reid.py +127 -0
  37. ovkit-0.2.0/src/ovkit/pipelines/scene.py +129 -0
  38. ovkit-0.2.0/src/ovkit/pipelines/temporal.py +226 -0
  39. ovkit-0.2.0/src/ovkit/pipelines/text.py +81 -0
  40. ovkit-0.2.0/src/ovkit/pipelines/tracking.py +120 -0
  41. ovkit-0.2.0/src/ovkit/recognize/audio.py +117 -0
  42. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/base.py +24 -3
  43. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/classify.py +57 -8
  44. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/detect.py +61 -1
  45. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/generic.py +30 -2
  46. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/segment.py +9 -3
  47. ovkit-0.2.0/src/ovkit/solutions/__init__.py +14 -0
  48. ovkit-0.2.0/src/ovkit/solutions/anomaly.py +93 -0
  49. ovkit-0.2.0/src/ovkit/solutions/ocr.py +7 -0
  50. ovkit-0.2.0/src/ovkit/solutions/reid.py +7 -0
  51. ovkit-0.2.0/src/ovkit/solutions/tracking.py +7 -0
  52. ovkit-0.1.2/src/ovkit/core/constants.py +0 -158
  53. ovkit-0.1.2/src/ovkit/core/results.py +0 -322
  54. ovkit-0.1.2/src/ovkit/face/__init__.py +0 -19
  55. ovkit-0.1.2/src/ovkit/face/analyzer.py +0 -26
  56. ovkit-0.1.2/src/ovkit/genai/__init__.py +0 -25
  57. ovkit-0.1.2/src/ovkit/manifests/genai.yaml +0 -38
  58. ovkit-0.1.2/src/ovkit/solutions/__init__.py +0 -9
  59. ovkit-0.1.2/src/ovkit/solutions/anomaly.py +0 -19
  60. ovkit-0.1.2/src/ovkit/solutions/ocr.py +0 -24
  61. ovkit-0.1.2/src/ovkit/solutions/reid.py +0 -26
  62. ovkit-0.1.2/src/ovkit/solutions/tracking.py +0 -23
  63. {ovkit-0.1.2 → ovkit-0.2.0}/.gitignore +0 -0
  64. {ovkit-0.1.2 → ovkit-0.2.0}/LICENSE +0 -0
  65. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/__init__.py +0 -0
  66. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/convert.py +0 -0
  67. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/errors.py +0 -0
  68. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/core/tasks.py +0 -0
  69. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/image/__init__.py +0 -0
  70. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/image/ops.py +0 -0
  71. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/manifests/models.yaml +0 -0
  72. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/py.typed +0 -0
  73. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/__init__.py +0 -0
  74. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/face.py +0 -0
  75. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/ocr.py +0 -0
  76. {ovkit-0.1.2 → ovkit-0.2.0}/src/ovkit/recognize/pose.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ovkit
3
- Version: 0.1.2
3
+ Version: 0.2.0
4
4
  Summary: A simple Python inference API for OpenVINO: one import, one Model class, clean Results — with AUTO/NPU, async, and INT8.
5
5
  Project-URL: Homepage, https://github.com/leeyunjai82/ovkit
6
6
  Project-URL: Documentation, https://leeyunjai82.github.io/ovkit/
@@ -69,14 +69,36 @@ ready models — with `AUTO`/`NPU`/`GPU` devices, async throughput, and INT8.
69
69
  ```python
70
70
  from ovkit import Model
71
71
 
72
- r = Model("detect")("image.jpg")[0] # download -> convert -> cache -> run
73
- r.save("out.jpg") # boxes drawn; r.boxes.xyxy / .conf / .cls
72
+ r = Model("detect", "image.jpg") # download -> convert -> cache -> run
73
+ print(r) # 2x person, car
74
+ r.save("out.jpg") # the boxes drawn on the photo
75
+ ```
76
+
77
+ One call fits every input — and Korean names work too:
78
+
79
+ ```python
80
+ r = Model("얼굴분석", "group.jpg") # == Model("face_analyze", ...)
81
+ for row in r.found: # [{'name': '사람', 'name_en': 'person',
82
+ print(row["name"], row["pos"]) # 'score': 0.93, 'box': [...], 'pos': '왼쪽 위'}]
83
+
84
+ for r in Model("track", 0): # webcam / video / "mic" -> a stream
85
+ print(r, r.elapsed_ms, r.device) # 2x person (#1, #4) 14.2 GPU
86
+ ```
87
+
88
+ The same call also runs whole **capabilities** — several models chained into one
89
+ answer, so you never have to crop, chain and stitch by hand:
90
+
91
+ ```python
92
+ r = Model("face_analyze", "group.jpg")
93
+ print(r) # 2 faces: age 31 · male 98% · happy 92%, age 27 · female 95% · neutral 88%
74
94
  ```
75
95
 
76
96
  Or without writing Python at all:
77
97
 
78
98
  ```bash
99
+ ovkit gui # a window: pick a capability, point it at your webcam
79
100
  ovkit run detect image.jpg # prints results, saves image_out.jpg
101
+ ovkit capabilities # what Model(name) can answer
80
102
  ```
81
103
 
82
104
  ## Install
@@ -94,9 +116,45 @@ python -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\act
94
116
  pip install -e ".[dev]"
95
117
  ```
96
118
 
119
+ ## Capabilities
120
+
121
+ A capability name gives you the answer, not the plumbing. Each one chains a
122
+ detector with the models that describe what it found — same `Model(...)` call,
123
+ same `Results`, and it takes the same sources (path, ndarray, folder, video,
124
+ camera index).
125
+
126
+ | `Model(...)` | Answers | Chains |
127
+ | ------------ | ------- | ------ |
128
+ | `scene` | `2 people (1 happy) · a laptop and a cup · floor 47%` | detection + segmentation + faces |
129
+ | `face_analyze` | `2 faces: age 31 · male 98% · happy 92%, ...` | face detection + age/gender + emotion (+ head pose, landmarks) |
130
+ | `person_analyze` | `3 people: male 0.98 · long pants 0.95 · bag 0.71, ...` | person detection + attributes |
131
+ | `vehicle_analyze` | `2 vehicles: type: car (0.98) · color: black (0.83), ...` | vehicle detection + type/colour |
132
+ | `read_text` | `'STOP AHEAD'` — every word, in reading order | text detection + text recognition |
133
+ | `read_plate` | `2 vehicles: black car — 12GA3456, ...` | plate detection + text recognition + vehicle attributes |
134
+ | `track` | `2x person (#1, #4)` — ids stable across frames | detection + IoU association |
135
+ | `drowsiness` | `EYES CLOSED 1.4s — drowsy` | face + landmarks + eye state + head pose, **over time** |
136
+ | `gesture` | `thumb up 0.94` | sign-language model over a rolling 8-frame clip |
137
+ | `gaze` | `1 face: looking right and slightly up` | face detection + landmarks + head pose + gaze |
138
+ | `attention` | `1 person looking at: laptop` | gaze + object detection (ray-cast into the boxes) |
139
+ | `anonymize` | the picture with every face pixelated | face (and plate) detection + redaction |
140
+ | `face_match` | `('yunjai', 0.81)` — who this is | embedding + cosine matching against your gallery |
141
+
142
+ ```python
143
+ from ovkit import Model, list_pipelines
144
+
145
+ list_pipelines() # every capability, described
146
+ Model("read_text")("sign.jpg")[0].text # 'STOP AHEAD'
147
+ Model("track")(0) # webcam, ids kept across frames
148
+ Model("face_analyze", attributes=("age_gender",)) # configure what runs
149
+ ```
150
+
151
+ Aliases: `ocr`, `anpr`, `blur`, `driver`, `describe`, `faces`, `people`, `vehicle`, `tracking`, `reid`.
152
+ `ovkit capabilities` prints the list; **`ovkit gui` opens a window** where you can
153
+ click through them against your webcam or a picture.
154
+
97
155
  ## Supported tasks
98
156
 
99
- Every task below runs end-to-end through the same 3 lines — swap the alias:
157
+ Every single-model task below runs end-to-end through the same 3 lines — swap the alias:
100
158
 
101
159
  | Task | Alias | Output | Example |
102
160
  | ---- | ----- | ------ | ------- |
@@ -106,6 +164,7 @@ Every task below runs end-to-end through the same 3 lines — swap the alias:
106
164
  | Pose / landmarks | `pose`, `face_landmarks` | `r.keypoints` | [pose.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/pose.py) |
107
165
  | Face analysis | `age_gender`, `emotion`, `head_pose`, `face_reid` | `r.text` (e.g. `"age 31 · male 98%"`) | [face_analysis.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_analysis.py) |
108
166
  | OCR | `text_recognition` | `r.text` | [ocr.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/ocr.py) |
167
+ | Tracking / matching | `track`, `face_match` | `r.track_ids` · `(label, score)` | [track.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/track.py) / [face_match.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_match.py) |
109
168
  | Super-resolution | `super_resolution` | upscaled image via `r.plot()` | [super_resolution.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/super_resolution.py) |
110
169
  | LLM / STT (GenAI) | `llm`, `stt` | generated text | [llm.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/llm.py) / [stt.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/stt.py) |
111
170
  | NLP / audio / time series | `qa`, `translation`, `noise_suppression`, `time_series` | tensors via `model.infer()` | [denoise_audio.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/denoise_audio.py) |
@@ -174,7 +233,10 @@ control for any model: `model.infer({name: tensor})` with `model.inputs`.
174
233
  | `r.boxes` | `xyxy`, `xywh`, `conf`, `cls` |
175
234
  | `r.masks` / `r.keypoints` / `r.probs` | masks · `[x,y,conf]` · `top1`/`top5` |
176
235
  | `r.text` | decoded text (OCR, face attributes) |
236
+ | `r.labels` / `r.track_ids` | per-box label a pipeline wrote · per-box track id |
177
237
  | `r.tensors` | raw `{name: ndarray}` |
238
+ | `r.summary()` | the whole result as one readable line |
239
+ | `r.crop(i)` / `r.to_dict()` / `r.to_json()` | cut a box out · plain Python · JSON |
178
240
  | `r.plot()` / `r.save(path)` | annotated image (or the model's output image) |
179
241
 
180
242
  </details>
@@ -24,14 +24,36 @@ ready models — with `AUTO`/`NPU`/`GPU` devices, async throughput, and INT8.
24
24
  ```python
25
25
  from ovkit import Model
26
26
 
27
- r = Model("detect")("image.jpg")[0] # download -> convert -> cache -> run
28
- r.save("out.jpg") # boxes drawn; r.boxes.xyxy / .conf / .cls
27
+ r = Model("detect", "image.jpg") # download -> convert -> cache -> run
28
+ print(r) # 2x person, car
29
+ r.save("out.jpg") # the boxes drawn on the photo
30
+ ```
31
+
32
+ One call fits every input — and Korean names work too:
33
+
34
+ ```python
35
+ r = Model("얼굴분석", "group.jpg") # == Model("face_analyze", ...)
36
+ for row in r.found: # [{'name': '사람', 'name_en': 'person',
37
+ print(row["name"], row["pos"]) # 'score': 0.93, 'box': [...], 'pos': '왼쪽 위'}]
38
+
39
+ for r in Model("track", 0): # webcam / video / "mic" -> a stream
40
+ print(r, r.elapsed_ms, r.device) # 2x person (#1, #4) 14.2 GPU
41
+ ```
42
+
43
+ The same call also runs whole **capabilities** — several models chained into one
44
+ answer, so you never have to crop, chain and stitch by hand:
45
+
46
+ ```python
47
+ r = Model("face_analyze", "group.jpg")
48
+ print(r) # 2 faces: age 31 · male 98% · happy 92%, age 27 · female 95% · neutral 88%
29
49
  ```
30
50
 
31
51
  Or without writing Python at all:
32
52
 
33
53
  ```bash
54
+ ovkit gui # a window: pick a capability, point it at your webcam
34
55
  ovkit run detect image.jpg # prints results, saves image_out.jpg
56
+ ovkit capabilities # what Model(name) can answer
35
57
  ```
36
58
 
37
59
  ## Install
@@ -49,9 +71,45 @@ python -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\act
49
71
  pip install -e ".[dev]"
50
72
  ```
51
73
 
74
+ ## Capabilities
75
+
76
+ A capability name gives you the answer, not the plumbing. Each one chains a
77
+ detector with the models that describe what it found — same `Model(...)` call,
78
+ same `Results`, and it takes the same sources (path, ndarray, folder, video,
79
+ camera index).
80
+
81
+ | `Model(...)` | Answers | Chains |
82
+ | ------------ | ------- | ------ |
83
+ | `scene` | `2 people (1 happy) · a laptop and a cup · floor 47%` | detection + segmentation + faces |
84
+ | `face_analyze` | `2 faces: age 31 · male 98% · happy 92%, ...` | face detection + age/gender + emotion (+ head pose, landmarks) |
85
+ | `person_analyze` | `3 people: male 0.98 · long pants 0.95 · bag 0.71, ...` | person detection + attributes |
86
+ | `vehicle_analyze` | `2 vehicles: type: car (0.98) · color: black (0.83), ...` | vehicle detection + type/colour |
87
+ | `read_text` | `'STOP AHEAD'` — every word, in reading order | text detection + text recognition |
88
+ | `read_plate` | `2 vehicles: black car — 12GA3456, ...` | plate detection + text recognition + vehicle attributes |
89
+ | `track` | `2x person (#1, #4)` — ids stable across frames | detection + IoU association |
90
+ | `drowsiness` | `EYES CLOSED 1.4s — drowsy` | face + landmarks + eye state + head pose, **over time** |
91
+ | `gesture` | `thumb up 0.94` | sign-language model over a rolling 8-frame clip |
92
+ | `gaze` | `1 face: looking right and slightly up` | face detection + landmarks + head pose + gaze |
93
+ | `attention` | `1 person looking at: laptop` | gaze + object detection (ray-cast into the boxes) |
94
+ | `anonymize` | the picture with every face pixelated | face (and plate) detection + redaction |
95
+ | `face_match` | `('yunjai', 0.81)` — who this is | embedding + cosine matching against your gallery |
96
+
97
+ ```python
98
+ from ovkit import Model, list_pipelines
99
+
100
+ list_pipelines() # every capability, described
101
+ Model("read_text")("sign.jpg")[0].text # 'STOP AHEAD'
102
+ Model("track")(0) # webcam, ids kept across frames
103
+ Model("face_analyze", attributes=("age_gender",)) # configure what runs
104
+ ```
105
+
106
+ Aliases: `ocr`, `anpr`, `blur`, `driver`, `describe`, `faces`, `people`, `vehicle`, `tracking`, `reid`.
107
+ `ovkit capabilities` prints the list; **`ovkit gui` opens a window** where you can
108
+ click through them against your webcam or a picture.
109
+
52
110
  ## Supported tasks
53
111
 
54
- Every task below runs end-to-end through the same 3 lines — swap the alias:
112
+ Every single-model task below runs end-to-end through the same 3 lines — swap the alias:
55
113
 
56
114
  | Task | Alias | Output | Example |
57
115
  | ---- | ----- | ------ | ------- |
@@ -61,6 +119,7 @@ Every task below runs end-to-end through the same 3 lines — swap the alias:
61
119
  | Pose / landmarks | `pose`, `face_landmarks` | `r.keypoints` | [pose.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/pose.py) |
62
120
  | Face analysis | `age_gender`, `emotion`, `head_pose`, `face_reid` | `r.text` (e.g. `"age 31 · male 98%"`) | [face_analysis.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_analysis.py) |
63
121
  | OCR | `text_recognition` | `r.text` | [ocr.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/ocr.py) |
122
+ | Tracking / matching | `track`, `face_match` | `r.track_ids` · `(label, score)` | [track.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/track.py) / [face_match.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/face_match.py) |
64
123
  | Super-resolution | `super_resolution` | upscaled image via `r.plot()` | [super_resolution.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/super_resolution.py) |
65
124
  | LLM / STT (GenAI) | `llm`, `stt` | generated text | [llm.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/llm.py) / [stt.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/stt.py) |
66
125
  | NLP / audio / time series | `qa`, `translation`, `noise_suppression`, `time_series` | tensors via `model.infer()` | [denoise_audio.py](https://github.com/leeyunjai82/ovkit/blob/main/examples/denoise_audio.py) |
@@ -129,7 +188,10 @@ control for any model: `model.infer({name: tensor})` with `model.inputs`.
129
188
  | `r.boxes` | `xyxy`, `xywh`, `conf`, `cls` |
130
189
  | `r.masks` / `r.keypoints` / `r.probs` | masks · `[x,y,conf]` · `top1`/`top5` |
131
190
  | `r.text` | decoded text (OCR, face attributes) |
191
+ | `r.labels` / `r.track_ids` | per-box label a pipeline wrote · per-box track id |
132
192
  | `r.tensors` | raw `{name: ndarray}` |
193
+ | `r.summary()` | the whole result as one readable line |
194
+ | `r.crop(i)` / `r.to_dict()` / `r.to_json()` | cut a box out · plain Python · JSON |
133
195
  | `r.plot()` / `r.save(path)` | annotated image (or the model's output image) |
134
196
 
135
197
  </details>
@@ -16,6 +16,7 @@ pip install -r examples/requirements.txt # fastapi / uvicorn (web demos only)
16
16
  | [`classify.py`](classify.py) | `classify` | top-5 classes |
17
17
  | [`ocr.py`](ocr.py) | `text_recognition` | decoded text |
18
18
  | [`super_resolution.py`](super_resolution.py) | `super_resolution` | upscaled image |
19
+ | [`background_matting.py`](background_matting.py) | `background_matting_mobilenetv2` | cut the subject out (needs frame + empty-scene photo) |
19
20
 
20
21
  ```bash
21
22
  python examples/detect.py photo.jpg # each is ~15 lines — read the source
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "ovkit"
7
- version = "0.1.2"
7
+ version = "0.2.0"
8
8
  description = "A simple Python inference API for OpenVINO: one import, one Model class, clean Results — with AUTO/NPU, async, and INT8."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -3,14 +3,23 @@
3
3
  One import, one :class:`Model` class, a callable object, and clean
4
4
  :class:`Results` — plus OpenVINO's strengths (AUTO/NPU devices, async, INT8).
5
5
 
6
- Example
7
- -------
6
+ One model
7
+ ---------
8
8
  >>> from ovkit import Model
9
9
  >>> model = Model("rtdetr_r50") # name -> auto download/convert/cache
10
10
  >>> results = model("img.jpg", conf=0.25)
11
11
  >>> for r in results:
12
- ... print(r.boxes.xyxy, r.boxes.conf, r.boxes.cls)
12
+ ... print(r.summary()) # '2x person, car'
13
13
  ... r.save("out.jpg")
14
+
15
+ Several models, one answer
16
+ --------------------------
17
+ >>> from ovkit import Model
18
+ >>> for r in Model("face_analyze")("group.jpg"):
19
+ ... print(r.summary()) # '2 faces: age 31 · male 98% · happy 92%, ...'
20
+
21
+ A capability name composes the models the answer needs — detection plus
22
+ whatever describes what was found. :func:`list_pipelines` shows them all.
14
23
  """
15
24
 
16
25
  from __future__ import annotations
@@ -29,11 +38,14 @@ from .core.errors import (
29
38
  from .core.model import Model
30
39
  from .core.registry import list_models
31
40
  from .core.results import Boxes, Keypoints, Masks, Probs, Results
41
+ from .pipelines import Pipeline, list_pipelines
32
42
 
33
- __version__ = "0.1.2"
43
+ __version__ = "0.2.0"
34
44
 
35
45
  __all__ = [
36
46
  "Model",
47
+ "Pipeline",
48
+ "list_pipelines",
37
49
  "Results",
38
50
  "Boxes",
39
51
  "Masks",
@@ -1,4 +1,4 @@
1
- """``ovkit`` command-line interface: ``run``, ``list``, ``info``, ``download``, ``devices``."""
1
+ """``ovkit`` command-line interface: ``gui``, ``run``, ``list``, ``info``, ``download``, ``devices``."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
@@ -41,6 +41,19 @@ def _cmd_list(_: argparse.Namespace) -> int:
41
41
  return 0
42
42
 
43
43
 
44
+ def _cmd_capabilities(_: argparse.Namespace) -> int:
45
+ """What ``Model(name)`` can answer beyond a single network."""
46
+ from .pipelines import ALIASES, list_pipelines
47
+
48
+ print("capabilities — Model(name) chains the models the answer needs:\n")
49
+ for name, description in list_pipelines().items():
50
+ print(f" {name:16s} {description}")
51
+ print("\naliases:")
52
+ for alias, target in sorted(ALIASES.items()):
53
+ print(f" {alias:16s} -> {target}")
54
+ return 0
55
+
56
+
44
57
  def _cmd_info(args: argparse.Namespace) -> int:
45
58
  entry = resolve(args.name)
46
59
  if entry is None:
@@ -96,25 +109,12 @@ def _cmd_run(args: argparse.Namespace) -> int:
96
109
  return 0
97
110
 
98
111
  for r in results:
99
- parts = [f"task={r.task}"]
100
- if r.text:
101
- parts.append(f'text="{r.text}"')
112
+ print(r.summary())
102
113
  if r.boxes is not None:
103
- parts.append(f"{len(r.boxes)} boxes")
104
114
  for x1, y1, x2, y2, c, cl in r.boxes.data[:20]:
105
115
  print(
106
116
  f" {r.name_for(int(cl)):16s} {c:.2f} [{int(x1)},{int(y1)},{int(x2)},{int(y2)}]"
107
117
  )
108
- if r.probs is not None:
109
- top = ", ".join(
110
- f"{r.name_for(int(i))} {r.probs.data[int(i)]:.2f}" for i in r.probs.top5
111
- )
112
- parts.append(f"top-5: {top}")
113
- if r.masks is not None:
114
- parts.append(f"masks {tuple(r.masks.data.shape)}")
115
- if r.keypoints is not None:
116
- parts.append(f"keypoints {tuple(r.keypoints.data.shape)}")
117
- print(" | ".join(parts))
118
118
 
119
119
  save = args.save
120
120
  if save is None and results and Path(str(args.source)).is_file():
@@ -125,15 +125,30 @@ def _cmd_run(args: argparse.Namespace) -> int:
125
125
  return 0
126
126
 
127
127
 
128
+ def _cmd_gui(args: argparse.Namespace) -> int:
129
+ """Open the desktop window: ``ovkit gui``."""
130
+ from .gui import main as gui_main
131
+
132
+ return gui_main(device=args.device, camera=args.camera)
133
+
134
+
128
135
  def main(argv: list[str] | None = None) -> int:
129
136
  """CLI entry point. Returns a process exit code."""
130
137
  parser = argparse.ArgumentParser(prog="ovkit", description="ovkit model utilities")
131
138
  parser.add_argument("--version", action="version", version=f"ovkit {__version__}")
132
139
  sub = parser.add_subparsers(dest="command", required=True)
133
140
 
141
+ p_gui = sub.add_parser("gui", help="open the desktop window (easiest way to try ovkit)")
142
+ p_gui.add_argument("--device", default="AUTO", help="AUTO | CPU | GPU | NPU")
143
+ p_gui.add_argument("--camera", type=int, default=0, help="camera index for the webcam button")
144
+ p_gui.set_defaults(func=_cmd_gui)
145
+
134
146
  p_list = sub.add_parser("list", help="list registered models")
135
147
  p_list.set_defaults(func=_cmd_list)
136
148
 
149
+ p_caps = sub.add_parser("capabilities", help="list composed capabilities (Model(name))")
150
+ p_caps.set_defaults(func=_cmd_capabilities)
151
+
137
152
  p_info = sub.add_parser("info", help="show details for a model")
138
153
  p_info.add_argument("name")
139
154
  p_info.set_defaults(func=_cmd_info)
@@ -0,0 +1,13 @@
1
+ """Audio I/O helpers: read a file into the array a model actually wants.
2
+
3
+ Models want float32 mono at a fixed sample rate; files are int16 stereo at
4
+ whatever rate the recorder used. These helpers close that gap so callers never
5
+ have to touch :mod:`wave` or write their own resampler.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from .ops import read_audio, read_wav, resample, to_mono, write_wav
11
+ from .plot import waveform
12
+
13
+ __all__ = ["read_audio", "read_wav", "resample", "to_mono", "waveform", "write_wav"]
@@ -0,0 +1,108 @@
1
+ """Read, convert and write audio as plain float32 numpy arrays."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import io
6
+ import wave
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ import numpy as np
11
+
12
+ from ..core.errors import OVKitError
13
+
14
+ #: Sample formats ``wave`` reports, mapped to (numpy dtype, full-scale value).
15
+ _PCM = {1: (np.int8, 128.0), 2: (np.int16, 32768.0), 4: (np.int32, 2147483648.0)}
16
+
17
+
18
+ def to_mono(audio: np.ndarray) -> np.ndarray:
19
+ """Average an ``(N, channels)`` array down to a 1-D mono signal."""
20
+ arr = np.asarray(audio, dtype=np.float32)
21
+ return arr.mean(axis=1) if arr.ndim == 2 and arr.shape[1] > 1 else arr.reshape(-1)
22
+
23
+
24
+ def resample(audio: np.ndarray, sr: int, target_sr: int) -> np.ndarray:
25
+ """Resample a 1-D signal to ``target_sr`` by linear interpolation.
26
+
27
+ Linear interpolation is not a studio-grade resampler, but it is exact when
28
+ the rates match, needs no extra dependency, and is well within the tolerance
29
+ of the classifiers and ASR models ovkit serves.
30
+ """
31
+ arr = np.asarray(audio, dtype=np.float32).reshape(-1)
32
+ if sr == target_sr or arr.size == 0:
33
+ return arr
34
+ n_out = int(round(arr.size * target_sr / float(sr)))
35
+ if n_out <= 1:
36
+ return arr[:1].copy()
37
+ src = np.linspace(0.0, arr.size - 1, num=n_out, dtype=np.float32)
38
+ return np.interp(src, np.arange(arr.size, dtype=np.float32), arr).astype(np.float32)
39
+
40
+
41
+ def read_audio(path: str | Path, target_sr: int | None = None) -> tuple[np.ndarray, int]:
42
+ """Read an audio file as ``(float32 mono in [-1, 1], sample_rate)``.
43
+
44
+ ``.wav`` is read with the standard library. Other formats are read through
45
+ ``soundfile`` when it is installed; otherwise the error says what to do
46
+ rather than failing deep inside a decoder. ``target_sr`` resamples the
47
+ result (and is what most models need, e.g. 16 kHz for speech).
48
+ """
49
+ src = Path(path)
50
+ if src.suffix.lower() == ".wav":
51
+ audio, sr = read_wav(src)
52
+ else:
53
+ audio, sr = _read_soundfile(src)
54
+ if target_sr:
55
+ audio = resample(audio, sr, target_sr)
56
+ sr = target_sr
57
+ return audio, sr
58
+
59
+
60
+ def read_wav(source: Any, target_sr: int | None = None) -> tuple[np.ndarray, int]:
61
+ """Read WAV from a path or an open/byte stream — any depth, any channels.
62
+
63
+ Uploads arrive as bytes and recordings arrive as stereo; both go through
64
+ here so callers never re-implement PCM decoding.
65
+ """
66
+ if isinstance(source, bytes):
67
+ source = io.BytesIO(source)
68
+ with wave.open(source if hasattr(source, "read") else str(source), "rb") as wf:
69
+ sr, width, channels = wf.getframerate(), wf.getsampwidth(), wf.getnchannels()
70
+ raw = wf.readframes(wf.getnframes())
71
+ if width not in _PCM:
72
+ raise OVKitError(
73
+ f"{width * 8}-bit WAV is not supported (8/16/32-bit PCM is). "
74
+ f"Re-encode it, e.g. ffmpeg -i in.wav -acodec pcm_s16le out.wav"
75
+ )
76
+ dtype, full_scale = _PCM[width]
77
+ samples = np.frombuffer(raw, dtype=dtype).astype(np.float32) / full_scale
78
+ if channels > 1:
79
+ samples = samples[: samples.size // channels * channels].reshape(-1, channels)
80
+ audio = to_mono(samples)
81
+ if target_sr:
82
+ return resample(audio, sr, target_sr), target_sr
83
+ return audio, sr
84
+
85
+
86
+ def _read_soundfile(path: Path) -> tuple[np.ndarray, int]:
87
+ try:
88
+ import soundfile as sf
89
+ except ImportError as exc:
90
+ raise OVKitError(
91
+ f"Reading '{path.suffix}' needs an extra decoder. Either convert to WAV "
92
+ f"(ffmpeg -i {path.name} -ar 16000 -ac 1 out.wav) or install one: "
93
+ f"pip install soundfile"
94
+ ) from exc
95
+ audio, sr = sf.read(str(path), dtype="float32", always_2d=True)
96
+ return to_mono(audio), int(sr)
97
+
98
+
99
+ def write_wav(path: str | Path, audio: np.ndarray, sr: int) -> Path:
100
+ """Write a float32 mono signal to a 16-bit PCM ``.wav`` and return the path."""
101
+ arr = np.clip(np.asarray(audio, dtype=np.float32).reshape(-1), -1.0, 1.0)
102
+ pcm = (arr * 32767.0).astype(np.int16)
103
+ with wave.open(str(path), "wb") as wf:
104
+ wf.setnchannels(1)
105
+ wf.setsampwidth(2)
106
+ wf.setframerate(int(sr))
107
+ wf.writeframes(pcm.tobytes())
108
+ return Path(path)
@@ -0,0 +1,47 @@
1
+ """Draw a waveform, so an audio result has something to show."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import numpy as np
6
+
7
+
8
+ def waveform(
9
+ audio: np.ndarray,
10
+ sr: int = 16_000,
11
+ width: int = 720,
12
+ height: int = 200,
13
+ color: tuple[int, int, int] = (200, 190, 120),
14
+ ) -> np.ndarray:
15
+ """Render a mono signal as a BGR image.
16
+
17
+ Audio results carry this as their image so ``plot()``, ``save()`` and the
18
+ web demo work the same way they do for a picture.
19
+ """
20
+ import cv2
21
+
22
+ img = np.full((height, width, 3), 24, np.uint8)
23
+ samples = np.asarray(audio, np.float32).reshape(-1)
24
+ mid = height // 2
25
+ cv2.line(img, (0, mid), (width, mid), (60, 60, 60), 1)
26
+ if samples.size:
27
+ # One vertical bar per column: the min/max of the samples it covers.
28
+ edges = np.linspace(0, samples.size, width + 1).astype(int)
29
+ peak = max(float(np.abs(samples).max()), 1e-6)
30
+ for x in range(width):
31
+ chunk = samples[edges[x] : max(edges[x + 1], edges[x] + 1)]
32
+ if not chunk.size:
33
+ continue
34
+ lo, hi = float(chunk.min()) / peak, float(chunk.max()) / peak
35
+ cv2.line(img, (x, mid - int(hi * mid * 0.9)), (x, mid - int(lo * mid * 0.9)), color, 1)
36
+ seconds = samples.size / float(sr or 1)
37
+ cv2.putText(
38
+ img,
39
+ f"{seconds:.1f}s @ {sr} Hz",
40
+ (8, height - 8),
41
+ cv2.FONT_HERSHEY_SIMPLEX,
42
+ 0.45,
43
+ (140, 140, 140),
44
+ 1,
45
+ cv2.LINE_AA,
46
+ )
47
+ return img
@@ -32,6 +32,39 @@ def available_devices() -> list[str]:
32
32
  return list(core().available_devices)
33
33
 
34
34
 
35
+ def _friendlier_compile_error(exc: Exception, model: str | Path | Any) -> Exception:
36
+ """Turn OpenVINO's nested compile errors into something actionable.
37
+
38
+ The most common one by far is IR whose ``.bin`` never arrived (a mirror
39
+ upload that dropped the weights, or an interrupted download). OpenVINO
40
+ reports it as "Empty weights data in bin file", buried under two layers of
41
+ "Exception from ...", which tells a user nothing about what to do.
42
+ """
43
+ from .errors import OVKitError
44
+
45
+ text = str(exc)
46
+ if "Empty weights data" not in text and "bin file" not in text:
47
+ return exc
48
+ if not isinstance(model, (str, Path)):
49
+ return exc
50
+ xml = Path(str(model))
51
+ if xml.suffix != ".xml":
52
+ return exc
53
+ bin_path = xml.with_suffix(".bin")
54
+ if not bin_path.exists():
55
+ state = "is missing"
56
+ elif bin_path.stat().st_size < 1024:
57
+ state = f"is only {bin_path.stat().st_size} bytes"
58
+ else:
59
+ return exc # weights look fine — a different problem, keep the original
60
+ return OVKitError(
61
+ f"The weights file for this model {state}: {bin_path}\n"
62
+ f"An OpenVINO IR needs both model.xml and model.bin. Delete the cached "
63
+ f"copy and download it again; if it keeps happening, the mirrored model "
64
+ f"itself is incomplete and needs re-uploading."
65
+ )
66
+
67
+
35
68
  class Backend:
36
69
  """A compiled model bound to a device, with sync and async inference.
37
70
 
@@ -47,7 +80,10 @@ class Backend:
47
80
  self.device = device
48
81
  c = core()
49
82
  src = str(model) if isinstance(model, (str, Path)) else model
50
- self.compiled = c.compile_model(src, device)
83
+ try:
84
+ self.compiled = c.compile_model(src, device)
85
+ except Exception as exc:
86
+ raise _friendlier_compile_error(exc, model) from exc
51
87
  self.inputs = self.compiled.inputs
52
88
  self.outputs = self.compiled.outputs
53
89
 
@@ -97,6 +133,15 @@ class Backend:
97
133
  return np.repeat(arr, 3, axis=1)
98
134
  return arr
99
135
 
136
+ @property
137
+ def actual_device(self) -> str:
138
+ """The device inference actually runs on (AUTO resolves to a real one)."""
139
+ try:
140
+ devices = self.compiled.get_property("EXECUTION_DEVICES")
141
+ return ",".join(devices) if devices else self.device
142
+ except Exception:
143
+ return self.device
144
+
100
145
  def output_signatures(self) -> list[tuple[str, tuple[int, ...]]]:
101
146
  """Return ``(name, shape)`` for each output (``-1`` for dynamic dims)."""
102
147
  sigs: list[tuple[str, tuple[int, ...]]] = []