omnismi 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- omnismi-1.0.0/LICENSE +21 -0
- omnismi-1.0.0/PKG-INFO +206 -0
- omnismi-1.0.0/README.md +166 -0
- omnismi-1.0.0/pyproject.toml +87 -0
- omnismi-1.0.0/setup.cfg +4 -0
- omnismi-1.0.0/src/omnismi/__init__.py +19 -0
- omnismi-1.0.0/src/omnismi/api.py +112 -0
- omnismi-1.0.0/src/omnismi/backends/__init__.py +15 -0
- omnismi-1.0.0/src/omnismi/backends/amd.py +314 -0
- omnismi-1.0.0/src/omnismi/backends/base.py +34 -0
- omnismi-1.0.0/src/omnismi/backends/nvidia.py +324 -0
- omnismi-1.0.0/src/omnismi/backends/registry.py +91 -0
- omnismi-1.0.0/src/omnismi/errors.py +15 -0
- omnismi-1.0.0/src/omnismi/models.py +35 -0
- omnismi-1.0.0/src/omnismi/normalize.py +131 -0
- omnismi-1.0.0/src/omnismi/validation/__init__.py +1 -0
- omnismi-1.0.0/src/omnismi/validation/parity.py +322 -0
- omnismi-1.0.0/src/omnismi.egg-info/PKG-INFO +206 -0
- omnismi-1.0.0/src/omnismi.egg-info/SOURCES.txt +26 -0
- omnismi-1.0.0/src/omnismi.egg-info/dependency_links.txt +1 -0
- omnismi-1.0.0/src/omnismi.egg-info/requires.txt +23 -0
- omnismi-1.0.0/src/omnismi.egg-info/top_level.txt +1 -0
- omnismi-1.0.0/tests/test_amd_backend_mock.py +92 -0
- omnismi-1.0.0/tests/test_api_contract.py +76 -0
- omnismi-1.0.0/tests/test_backend_registry.py +77 -0
- omnismi-1.0.0/tests/test_no_vendor_installed.py +13 -0
- omnismi-1.0.0/tests/test_normalization.py +33 -0
- omnismi-1.0.0/tests/test_nvidia_backend_mock.py +107 -0
omnismi-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 GreatHato
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
omnismi-1.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
Metadata-Version: 2.2
|
|
2
|
+
Name: omnismi
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Cross-vendor GPU observability for AI agents and Python apps (NVIDIA and AMD today)
|
|
5
|
+
Author: Omnismi Contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/omnismi/omnismi
|
|
8
|
+
Project-URL: Repository, https://github.com/omnismi/omnismi
|
|
9
|
+
Project-URL: Documentation, https://github.com/omnismi/omnismi/tree/main/docs
|
|
10
|
+
Keywords: gpu,cross-vendor,nvidia,amd,observability,ai-agent
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Provides-Extra: all
|
|
23
|
+
Requires-Dist: nvidia-ml-py>=11.525.84; extra == "all"
|
|
24
|
+
Requires-Dist: amdsmi>=6.0.0; extra == "all"
|
|
25
|
+
Provides-Extra: nvidia
|
|
26
|
+
Requires-Dist: nvidia-ml-py>=11.525.84; extra == "nvidia"
|
|
27
|
+
Provides-Extra: amd
|
|
28
|
+
Requires-Dist: amdsmi>=6.0.0; extra == "amd"
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
31
|
+
Requires-Dist: pytest-cov>=5.0; extra == "dev"
|
|
32
|
+
Requires-Dist: mypy>=1.8; extra == "dev"
|
|
33
|
+
Requires-Dist: black>=24.0; extra == "dev"
|
|
34
|
+
Requires-Dist: isort>=5.13; extra == "dev"
|
|
35
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
36
|
+
Provides-Extra: docs
|
|
37
|
+
Requires-Dist: mkdocs>=1.6; extra == "docs"
|
|
38
|
+
Requires-Dist: mkdocs-material>=9.5; extra == "docs"
|
|
39
|
+
Requires-Dist: mkdocstrings[python]>=0.24; extra == "docs"
|
|
40
|
+
|
|
41
|
+
# Omnismi
|
|
42
|
+
|
|
43
|
+
Cross-vendor GPU observability for AI agents and Python apps (NVIDIA and AMD today).
|
|
44
|
+
|
|
45
|
+
Omnismi provides a compact and stable Python API for reading GPU information and metrics across vendors.
|
|
46
|
+
The long-term goal is broader vendor coverage without changing user-facing API contracts.
|
|
47
|
+
|
|
48
|
+
## Why Omnismi
|
|
49
|
+
|
|
50
|
+
- Unified API across vendors: `count`, `gpus`, `gpu`, `info`, `metrics`.
|
|
51
|
+
- Fixed normalized units: bytes, percent, Celsius, Watts, MHz.
|
|
52
|
+
- Graceful degradation: unavailable metrics return `None` instead of raising by default.
|
|
53
|
+
- NVIDIA psutil-style cached sampling plus `GPU.realtime()` for forced live reads.
|
|
54
|
+
- Built-in parity checker to compare normalized output with direct vendor readings.
|
|
55
|
+
|
|
56
|
+
## Why not just torch/pynvml/amdsmi?
|
|
57
|
+
|
|
58
|
+
PyTorch memory APIs are useful in framework workflows, but are framework-scoped and not designed as a
|
|
59
|
+
cross-vendor observability contract for general runtime checks. Direct vendor bindings are essential,
|
|
60
|
+
but each has different lifecycle, naming, and compatibility details. Omnismi adds a stable cross-vendor
|
|
61
|
+
contract for agent preflight and application telemetry. See [docs/why-omnismi.md](docs/why-omnismi.md).
|
|
62
|
+
|
|
63
|
+
## Adapter Matrix (Ground Truth Libraries)
|
|
64
|
+
|
|
65
|
+
| Vendor | Runtime/Driver Stack | Ground Truth Library | Router Status |
|
|
66
|
+
|---|---|---|---|
|
|
67
|
+
| NVIDIA | CUDA + NVML | `nvidia-ml-py` | ✅ Supported |
|
|
68
|
+
| AMD | ROCm + AMD SMI | `amdsmi` | ✅ Supported |
|
|
69
|
+
| Intel | oneAPI + Level Zero | TBD | ⬜ Planned |
|
|
70
|
+
| Apple | Metal | TBD | ⬜ Planned |
|
|
71
|
+
|
|
72
|
+
Status legend:
|
|
73
|
+
- `✅ Supported`: Adapter path is integrated and maintained.
|
|
74
|
+
- `🟡 Partial`: Adapter is integrated but some metrics/features are incomplete.
|
|
75
|
+
- `🧪 Awaiting User Validation`: Adapter path exists; model/version evidence is still needed.
|
|
76
|
+
- `⬜ Planned`: Vendor adapter is not integrated yet.
|
|
77
|
+
|
|
78
|
+
## Hardware Validation Status
|
|
79
|
+
|
|
80
|
+
| Vendor | Model | Status | Evidence |
|
|
81
|
+
|---|---|---|---|
|
|
82
|
+
| NVIDIA | H20 | ✅ Verified | [v1.0.0 release note](CHANGELOG.md#100---2026-02-25) |
|
|
83
|
+
| AMD | MI300X | ✅ Verified | [v1.0.0 release note](CHANGELOG.md#100---2026-02-25) |
|
|
84
|
+
|
|
85
|
+
See full matrix in [docs/compatibility.md](docs/compatibility.md).
|
|
86
|
+
|
|
87
|
+
## Install
|
|
88
|
+
|
|
89
|
+
Omnismi core is lightweight and has no mandatory vendor dependency.
|
|
90
|
+
Pick the install command that matches your environment:
|
|
91
|
+
|
|
92
|
+
| Your environment | What to install | Command |
|
|
93
|
+
|---|---|---|
|
|
94
|
+
| No GPU / CI / just developing API integration | Core package only | `pip install omnismi` |
|
|
95
|
+
| NVIDIA GPUs only | Core + NVIDIA backend dependency | `pip install "omnismi[nvidia]"` |
|
|
96
|
+
| AMD GPUs only | Core + AMD backend dependency | `pip install "omnismi[amd]"` |
|
|
97
|
+
| Mixed cluster or shared image | Core + NVIDIA + AMD dependencies | `pip install "omnismi[all]"` |
|
|
98
|
+
|
|
99
|
+
### Install From Local Source
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
# from repo root
|
|
103
|
+
python -m pip install -e ".[all]"
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
If you only need one vendor backend during local development:
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
python -m pip install -e ".[nvidia]"
|
|
110
|
+
# or
|
|
111
|
+
python -m pip install -e ".[amd]"
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## Quick Start
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
import omnismi as omi
|
|
118
|
+
|
|
119
|
+
# 1) Count GPUs
|
|
120
|
+
gpu_count = omi.count()
|
|
121
|
+
|
|
122
|
+
# 2) Check whether GPU exists
|
|
123
|
+
has_gpu = gpu_count > 0
|
|
124
|
+
|
|
125
|
+
# 3) Get max total GPU memory (bytes) across visible devices
|
|
126
|
+
max_memory_bytes = max(
|
|
127
|
+
(dev.info().memory_total_bytes or 0 for dev in omi.gpus()),
|
|
128
|
+
default=0,
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
print(f"gpu_count={gpu_count}")
|
|
132
|
+
print(f"has_gpu={has_gpu}")
|
|
133
|
+
print(f"max_memory_bytes={max_memory_bytes}")
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
## API
|
|
137
|
+
|
|
138
|
+
- `omi.count() -> int`
|
|
139
|
+
- `omi.gpus() -> list[GPU]`
|
|
140
|
+
- `omi.gpu(index: int) -> GPU | None`
|
|
141
|
+
- `GPU.info() -> GPUInfo`
|
|
142
|
+
- `GPU.metrics() -> GPUMetrics`
|
|
143
|
+
- `GPU.realtime() -> context manager` (force live reads when backend supports it)
|
|
144
|
+
|
|
145
|
+
## Current Support and Semantics
|
|
146
|
+
|
|
147
|
+
| Vendor | Status | Backend dependency | Read semantics |
|
|
148
|
+
|---|---|---|---|
|
|
149
|
+
| NVIDIA | Supported | `nvidia-ml-py` | Read-only, normalized units, unavailable values return `None` |
|
|
150
|
+
| AMD | Supported | `amdsmi` | Read-only, normalized units, unavailable values return `None` |
|
|
151
|
+
| Other vendors (Intel, Apple, etc.) | Planned | TBD | Same API contract (`count/gpus/gpu`, `info/metrics`) |
|
|
152
|
+
|
|
153
|
+
| Metric field | Unit | Semantic |
|
|
154
|
+
|---|---|---|
|
|
155
|
+
| `utilization_percent` | `%` | GPU utilization percentage when available |
|
|
156
|
+
| `memory_used_bytes` / `memory_total_bytes` | `bytes` | Memory usage/total in bytes |
|
|
157
|
+
| `temperature_c` | `C` | Device temperature in Celsius |
|
|
158
|
+
| `power_w` | `W` | Power usage in Watts |
|
|
159
|
+
| `core_clock_mhz` / `memory_clock_mhz` | `MHz` | Core/memory clock when available |
|
|
160
|
+
|
|
161
|
+
### Sampling Semantics (NVIDIA)
|
|
162
|
+
|
|
163
|
+
- Omnismi initializes NVML lazily on first NVIDIA backend use (`nvmlInit`).
|
|
164
|
+
- `GPU.metrics()` is psutil-style for NVIDIA: repeated calls return the latest cached sample instead of reading NVML every call.
|
|
165
|
+
- A background sampler refreshes cached metrics periodically (default 0.5s interval).
|
|
166
|
+
- On process exit or backend teardown, Omnismi calls `nvmlShutdown()`.
|
|
167
|
+
|
|
168
|
+
Use realtime mode only when you explicitly need per-call direct reads:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
import omnismi as omi
|
|
172
|
+
|
|
173
|
+
dev = omi.gpu(0)
|
|
174
|
+
if dev is not None:
|
|
175
|
+
with dev.realtime():
|
|
176
|
+
live = dev.metrics() # bypass cache for this call path
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
## Roadmap (Todo)
|
|
180
|
+
|
|
181
|
+
- Extend backend coverage to more GPU vendors.
|
|
182
|
+
- Improve compatibility matrix depth across drivers/runtimes/architectures.
|
|
183
|
+
- Strengthen parity validation workflow and reporting.
|
|
184
|
+
- Expand hardware-backed tests and reproducibility tooling.
|
|
185
|
+
- Keep API minimal while improving metric quality and consistency.
|
|
186
|
+
|
|
187
|
+
## Documentation
|
|
188
|
+
|
|
189
|
+
- API and usage docs: `docs/`
|
|
190
|
+
- Build docs locally: `mkdocs serve`
|
|
191
|
+
- Parity validation: `python -m omnismi.validation.parity --vendor nvidia --samples 3`
|
|
192
|
+
|
|
193
|
+
## Local Validation
|
|
194
|
+
|
|
195
|
+
```bash
|
|
196
|
+
# run unit tests
|
|
197
|
+
PYTHONPATH=src pytest -q
|
|
198
|
+
|
|
199
|
+
# compare normalized output against direct vendor API
|
|
200
|
+
PYTHONPATH=src python -m omnismi.validation.parity --vendor nvidia --samples 3
|
|
201
|
+
PYTHONPATH=src python -m omnismi.validation.parity --vendor amd --samples 3
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
## License
|
|
205
|
+
|
|
206
|
+
MIT
|
omnismi-1.0.0/README.md
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# Omnismi
|
|
2
|
+
|
|
3
|
+
Cross-vendor GPU observability for AI agents and Python apps (NVIDIA and AMD today).
|
|
4
|
+
|
|
5
|
+
Omnismi provides a compact and stable Python API for reading GPU information and metrics across vendors.
|
|
6
|
+
The long-term goal is broader vendor coverage without changing user-facing API contracts.
|
|
7
|
+
|
|
8
|
+
## Why Omnismi
|
|
9
|
+
|
|
10
|
+
- Unified API across vendors: `count`, `gpus`, `gpu`, `info`, `metrics`.
|
|
11
|
+
- Fixed normalized units: bytes, percent, Celsius, Watts, MHz.
|
|
12
|
+
- Graceful degradation: unavailable metrics return `None` instead of raising by default.
|
|
13
|
+
- NVIDIA psutil-style cached sampling plus `GPU.realtime()` for forced live reads.
|
|
14
|
+
- Built-in parity checker to compare normalized output with direct vendor readings.
|
|
15
|
+
|
|
16
|
+
## Why not just torch/pynvml/amdsmi?
|
|
17
|
+
|
|
18
|
+
PyTorch memory APIs are useful in framework workflows, but are framework-scoped and not designed as a
|
|
19
|
+
cross-vendor observability contract for general runtime checks. Direct vendor bindings are essential,
|
|
20
|
+
but each has different lifecycle, naming, and compatibility details. Omnismi adds a stable cross-vendor
|
|
21
|
+
contract for agent preflight and application telemetry. See [docs/why-omnismi.md](docs/why-omnismi.md).
|
|
22
|
+
|
|
23
|
+
## Adapter Matrix (Ground Truth Libraries)
|
|
24
|
+
|
|
25
|
+
| Vendor | Runtime/Driver Stack | Ground Truth Library | Router Status |
|
|
26
|
+
|---|---|---|---|
|
|
27
|
+
| NVIDIA | CUDA + NVML | `nvidia-ml-py` | ✅ Supported |
|
|
28
|
+
| AMD | ROCm + AMD SMI | `amdsmi` | ✅ Supported |
|
|
29
|
+
| Intel | oneAPI + Level Zero | TBD | ⬜ Planned |
|
|
30
|
+
| Apple | Metal | TBD | ⬜ Planned |
|
|
31
|
+
|
|
32
|
+
Status legend:
|
|
33
|
+
- `✅ Supported`: Adapter path is integrated and maintained.
|
|
34
|
+
- `🟡 Partial`: Adapter is integrated but some metrics/features are incomplete.
|
|
35
|
+
- `🧪 Awaiting User Validation`: Adapter path exists; model/version evidence is still needed.
|
|
36
|
+
- `⬜ Planned`: Vendor adapter is not integrated yet.
|
|
37
|
+
|
|
38
|
+
## Hardware Validation Status
|
|
39
|
+
|
|
40
|
+
| Vendor | Model | Status | Evidence |
|
|
41
|
+
|---|---|---|---|
|
|
42
|
+
| NVIDIA | H20 | ✅ Verified | [v1.0.0 release note](CHANGELOG.md#100---2026-02-25) |
|
|
43
|
+
| AMD | MI300X | ✅ Verified | [v1.0.0 release note](CHANGELOG.md#100---2026-02-25) |
|
|
44
|
+
|
|
45
|
+
See full matrix in [docs/compatibility.md](docs/compatibility.md).
|
|
46
|
+
|
|
47
|
+
## Install
|
|
48
|
+
|
|
49
|
+
Omnismi core is lightweight and has no mandatory vendor dependency.
|
|
50
|
+
Pick the install command that matches your environment:
|
|
51
|
+
|
|
52
|
+
| Your environment | What to install | Command |
|
|
53
|
+
|---|---|---|
|
|
54
|
+
| No GPU / CI / just developing API integration | Core package only | `pip install omnismi` |
|
|
55
|
+
| NVIDIA GPUs only | Core + NVIDIA backend dependency | `pip install "omnismi[nvidia]"` |
|
|
56
|
+
| AMD GPUs only | Core + AMD backend dependency | `pip install "omnismi[amd]"` |
|
|
57
|
+
| Mixed cluster or shared image | Core + NVIDIA + AMD dependencies | `pip install "omnismi[all]"` |
|
|
58
|
+
|
|
59
|
+
### Install From Local Source
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
# from repo root
|
|
63
|
+
python -m pip install -e ".[all]"
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
If you only need one vendor backend during local development:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
python -m pip install -e ".[nvidia]"
|
|
70
|
+
# or
|
|
71
|
+
python -m pip install -e ".[amd]"
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Quick Start
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
import omnismi as omi
|
|
78
|
+
|
|
79
|
+
# 1) Count GPUs
|
|
80
|
+
gpu_count = omi.count()
|
|
81
|
+
|
|
82
|
+
# 2) Check whether GPU exists
|
|
83
|
+
has_gpu = gpu_count > 0
|
|
84
|
+
|
|
85
|
+
# 3) Get max total GPU memory (bytes) across visible devices
|
|
86
|
+
max_memory_bytes = max(
|
|
87
|
+
(dev.info().memory_total_bytes or 0 for dev in omi.gpus()),
|
|
88
|
+
default=0,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
print(f"gpu_count={gpu_count}")
|
|
92
|
+
print(f"has_gpu={has_gpu}")
|
|
93
|
+
print(f"max_memory_bytes={max_memory_bytes}")
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## API
|
|
97
|
+
|
|
98
|
+
- `omi.count() -> int`
|
|
99
|
+
- `omi.gpus() -> list[GPU]`
|
|
100
|
+
- `omi.gpu(index: int) -> GPU | None`
|
|
101
|
+
- `GPU.info() -> GPUInfo`
|
|
102
|
+
- `GPU.metrics() -> GPUMetrics`
|
|
103
|
+
- `GPU.realtime() -> context manager` (force live reads when backend supports it)
|
|
104
|
+
|
|
105
|
+
## Current Support and Semantics
|
|
106
|
+
|
|
107
|
+
| Vendor | Status | Backend dependency | Read semantics |
|
|
108
|
+
|---|---|---|---|
|
|
109
|
+
| NVIDIA | Supported | `nvidia-ml-py` | Read-only, normalized units, unavailable values return `None` |
|
|
110
|
+
| AMD | Supported | `amdsmi` | Read-only, normalized units, unavailable values return `None` |
|
|
111
|
+
| Other vendors (Intel, Apple, etc.) | Planned | TBD | Same API contract (`count/gpus/gpu`, `info/metrics`) |
|
|
112
|
+
|
|
113
|
+
| Metric field | Unit | Semantic |
|
|
114
|
+
|---|---|---|
|
|
115
|
+
| `utilization_percent` | `%` | GPU utilization percentage when available |
|
|
116
|
+
| `memory_used_bytes` / `memory_total_bytes` | `bytes` | Memory usage/total in bytes |
|
|
117
|
+
| `temperature_c` | `C` | Device temperature in Celsius |
|
|
118
|
+
| `power_w` | `W` | Power usage in Watts |
|
|
119
|
+
| `core_clock_mhz` / `memory_clock_mhz` | `MHz` | Core/memory clock when available |
|
|
120
|
+
|
|
121
|
+
### Sampling Semantics (NVIDIA)
|
|
122
|
+
|
|
123
|
+
- Omnismi initializes NVML lazily on first NVIDIA backend use (`nvmlInit`).
|
|
124
|
+
- `GPU.metrics()` is psutil-style for NVIDIA: repeated calls return the latest cached sample instead of reading NVML every call.
|
|
125
|
+
- A background sampler refreshes cached metrics periodically (default 0.5s interval).
|
|
126
|
+
- On process exit or backend teardown, Omnismi calls `nvmlShutdown()`.
|
|
127
|
+
|
|
128
|
+
Use realtime mode only when you explicitly need per-call direct reads:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
import omnismi as omi
|
|
132
|
+
|
|
133
|
+
dev = omi.gpu(0)
|
|
134
|
+
if dev is not None:
|
|
135
|
+
with dev.realtime():
|
|
136
|
+
live = dev.metrics() # bypass cache for this call path
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## Roadmap (Todo)
|
|
140
|
+
|
|
141
|
+
- Extend backend coverage to more GPU vendors.
|
|
142
|
+
- Improve compatibility matrix depth across drivers/runtimes/architectures.
|
|
143
|
+
- Strengthen parity validation workflow and reporting.
|
|
144
|
+
- Expand hardware-backed tests and reproducibility tooling.
|
|
145
|
+
- Keep API minimal while improving metric quality and consistency.
|
|
146
|
+
|
|
147
|
+
## Documentation
|
|
148
|
+
|
|
149
|
+
- API and usage docs: `docs/`
|
|
150
|
+
- Build docs locally: `mkdocs serve`
|
|
151
|
+
- Parity validation: `python -m omnismi.validation.parity --vendor nvidia --samples 3`
|
|
152
|
+
|
|
153
|
+
## Local Validation
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
# run unit tests
|
|
157
|
+
PYTHONPATH=src pytest -q
|
|
158
|
+
|
|
159
|
+
# compare normalized output against direct vendor API
|
|
160
|
+
PYTHONPATH=src python -m omnismi.validation.parity --vendor nvidia --samples 3
|
|
161
|
+
PYTHONPATH=src python -m omnismi.validation.parity --vendor amd --samples 3
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## License
|
|
165
|
+
|
|
166
|
+
MIT
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=64.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "omnismi"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "Cross-vendor GPU observability for AI agents and Python apps (NVIDIA and AMD today)"
|
|
9
|
+
keywords = ["gpu", "cross-vendor", "nvidia", "amd", "observability", "ai-agent"]
|
|
10
|
+
readme = "README.md"
|
|
11
|
+
license = {text = "MIT"}
|
|
12
|
+
requires-python = ">=3.9"
|
|
13
|
+
authors = [
|
|
14
|
+
{name = "Omnismi Contributors"}
|
|
15
|
+
]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 4 - Beta",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"License :: OSI Approved :: MIT License",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.9",
|
|
22
|
+
"Programming Language :: Python :: 3.10",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
]
|
|
26
|
+
dependencies = []
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
all = [
|
|
30
|
+
"nvidia-ml-py>=11.525.84",
|
|
31
|
+
"amdsmi>=6.0.0",
|
|
32
|
+
]
|
|
33
|
+
nvidia = [
|
|
34
|
+
"nvidia-ml-py>=11.525.84",
|
|
35
|
+
]
|
|
36
|
+
amd = [
|
|
37
|
+
"amdsmi>=6.0.0",
|
|
38
|
+
]
|
|
39
|
+
dev = [
|
|
40
|
+
"pytest>=8.0",
|
|
41
|
+
"pytest-cov>=5.0",
|
|
42
|
+
"mypy>=1.8",
|
|
43
|
+
"black>=24.0",
|
|
44
|
+
"isort>=5.13",
|
|
45
|
+
"ruff>=0.6",
|
|
46
|
+
]
|
|
47
|
+
docs = [
|
|
48
|
+
"mkdocs>=1.6",
|
|
49
|
+
"mkdocs-material>=9.5",
|
|
50
|
+
"mkdocstrings[python]>=0.24",
|
|
51
|
+
]
|
|
52
|
+
|
|
53
|
+
[project.urls]
|
|
54
|
+
Homepage = "https://github.com/omnismi/omnismi"
|
|
55
|
+
Repository = "https://github.com/omnismi/omnismi"
|
|
56
|
+
Documentation = "https://github.com/omnismi/omnismi/tree/main/docs"
|
|
57
|
+
|
|
58
|
+
[tool.setuptools.packages.find]
|
|
59
|
+
where = ["src"]
|
|
60
|
+
|
|
61
|
+
[tool.pytest.ini_options]
|
|
62
|
+
testpaths = ["tests"]
|
|
63
|
+
python_files = ["test_*.py"]
|
|
64
|
+
python_classes = ["Test*"]
|
|
65
|
+
python_functions = ["test_*"]
|
|
66
|
+
|
|
67
|
+
[tool.black]
|
|
68
|
+
line-length = 88
|
|
69
|
+
target-version = ["py39", "py310", "py311", "py312"]
|
|
70
|
+
|
|
71
|
+
[tool.isort]
|
|
72
|
+
profile = "black"
|
|
73
|
+
line-length = 88
|
|
74
|
+
|
|
75
|
+
[tool.ruff]
|
|
76
|
+
line-length = 88
|
|
77
|
+
target-version = "py39"
|
|
78
|
+
|
|
79
|
+
[tool.ruff.lint]
|
|
80
|
+
select = ["E", "F", "W", "I", "N", "UP"]
|
|
81
|
+
ignore = []
|
|
82
|
+
|
|
83
|
+
[tool.mypy]
|
|
84
|
+
python_version = "3.9"
|
|
85
|
+
strict = false
|
|
86
|
+
warn_return_any = true
|
|
87
|
+
warn_unused_configs = true
|
omnismi-1.0.0/setup.cfg
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Omnismi: unified, minimal GPU observability API."""
|
|
2
|
+
|
|
3
|
+
from omnismi.api import GPU, count, gpu, gpus
|
|
4
|
+
from omnismi.errors import BackendError, OmnismiError, ValidationError
|
|
5
|
+
from omnismi.models import GPUMetrics, GPUInfo
|
|
6
|
+
|
|
7
|
+
__version__ = "1.0.0"
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"BackendError",
|
|
11
|
+
"GPU",
|
|
12
|
+
"GPUInfo",
|
|
13
|
+
"GPUMetrics",
|
|
14
|
+
"OmnismiError",
|
|
15
|
+
"ValidationError",
|
|
16
|
+
"count",
|
|
17
|
+
"gpu",
|
|
18
|
+
"gpus",
|
|
19
|
+
]
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Public Omnismi API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from contextlib import nullcontext
|
|
6
|
+
import time
|
|
7
|
+
from dataclasses import replace
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from omnismi.backends.base import BaseBackend
|
|
11
|
+
from omnismi.backends.registry import active_backends
|
|
12
|
+
from omnismi.models import GPUMetrics, GPUInfo
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class GPU:
|
|
16
|
+
"""Lightweight GPU object with psutil-style read methods."""
|
|
17
|
+
|
|
18
|
+
__slots__ = ("_backend", "_device", "_index")
|
|
19
|
+
|
|
20
|
+
def __init__(self, backend: BaseBackend, device: Any, index: int) -> None:
|
|
21
|
+
self._backend = backend
|
|
22
|
+
self._device = device
|
|
23
|
+
self._index = index
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def index(self) -> int:
|
|
27
|
+
"""Global Omnismi GPU index."""
|
|
28
|
+
return self._index
|
|
29
|
+
|
|
30
|
+
def info(self) -> GPUInfo:
|
|
31
|
+
"""Return static information for this GPU."""
|
|
32
|
+
try:
|
|
33
|
+
info = self._backend.info(self._device, self._index)
|
|
34
|
+
if info.index != self._index:
|
|
35
|
+
info = replace(info, index=self._index)
|
|
36
|
+
return info
|
|
37
|
+
except Exception:
|
|
38
|
+
return GPUInfo(
|
|
39
|
+
index=self._index,
|
|
40
|
+
vendor=self._backend.vendor,
|
|
41
|
+
name=f"{self._backend.vendor.upper()} GPU {self._index}",
|
|
42
|
+
uuid=None,
|
|
43
|
+
driver=None,
|
|
44
|
+
memory_total_bytes=None,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
def metrics(self) -> GPUMetrics:
|
|
48
|
+
"""Return current metrics for this GPU."""
|
|
49
|
+
try:
|
|
50
|
+
metrics = self._backend.metrics(self._device, self._index)
|
|
51
|
+
if metrics.index != self._index:
|
|
52
|
+
metrics = replace(metrics, index=self._index)
|
|
53
|
+
return metrics
|
|
54
|
+
except Exception:
|
|
55
|
+
return GPUMetrics(
|
|
56
|
+
index=self._index,
|
|
57
|
+
utilization_percent=None,
|
|
58
|
+
memory_used_bytes=None,
|
|
59
|
+
memory_total_bytes=None,
|
|
60
|
+
temperature_c=None,
|
|
61
|
+
power_w=None,
|
|
62
|
+
core_clock_mhz=None,
|
|
63
|
+
memory_clock_mhz=None,
|
|
64
|
+
timestamp_ns=time.time_ns(),
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
def realtime(self):
|
|
68
|
+
"""Return a context that forces live metric reads when supported by backend."""
|
|
69
|
+
mode = getattr(self._backend, "realtime_mode", None)
|
|
70
|
+
if callable(mode):
|
|
71
|
+
try:
|
|
72
|
+
return mode()
|
|
73
|
+
except Exception:
|
|
74
|
+
return nullcontext()
|
|
75
|
+
return nullcontext()
|
|
76
|
+
|
|
77
|
+
def __repr__(self) -> str:
|
|
78
|
+
return f"GPU(index={self._index}, vendor={self._backend.vendor!r})"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def gpus() -> list[GPU]:
|
|
82
|
+
"""Return all currently visible GPUs across all active backends."""
|
|
83
|
+
devices: list[GPU] = []
|
|
84
|
+
next_index = 0
|
|
85
|
+
|
|
86
|
+
for backend in active_backends():
|
|
87
|
+
try:
|
|
88
|
+
backend_devices = backend.devices()
|
|
89
|
+
except Exception:
|
|
90
|
+
continue
|
|
91
|
+
|
|
92
|
+
for device in backend_devices:
|
|
93
|
+
devices.append(GPU(backend=backend, device=device, index=next_index))
|
|
94
|
+
next_index += 1
|
|
95
|
+
|
|
96
|
+
return devices
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def count() -> int:
|
|
100
|
+
"""Return the number of currently visible GPUs."""
|
|
101
|
+
return len(gpus())
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def gpu(index: int) -> GPU | None:
|
|
105
|
+
"""Return a single GPU by global index, or None if out of range."""
|
|
106
|
+
if index < 0:
|
|
107
|
+
return None
|
|
108
|
+
|
|
109
|
+
for device in gpus():
|
|
110
|
+
if device.index == index:
|
|
111
|
+
return device
|
|
112
|
+
return None
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Backend implementations for supported GPU vendors."""
|
|
2
|
+
|
|
3
|
+
from omnismi.backends.amd import AmdBackend
|
|
4
|
+
from omnismi.backends.base import BaseBackend
|
|
5
|
+
from omnismi.backends.nvidia import NvidiaBackend
|
|
6
|
+
from omnismi.backends.registry import active_backends, close_all, registered_backends
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"AmdBackend",
|
|
10
|
+
"BaseBackend",
|
|
11
|
+
"NvidiaBackend",
|
|
12
|
+
"active_backends",
|
|
13
|
+
"registered_backends",
|
|
14
|
+
"close_all",
|
|
15
|
+
]
|