llm-inspector 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. llm_inspector-0.6.0/.github/workflows/publish.yml +37 -0
  2. llm_inspector-0.6.0/.gitignore +18 -0
  3. llm_inspector-0.6.0/LICENSE +21 -0
  4. llm_inspector-0.6.0/PKG-INFO +197 -0
  5. llm_inspector-0.6.0/README.md +159 -0
  6. llm_inspector-0.6.0/docs/DGX_GUIDE.md +342 -0
  7. llm_inspector-0.6.0/docs/INSTALL_GUIDE.md +416 -0
  8. llm_inspector-0.6.0/docs/assets/llm-inspector-demo.gif +0 -0
  9. llm_inspector-0.6.0/pyproject.toml +71 -0
  10. llm_inspector-0.6.0/src/llm_inspector/__init__.py +15 -0
  11. llm_inspector-0.6.0/src/llm_inspector/backends/__init__.py +11 -0
  12. llm_inspector-0.6.0/src/llm_inspector/backends/base.py +82 -0
  13. llm_inspector-0.6.0/src/llm_inspector/backends/cpu.py +35 -0
  14. llm_inspector-0.6.0/src/llm_inspector/backends/cuda.py +156 -0
  15. llm_inspector-0.6.0/src/llm_inspector/backends/registry.py +50 -0
  16. llm_inspector-0.6.0/src/llm_inspector/cli.py +125 -0
  17. llm_inspector-0.6.0/src/llm_inspector/collectors/__init__.py +9 -0
  18. llm_inspector-0.6.0/src/llm_inspector/collectors/base.py +107 -0
  19. llm_inspector-0.6.0/src/llm_inspector/collectors/hardware.py +130 -0
  20. llm_inspector-0.6.0/src/llm_inspector/collectors/memory.py +101 -0
  21. llm_inspector-0.6.0/src/llm_inspector/collectors/memory_breakdown.py +209 -0
  22. llm_inspector-0.6.0/src/llm_inspector/collectors/model.py +156 -0
  23. llm_inspector-0.6.0/src/llm_inspector/collectors/process.py +81 -0
  24. llm_inspector-0.6.0/src/llm_inspector/collectors/runtime.py +51 -0
  25. llm_inspector-0.6.0/src/llm_inspector/embedded/__init__.py +5 -0
  26. llm_inspector-0.6.0/src/llm_inspector/embedded/adapters/__init__.py +5 -0
  27. llm_inspector-0.6.0/src/llm_inspector/embedded/adapters/base.py +54 -0
  28. llm_inspector-0.6.0/src/llm_inspector/embedded/adapters/huggingface.py +115 -0
  29. llm_inspector-0.6.0/src/llm_inspector/embedded/adapters/pytorch.py +66 -0
  30. llm_inspector-0.6.0/src/llm_inspector/embedded/adapters/registry.py +44 -0
  31. llm_inspector-0.6.0/src/llm_inspector/embedded/adapters/vllm.py +351 -0
  32. llm_inspector-0.6.0/src/llm_inspector/embedded/attach.py +237 -0
  33. llm_inspector-0.6.0/src/llm_inspector/embedded/collectors.py +422 -0
  34. llm_inspector-0.6.0/src/llm_inspector/embedded/hooks.py +143 -0
  35. llm_inspector-0.6.0/src/llm_inspector/embedded/model_details.py +369 -0
  36. llm_inspector-0.6.0/src/llm_inspector/embedded/streaming.py +90 -0
  37. llm_inspector-0.6.0/src/llm_inspector/inspector/__init__.py +1 -0
  38. llm_inspector-0.6.0/src/llm_inspector/inspector/context.py +36 -0
  39. llm_inspector-0.6.0/src/llm_inspector/inspector/core.py +207 -0
  40. llm_inspector-0.6.0/src/llm_inspector/inspector/pipeline.py +167 -0
  41. llm_inspector-0.6.0/src/llm_inspector/models/__init__.py +7 -0
  42. llm_inspector-0.6.0/src/llm_inspector/models/enums.py +47 -0
  43. llm_inspector-0.6.0/src/llm_inspector/models/measurement.py +89 -0
  44. llm_inspector-0.6.0/src/llm_inspector/models/report.py +60 -0
  45. llm_inspector-0.6.0/src/llm_inspector/models/results.py +315 -0
  46. llm_inspector-0.6.0/src/llm_inspector/optimization/__init__.py +8 -0
  47. llm_inspector-0.6.0/src/llm_inspector/optimization/analyzer.py +96 -0
  48. llm_inspector-0.6.0/src/llm_inspector/optimization/base.py +43 -0
  49. llm_inspector-0.6.0/src/llm_inspector/optimization/models.py +84 -0
  50. llm_inspector-0.6.0/src/llm_inspector/optimization/quantization/__init__.py +1 -0
  51. llm_inspector-0.6.0/src/llm_inspector/optimization/quantization/base.py +99 -0
  52. llm_inspector-0.6.0/src/llm_inspector/optimization/quantization/methods.py +123 -0
  53. llm_inspector-0.6.0/src/llm_inspector/optimization/quantization/strategy.py +120 -0
  54. llm_inspector-0.6.0/src/llm_inspector/optimization/recommendations.py +108 -0
  55. llm_inspector-0.6.0/src/llm_inspector/plugins/__init__.py +1 -0
  56. llm_inspector-0.6.0/src/llm_inspector/plugins/base.py +96 -0
  57. llm_inspector-0.6.0/src/llm_inspector/plugins/fastapi.py +283 -0
  58. llm_inspector-0.6.0/src/llm_inspector/plugins/huggingface.py +210 -0
  59. llm_inspector-0.6.0/src/llm_inspector/plugins/ollama.py +246 -0
  60. llm_inspector-0.6.0/src/llm_inspector/plugins/registry.py +87 -0
  61. llm_inspector-0.6.0/src/llm_inspector/plugins/unknown.py +23 -0
  62. llm_inspector-0.6.0/src/llm_inspector/plugins/vllm.py +334 -0
  63. llm_inspector-0.6.0/src/llm_inspector/rpc.py +40 -0
  64. llm_inspector-0.6.0/src/llm_inspector/serialization.py +39 -0
  65. llm_inspector-0.6.0/src/llm_inspector/sources/__init__.py +7 -0
  66. llm_inspector-0.6.0/src/llm_inspector/sources/base.py +32 -0
  67. llm_inspector-0.6.0/src/llm_inspector/sources/embedded.py +56 -0
  68. llm_inspector-0.6.0/src/llm_inspector/sources/resolver.py +51 -0
  69. llm_inspector-0.6.0/src/llm_inspector/transport/__init__.py +12 -0
  70. llm_inspector-0.6.0/src/llm_inspector/transport/base.py +33 -0
  71. llm_inspector-0.6.0/src/llm_inspector/transport/discovery.py +50 -0
  72. llm_inspector-0.6.0/src/llm_inspector/transport/unix.py +194 -0
  73. llm_inspector-0.6.0/src/llm_inspector/ui/__init__.py +1 -0
  74. llm_inspector-0.6.0/src/llm_inspector/ui/format.py +86 -0
  75. llm_inspector-0.6.0/src/llm_inspector/ui/panels.py +504 -0
  76. llm_inspector-0.6.0/src/llm_inspector/ui/tables.py +82 -0
  77. llm_inspector-0.6.0/src/llm_inspector/utils/__init__.py +1 -0
  78. llm_inspector-0.6.0/src/llm_inspector/utils/cmdline.py +70 -0
  79. llm_inspector-0.6.0/src/llm_inspector/utils/http.py +112 -0
  80. llm_inspector-0.6.0/src/llm_inspector/utils/proc.py +114 -0
  81. llm_inspector-0.6.0/src/llm_inspector/utils/procfs.py +352 -0
  82. llm_inspector-0.6.0/src/llm_inspector/utils/vllm_urls.py +70 -0
  83. llm_inspector-0.6.0/tests/__init__.py +0 -0
  84. llm_inspector-0.6.0/tests/test_cmdline_utils.py +86 -0
  85. llm_inspector-0.6.0/tests/test_embedded.py +146 -0
  86. llm_inspector-0.6.0/tests/test_embedded_source.py +86 -0
  87. llm_inspector-0.6.0/tests/test_hardware_collector.py +104 -0
  88. llm_inspector-0.6.0/tests/test_http_utils.py +77 -0
  89. llm_inspector-0.6.0/tests/test_huggingface_plugin.py +128 -0
  90. llm_inspector-0.6.0/tests/test_measurement.py +40 -0
  91. llm_inspector-0.6.0/tests/test_memory_breakdown_collector.py +90 -0
  92. llm_inspector-0.6.0/tests/test_memory_collector.py +87 -0
  93. llm_inspector-0.6.0/tests/test_model_collector.py +125 -0
  94. llm_inspector-0.6.0/tests/test_model_details.py +117 -0
  95. llm_inspector-0.6.0/tests/test_ollama_plugin.py +153 -0
  96. llm_inspector-0.6.0/tests/test_optimization.py +121 -0
  97. llm_inspector-0.6.0/tests/test_phase4_breakdown.py +123 -0
  98. llm_inspector-0.6.0/tests/test_plugin_registry.py +44 -0
  99. llm_inspector-0.6.0/tests/test_process_collector.py +61 -0
  100. llm_inspector-0.6.0/tests/test_rpc.py +28 -0
  101. llm_inspector-0.6.0/tests/test_transport.py +51 -0
  102. llm_inspector-0.6.0/tests/test_vllm_adapter.py +176 -0
  103. llm_inspector-0.6.0/tests/test_vllm_plugin.py +179 -0
@@ -0,0 +1,37 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ permissions:
8
+ contents: read
9
+
10
+ jobs:
11
+ publish:
12
+ name: Build and publish
13
+ runs-on: ubuntu-latest
14
+ environment:
15
+ name: pypi
16
+ url: https://pypi.org/p/llm-inspector
17
+ permissions:
18
+ contents: read # required: job permissions replace workflow defaults
19
+ id-token: write # Trusted Publishing (OIDC) — no API token in GitHub secrets
20
+
21
+ steps:
22
+ - name: Checkout
23
+ uses: actions/checkout@v4
24
+
25
+ - name: Set up Python
26
+ uses: actions/setup-python@v5
27
+ with:
28
+ python-version: "3.12"
29
+
30
+ - name: Install build
31
+ run: python -m pip install --upgrade pip build
32
+
33
+ - name: Build sdist and wheel
34
+ run: python -m build
35
+
36
+ - name: Publish to PyPI
37
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,18 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.pyo
4
+ *.pyd
5
+ *.so
6
+ *.egg
7
+ *.egg-info/
8
+ dist/
9
+ build/
10
+ .venv/
11
+ venv/
12
+ .env
13
+ .mypy_cache/
14
+ .ruff_cache/
15
+ .pytest_cache/
16
+ .coverage
17
+ htmlcov/
18
+ *.prof
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Hela Saoudi
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,197 @@
1
+ Metadata-Version: 2.4
2
+ Name: llm-inspector
3
+ Version: 0.6.0
4
+ Summary: The htop for LLM inference. Measured. Not guessed.
5
+ Project-URL: Homepage, https://github.com/helasaoudi/llm-inspector
6
+ Project-URL: Repository, https://github.com/helasaoudi/llm-inspector
7
+ Project-URL: Bug Tracker, https://github.com/helasaoudi/llm-inspector/issues
8
+ Author-email: Hela Saoudi <helasaoudi024@gmail.com>
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: cli,gpu,inference,llm,monitoring,observability,ollama,vllm
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Environment :: Console
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: System :: Monitoring
21
+ Requires-Python: >=3.11
22
+ Requires-Dist: nvidia-ml-py>=12.535
23
+ Requires-Dist: psutil>=5.9
24
+ Requires-Dist: pydantic>=2.7
25
+ Requires-Dist: python-dateutil>=2.9
26
+ Requires-Dist: requests>=2.31
27
+ Requires-Dist: rich>=13.7
28
+ Requires-Dist: typer>=0.12
29
+ Provides-Extra: dev
30
+ Requires-Dist: mypy>=1.10; extra == 'dev'
31
+ Requires-Dist: pytest-cov>=5.0; extra == 'dev'
32
+ Requires-Dist: pytest-mock>=3.14; extra == 'dev'
33
+ Requires-Dist: pytest>=8.2; extra == 'dev'
34
+ Requires-Dist: ruff>=0.4; extra == 'dev'
35
+ Provides-Extra: torch
36
+ Requires-Dist: torch>=2.2; extra == 'torch'
37
+ Description-Content-Type: text/markdown
38
+
39
+ <div align="center">
40
+
41
+ # LLM Inspector
42
+
43
+ **The `htop` for LLM inference.**
44
+
45
+ *Measured. Not guessed.*
46
+
47
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue?logo=python&logoColor=white)](https://python.org)
48
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
49
+ [![Platforms](https://img.shields.io/badge/platform-macOS%20%7C%20Linux%20%7C%20GPU%20servers-lightgrey)](#)
50
+
51
+ **Inspect → Understand → Optimize**
52
+
53
+ </div>
54
+
55
+ ---
56
+
57
+ LLM Inspector inspects live inference processes and shows exactly how GPU memory is being used, what model is running, how the runtime is configured, and where every reported value comes from.
58
+
59
+ Unlike traditional monitoring tools, it doesn't stop at inspection. It also analyzes the running workload and projects optimization opportunities—starting with quantization—to help you understand how different strategies would impact GPU memory **before** making any changes.
60
+
61
+ ---
62
+
63
+ ## Example
64
+
65
+ <div align="center">
66
+
67
+ ![LLM Inspector demo](https://raw.githubusercontent.com/helasaoudi/llm-inspector/main/docs/assets/llm-inspector-demo.gif)
68
+
69
+ </div>
70
+
71
+ ---
72
+
73
+ ## Why?
74
+
75
+ Existing tools tell you that your GPU is using 18 GB.
76
+
77
+ LLM Inspector tells you **why**:
78
+
79
+ ```text
80
+ Weights 7 GB
81
+ KV Cache 8 GB
82
+ Workspace 1 GB
83
+ Other 2 GB
84
+ ```
85
+
86
+ Then it tells you what would happen if you optimized:
87
+
88
+ ```text
89
+ FP8 would save ~3 GB of weights.
90
+ AWQ would save ~5 GB of weights.
91
+ Weight quantization won't fix an 8 GB KV Cache bottleneck.
92
+ ```
93
+
94
+ That is the difference between a GPU monitor and an inference advisor.
95
+
96
+ ---
97
+
98
+ ## Embedded goes deeper
99
+
100
+ Most tools stop here:
101
+
102
+ ```text
103
+ GPU
104
+ └── Process A
105
+ └── 17.3 GB
106
+ ```
107
+
108
+ LLM Inspector goes one level deeper:
109
+
110
+ ```text
111
+ GPU
112
+ └── Process A
113
+ ├── Weights
114
+ ├── KV Cache
115
+ ├── Workspace
116
+ ├── Activations
117
+ └── Optimization Analysis (Projected)
118
+ ```
119
+
120
+ That bridge—**system observability + model internals**, with an `htop`-style CLI—is the innovation. Deep metrics come from optional embedded `attach()` inside the inference process (see the install guide).
121
+
122
+ ---
123
+
124
+ ## Measured vs Projected
125
+
126
+ | Section | Kind |
127
+ |---------|------|
128
+ | Process, Hardware, Model, Memory, Runtime | **Measured** from live sources (or `Unavailable` with a reason) |
129
+ | Optimization Analysis | **Projected** from measured inputs — never mutates the model |
130
+
131
+ ```bash
132
+ llminspect inspect <pid> --verbose # show provenance for every field
133
+ ```
134
+
135
+ ---
136
+
137
+ ## Install
138
+
139
+ Works on any NVIDIA GPU machine — laptop, workstation, cloud VM, bare-metal server, or DGX. Docker is optional.
140
+
141
+ ```bash
142
+ pip install llm-inspector
143
+ ```
144
+
145
+ For embedded deep metrics (Weights / KV / Activations) in the same Python env as the model:
146
+
147
+ ```bash
148
+ pip install "llm-inspector[torch]"
149
+ ```
150
+
151
+ From source (contributors):
152
+
153
+ ```bash
154
+ git clone https://github.com/helasaoudi/llm-inspector
155
+ cd llm-inspector
156
+ python -m venv .venv && source .venv/bin/activate
157
+ pip install -e ".[torch]"
158
+ ```
159
+
160
+ ---
161
+
162
+ ## Try it
163
+
164
+ ```bash
165
+ llminspect ps
166
+ llminspect inspect <pid>
167
+ llminspect inspect <pid> --verbose
168
+ ```
169
+
170
+ With Ollama on the host (no Docker):
171
+
172
+ ```bash
173
+ ollama run llama3
174
+ llminspect inspect $(pgrep -f "ollama serve") --verbose
175
+ ```
176
+
177
+ ---
178
+
179
+ ## Supports
180
+
181
+ Ollama · vLLM · HuggingFace Transformers · FastAPI · custom PyTorch
182
+ macOS · Linux · any NVIDIA GPU server (including DGX)
183
+
184
+ ---
185
+
186
+ ## Documentation
187
+
188
+ | Guide | When to read it |
189
+ |-------|-----------------|
190
+ | **[Install & Integration](docs/INSTALL_GUIDE.md)** | Host + Docker install, `attach()`, vLLM plugin, troubleshooting |
191
+ | [DGX notes](docs/DGX_GUIDE.md) | Extra detail from a DGX Spark inference-service setup |
192
+
193
+ ---
194
+
195
+ ## License
196
+
197
+ MIT
@@ -0,0 +1,159 @@
1
+ <div align="center">
2
+
3
+ # LLM Inspector
4
+
5
+ **The `htop` for LLM inference.**
6
+
7
+ *Measured. Not guessed.*
8
+
9
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue?logo=python&logoColor=white)](https://python.org)
10
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
11
+ [![Platforms](https://img.shields.io/badge/platform-macOS%20%7C%20Linux%20%7C%20GPU%20servers-lightgrey)](#)
12
+
13
+ **Inspect → Understand → Optimize**
14
+
15
+ </div>
16
+
17
+ ---
18
+
19
+ LLM Inspector inspects live inference processes and shows exactly how GPU memory is being used, what model is running, how the runtime is configured, and where every reported value comes from.
20
+
21
+ Unlike traditional monitoring tools, it doesn't stop at inspection. It also analyzes the running workload and projects optimization opportunities—starting with quantization—to help you understand how different strategies would impact GPU memory **before** making any changes.
22
+
23
+ ---
24
+
25
+ ## Example
26
+
27
+ <div align="center">
28
+
29
+ ![LLM Inspector demo](https://raw.githubusercontent.com/helasaoudi/llm-inspector/main/docs/assets/llm-inspector-demo.gif)
30
+
31
+ </div>
32
+
33
+ ---
34
+
35
+ ## Why?
36
+
37
+ Existing tools tell you that your GPU is using 18 GB.
38
+
39
+ LLM Inspector tells you **why**:
40
+
41
+ ```text
42
+ Weights 7 GB
43
+ KV Cache 8 GB
44
+ Workspace 1 GB
45
+ Other 2 GB
46
+ ```
47
+
48
+ Then it tells you what would happen if you optimized:
49
+
50
+ ```text
51
+ FP8 would save ~3 GB of weights.
52
+ AWQ would save ~5 GB of weights.
53
+ Weight quantization won't fix an 8 GB KV Cache bottleneck.
54
+ ```
55
+
56
+ That is the difference between a GPU monitor and an inference advisor.
57
+
58
+ ---
59
+
60
+ ## Embedded goes deeper
61
+
62
+ Most tools stop here:
63
+
64
+ ```text
65
+ GPU
66
+ └── Process A
67
+ └── 17.3 GB
68
+ ```
69
+
70
+ LLM Inspector goes one level deeper:
71
+
72
+ ```text
73
+ GPU
74
+ └── Process A
75
+ ├── Weights
76
+ ├── KV Cache
77
+ ├── Workspace
78
+ ├── Activations
79
+ └── Optimization Analysis (Projected)
80
+ ```
81
+
82
+ That bridge—**system observability + model internals**, with an `htop`-style CLI—is the innovation. Deep metrics come from optional embedded `attach()` inside the inference process (see the install guide).
83
+
84
+ ---
85
+
86
+ ## Measured vs Projected
87
+
88
+ | Section | Kind |
89
+ |---------|------|
90
+ | Process, Hardware, Model, Memory, Runtime | **Measured** from live sources (or `Unavailable` with a reason) |
91
+ | Optimization Analysis | **Projected** from measured inputs — never mutates the model |
92
+
93
+ ```bash
94
+ llminspect inspect <pid> --verbose # show provenance for every field
95
+ ```
96
+
97
+ ---
98
+
99
+ ## Install
100
+
101
+ Works on any NVIDIA GPU machine — laptop, workstation, cloud VM, bare-metal server, or DGX. Docker is optional.
102
+
103
+ ```bash
104
+ pip install llm-inspector
105
+ ```
106
+
107
+ For embedded deep metrics (Weights / KV / Activations) in the same Python env as the model:
108
+
109
+ ```bash
110
+ pip install "llm-inspector[torch]"
111
+ ```
112
+
113
+ From source (contributors):
114
+
115
+ ```bash
116
+ git clone https://github.com/helasaoudi/llm-inspector
117
+ cd llm-inspector
118
+ python -m venv .venv && source .venv/bin/activate
119
+ pip install -e ".[torch]"
120
+ ```
121
+
122
+ ---
123
+
124
+ ## Try it
125
+
126
+ ```bash
127
+ llminspect ps
128
+ llminspect inspect <pid>
129
+ llminspect inspect <pid> --verbose
130
+ ```
131
+
132
+ With Ollama on the host (no Docker):
133
+
134
+ ```bash
135
+ ollama run llama3
136
+ llminspect inspect $(pgrep -f "ollama serve") --verbose
137
+ ```
138
+
139
+ ---
140
+
141
+ ## Supports
142
+
143
+ Ollama · vLLM · HuggingFace Transformers · FastAPI · custom PyTorch
144
+ macOS · Linux · any NVIDIA GPU server (including DGX)
145
+
146
+ ---
147
+
148
+ ## Documentation
149
+
150
+ | Guide | When to read it |
151
+ |-------|-----------------|
152
+ | **[Install & Integration](docs/INSTALL_GUIDE.md)** | Host + Docker install, `attach()`, vLLM plugin, troubleshooting |
153
+ | [DGX notes](docs/DGX_GUIDE.md) | Extra detail from a DGX Spark inference-service setup |
154
+
155
+ ---
156
+
157
+ ## License
158
+
159
+ MIT