kaggle-vllm 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kaggle_vllm-0.1.0/LICENSE +17 -0
- kaggle_vllm-0.1.0/PKG-INFO +305 -0
- kaggle_vllm-0.1.0/README.md +279 -0
- kaggle_vllm-0.1.0/pyproject.toml +48 -0
- kaggle_vllm-0.1.0/setup.cfg +4 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/__init__.py +17 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/bootstrap.py +465 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/checksums.py +33 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/cli.py +215 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/doctor.py +105 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/download.py +122 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/environment.py +169 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/exceptions.py +37 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/installation.py +138 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/llm.py +114 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/profiles/kaggle-t4x2-cu128/overlay-lock.txt +47 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/profiles/kaggle-t4x2-cu128/overlay-requirements.txt +54 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/profiles/kaggle-t4x2-cu128/profile.json +40 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/profiles.py +117 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/runtime.py +42 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/server.py +88 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm/sharding.py +142 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/PKG-INFO +305 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/SOURCES.txt +39 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/dependency_links.txt +1 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/entry_points.txt +2 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/requires.txt +6 -0
- kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/top_level.txt +1 -0
- kaggle_vllm-0.1.0/tests/test_bootstrap.py +196 -0
- kaggle_vllm-0.1.0/tests/test_checksums.py +21 -0
- kaggle_vllm-0.1.0/tests/test_cli.py +77 -0
- kaggle_vllm-0.1.0/tests/test_doctor.py +34 -0
- kaggle_vllm-0.1.0/tests/test_download.py +68 -0
- kaggle_vllm-0.1.0/tests/test_environment.py +54 -0
- kaggle_vllm-0.1.0/tests/test_installation.py +66 -0
- kaggle_vllm-0.1.0/tests/test_integration_gpu.py +13 -0
- kaggle_vllm-0.1.0/tests/test_llm.py +120 -0
- kaggle_vllm-0.1.0/tests/test_profiles.py +22 -0
- kaggle_vllm-0.1.0/tests/test_runtime.py +19 -0
- kaggle_vllm-0.1.0/tests/test_server.py +30 -0
- kaggle_vllm-0.1.0/tests/test_sharding.py +31 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
Apache License
|
|
2
|
+
Version 2.0, January 2004
|
|
3
|
+
http://www.apache.org/licenses/
|
|
4
|
+
|
|
5
|
+
Copyright 2026 kaggle-vllm contributors
|
|
6
|
+
|
|
7
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
8
|
+
you may not use this file except in compliance with the License.
|
|
9
|
+
You may obtain a copy of the License at
|
|
10
|
+
|
|
11
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
12
|
+
|
|
13
|
+
Unless required by applicable law or agreed to in writing, software
|
|
14
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
15
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
16
|
+
See the License for the specific language governing permissions and
|
|
17
|
+
limitations under the License.
|
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: kaggle-vllm
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A lightweight Kaggle compatibility SDK around upstream vLLM for Tesla T4 GPUs.
|
|
5
|
+
Author: kaggle-vllm contributors
|
|
6
|
+
License-Expression: Apache-2.0
|
|
7
|
+
Project-URL: Documentation, https://github.com/kaggle-vllm/kaggle-vllm/tree/main/docs
|
|
8
|
+
Project-URL: Repository, https://github.com/kaggle-vllm/kaggle-vllm
|
|
9
|
+
Project-URL: Issues, https://github.com/kaggle-vllm/kaggle-vllm/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/kaggle-vllm/kaggle-vllm/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: Hugging Face binaries, https://huggingface.co/waqasm86/vllm-kaggle-binaries
|
|
12
|
+
Project-URL: Hugging Face TP=2 model, https://huggingface.co/waqasm86/vllm-kaggle-models
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Provides-Extra: test
|
|
22
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
23
|
+
Provides-Extra: hub
|
|
24
|
+
Requires-Dist: huggingface_hub>=0.36; extra == "hub"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# kaggle-vllm
|
|
28
|
+
|
|
29
|
+
`kaggle-vllm` is a lightweight Python SDK and compatibility toolkit around
|
|
30
|
+
[upstream vLLM](https://github.com/vllm-project/vllm) for Kaggle's NVIDIA Tesla
|
|
31
|
+
T4 environment. It validates the runtime, protects Kaggle's preinstalled
|
|
32
|
+
PyTorch/CUDA stack during explicit artifact staging, wraps `vllm.LLM`, inspects
|
|
33
|
+
vLLM-native persistent sharded checkpoints, and safely launches vLLM's
|
|
34
|
+
OpenAI-compatible server.
|
|
35
|
+
|
|
36
|
+
It is **not a fork, reimplementation, or replacement for vLLM**. Inference,
|
|
37
|
+
tensor parallelism, sharded-state persistence, and serving remain upstream vLLM
|
|
38
|
+
capabilities.
|
|
39
|
+
|
|
40
|
+
> **Status:** v0.1 release candidate. Functionally validated on the documented
|
|
41
|
+
> Kaggle dual-T4 environment. PyPI publication is pending Trusted Publisher
|
|
42
|
+
> configuration; it is not described as production-ready.
|
|
43
|
+
|
|
44
|
+
## Validated environment
|
|
45
|
+
|
|
46
|
+
The archived 2026-08-22/23 Kaggle runs recorded:
|
|
47
|
+
|
|
48
|
+
| Component | Validated value |
|
|
49
|
+
|---|---|
|
|
50
|
+
| Platform | Kaggle Notebook, Linux/glibc 2.35 |
|
|
51
|
+
| Python | 3.12.13 |
|
|
52
|
+
| PyTorch | 2.10.0+cu128 (preserved system install) |
|
|
53
|
+
| CUDA toolkit | 12.8.93 |
|
|
54
|
+
| Driver | 580.159.04; `nvidia-smi` CUDA capability 13.0 |
|
|
55
|
+
| GPU | 2 × NVIDIA Tesla T4, 15,360 MiB each |
|
|
56
|
+
| Compute capability | 7.5 / SM75 |
|
|
57
|
+
| NCCL | 2.27.5 |
|
|
58
|
+
| CMake / GCC | 3.31.10 / 11.4.0 |
|
|
59
|
+
| vLLM source | tag v0.18.1, commit `a26e8dc7ff2111a005144d775ecf9cebf56c45b2` |
|
|
60
|
+
|
|
61
|
+
The generated wheel is:
|
|
62
|
+
|
|
63
|
+
```text
|
|
64
|
+
vllm-0.18.2.dev0+ga26e8dc7f.d20260822.cu128-cp312-cp312-linux_x86_64.whl
|
|
65
|
+
SHA256 5a9bd710b8a19fdd23abb3442baad892da977466f996334decd533a225f5fd0c
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The source identity and distribution version are not contradictory. The source
|
|
69
|
+
checkout is upstream `v0.18.1` at the commit above; the wheel filename/version
|
|
70
|
+
is generated build metadata from vLLM's `setuptools_scm` configuration, which
|
|
71
|
+
reported the next development version plus Git/date and local CUDA metadata.
|
|
72
|
+
It does not mean the source was the upstream v0.18.2 release.
|
|
73
|
+
|
|
74
|
+
## Install the lightweight SDK
|
|
75
|
+
|
|
76
|
+
The SDK intentionally has no hard dependency on vLLM, Torch, or CUDA. Once the
|
|
77
|
+
0.1.0 distribution is published to PyPI, the primary Kaggle flow is:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install kaggle_vllm
|
|
81
|
+
kaggle-vllm bootstrap
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
The one-line form is:
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
pip install kaggle_vllm && kaggle-vllm bootstrap
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The canonical distribution spelling is equivalent:
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
python -m pip install kaggle-vllm
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
`pip install kaggle_vllm` installs only the small `kaggle-vllm` distribution;
|
|
97
|
+
Python packaging normalizes `_` and `-` in project names. The explicit
|
|
98
|
+
`bootstrap` command then downloads the exact native wheel from the Hugging Face
|
|
99
|
+
Hub/Xet-backed repository, checks its immutable revision and SHA256, stages it
|
|
100
|
+
with `pip --target --no-deps`, and creates the validated dependency overlay.
|
|
101
|
+
It never replaces or reinstalls Kaggle's Torch packages.
|
|
102
|
+
|
|
103
|
+
Importing `kaggle_vllm` never downloads or installs anything. The native wheel
|
|
104
|
+
is CPython 3.12 (`cp312`) and bootstrap rejects Python 3.11 even though the
|
|
105
|
+
lightweight SDK itself can be developed and tested with Python 3.11. Until PyPI
|
|
106
|
+
publication completes, install a locally built SDK wheel or use the immutable
|
|
107
|
+
Hugging Face SDK fallback documented in [installation](docs/installation.md).
|
|
108
|
+
|
|
109
|
+
Inspect the complete plan without network or filesystem changes:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
kaggle-vllm bootstrap --dry-run --strict
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
## Python inference API
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
from kaggle_vllm import KaggleLLM
|
|
119
|
+
|
|
120
|
+
llm = KaggleLLM(
|
|
121
|
+
model="Qwen/Qwen2.5-3B-Instruct",
|
|
122
|
+
tensor_parallel_size=2,
|
|
123
|
+
max_model_len=2048,
|
|
124
|
+
gpu_memory_utilization=0.70,
|
|
125
|
+
)
|
|
126
|
+
outputs = llm.generate(["Explain tensor parallelism."], sampling_params)
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
`KaggleLLM` lazily imports and wraps upstream `vllm.LLM`. It validates the TP
|
|
130
|
+
degree against visible GPUs and forwards advanced keyword arguments. The SDK
|
|
131
|
+
supplies the conservative settings validated on Kaggle T4 by default:
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
dtype="float16"
|
|
135
|
+
enforce_eager=True
|
|
136
|
+
disable_custom_all_reduce=True
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
These are validated conservative defaults for this Kaggle T4 configuration,
|
|
140
|
+
not claims of universal optimality. Every value can be overridden explicitly.
|
|
141
|
+
|
|
142
|
+
## Tensor parallelism is not persistent sharding
|
|
143
|
+
|
|
144
|
+
Runtime tensor parallelism partitions model execution across visible devices:
|
|
145
|
+
|
|
146
|
+
```python
|
|
147
|
+
KaggleLLM(model="Qwen/Qwen2.5-3B-Instruct", tensor_parallel_size=2)
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
A persistent TP-aware checkpoint is a different artifact. The experiment used
|
|
151
|
+
vLLM's native `save_sharded_state` machinery to write rank-specific files:
|
|
152
|
+
|
|
153
|
+
```text
|
|
154
|
+
model-rank-0-part-0.safetensors
|
|
155
|
+
model-rank-0-part-1.safetensors
|
|
156
|
+
model-rank-1-part-0.safetensors
|
|
157
|
+
model-rank-1-part-1.safetensors
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Save and inspect one through the wrapper:
|
|
161
|
+
|
|
162
|
+
```python
|
|
163
|
+
inspection = llm.save_sharded_model(
|
|
164
|
+
"/kaggle/working/qwen2.5-3b-t4x2-sharded"
|
|
165
|
+
)
|
|
166
|
+
print(inspection.rank_count) # 2
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Reload using the topology for which it was created:
|
|
170
|
+
|
|
171
|
+
```python
|
|
172
|
+
llm = KaggleLLM(
|
|
173
|
+
model="/kaggle/input/qwen2.5-3b-t4x2-sharded",
|
|
174
|
+
tensor_parallel_size=2,
|
|
175
|
+
load_format="sharded_state",
|
|
176
|
+
max_model_len=2048,
|
|
177
|
+
gpu_memory_utilization=0.70,
|
|
178
|
+
)
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
This is not arbitrary tensor splitting, uneven 1/3–2/3 GPU allocation, or a
|
|
182
|
+
claim of topology-independent portability. See [persistent sharded state](docs/sharded-state.md).
|
|
183
|
+
|
|
184
|
+
## Explicit native bootstrap and activation
|
|
185
|
+
|
|
186
|
+
Normal dependency resolution can replace Kaggle's tightly coupled Torch/CUDA
|
|
187
|
+
packages. Bootstrap uses the packaged `kaggle-t4x2-cu128` profile and pins the
|
|
188
|
+
native artifact to Hugging Face commit
|
|
189
|
+
`f6b4f10de54924ed6fe9e28cceab84eca7276ab6`:
|
|
190
|
+
|
|
191
|
+
```bash
|
|
192
|
+
kaggle-vllm bootstrap --strict
|
|
193
|
+
eval "$(kaggle-vllm env)" # optional for subsequent shell commands
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
By default it uses `/kaggle/working/vllm-staged`,
|
|
197
|
+
`/kaggle/working/vllm-runtime-overlay`, and
|
|
198
|
+
`/kaggle/working/kaggle-vllm-cache`; every path is overridable. The packaged
|
|
199
|
+
overlay lock is the exact small reproducibility input from the successful
|
|
200
|
+
Kaggle recovery. Bootstrap rejects `torch`, `torchvision`, and `torchaudio`
|
|
201
|
+
entries, writes a runtime manifest, and refuses incompatible non-empty runtime
|
|
202
|
+
directories. `KaggleLLM` may activate an already-completed default manifest,
|
|
203
|
+
but it never bootstraps implicitly. See [installation](docs/installation.md).
|
|
204
|
+
|
|
205
|
+
## Kaggle CUDA-driver discovery
|
|
206
|
+
|
|
207
|
+
The toolkit was at `/usr/local/cuda-12.8`, while the mounted live driver was
|
|
208
|
+
`/usr/local/nvidia/lib64/libcuda.so`. CMake found the toolkit but initially did
|
|
209
|
+
not expose `CUDA::cuda_driver`. The successful build made the driver directory
|
|
210
|
+
visible with:
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
export CMAKE_LIBRARY_PATH=/usr/local/nvidia/lib64
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
The source-build scripts retain this workaround. The wheel itself is excluded
|
|
217
|
+
from Git.
|
|
218
|
+
|
|
219
|
+
## OpenAI-compatible serving
|
|
220
|
+
|
|
221
|
+
The server helper creates an argument array and invokes upstream `vllm serve`
|
|
222
|
+
without a shell:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
kaggle-vllm serve /kaggle/input/qwen2.5-3b-t4x2-sharded \
|
|
226
|
+
--served-model-name qwen2.5-3b-kaggle-t4x2 \
|
|
227
|
+
--load-format sharded_state \
|
|
228
|
+
--tensor-parallel-size 2 \
|
|
229
|
+
--dtype float16 \
|
|
230
|
+
--max-model-len 2048 \
|
|
231
|
+
--gpu-memory-utilization 0.70 \
|
|
232
|
+
--host 127.0.0.1 \
|
|
233
|
+
--port 8001
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
The archived Qwen run returned HTTP 200 from both `GET /v1/models` and
|
|
237
|
+
`POST /v1/chat/completions`. See [OpenAI serving](docs/openai-serving.md).
|
|
238
|
+
|
|
239
|
+
## What was functionally validated
|
|
240
|
+
|
|
241
|
+
- CUDA-enabled vLLM wheel build and SHA256 verification
|
|
242
|
+
- staged native imports (`vllm._C`, `vllm._moe_C`, allocator)
|
|
243
|
+
- isolated dependency overlay while preserving system PyTorch
|
|
244
|
+
- single-T4 FP16 inference with `facebook/opt-125m`
|
|
245
|
+
- raw two-rank NCCL all-reduce (`3.0` on both ranks)
|
|
246
|
+
- real vLLM TP=2 inference with `facebook/opt-125m`
|
|
247
|
+
- Qwen/Qwen2.5-3B-Instruct FP16 TP=2 inference
|
|
248
|
+
- persistent TP=2 sharded-state creation and reload
|
|
249
|
+
- OpenAI-compatible TP=2 serving from the sharded Qwen checkpoint
|
|
250
|
+
|
|
251
|
+
The curated evidence is in [`artifacts/kaggle-2026-08-23`](artifacts/kaggle-2026-08-23/README.md),
|
|
252
|
+
with the larger immutable evidence and model archives kept outside Git.
|
|
253
|
+
|
|
254
|
+
## Tesla T4 / SM75 behavior
|
|
255
|
+
|
|
256
|
+
FlashAttention 2 requires compute capability 8.0 or newer and was unavailable
|
|
257
|
+
on SM75. vLLM selected `TRITON_ATTN` in the recorded runs. SymmMem communicator
|
|
258
|
+
warnings are also expected because that capability is unavailable on SM75;
|
|
259
|
+
ordinary NCCL communication and TP=2 inference still completed successfully.
|
|
260
|
+
|
|
261
|
+
## CLI
|
|
262
|
+
|
|
263
|
+
```text
|
|
264
|
+
kaggle-vllm doctor
|
|
265
|
+
kaggle-vllm fingerprint
|
|
266
|
+
kaggle-vllm bootstrap [--strict] [--dry-run]
|
|
267
|
+
kaggle-vllm env [--manifest PATH]
|
|
268
|
+
kaggle-vllm verify-gpus --tensor-parallel-size 2
|
|
269
|
+
kaggle-vllm inspect-shards PATH --json
|
|
270
|
+
kaggle-vllm verify-wheel PATH [--sha256 DIGEST]
|
|
271
|
+
kaggle-vllm stage-wheel PATH --target TARGET [--sha256 DIGEST]
|
|
272
|
+
kaggle-vllm serve MODEL ...
|
|
273
|
+
```
|
|
274
|
+
|
|
275
|
+
## Artifact distribution and security
|
|
276
|
+
|
|
277
|
+
Verified release artifacts are published separately from the source repository:
|
|
278
|
+
|
|
279
|
+
- [validated Kaggle dual-T4 vLLM wheel and metadata](https://huggingface.co/waqasm86/vllm-kaggle-binaries)
|
|
280
|
+
- [Qwen2.5-3B-Instruct TP=2 persistent sharded state](https://huggingface.co/waqasm86/vllm-kaggle-models)
|
|
281
|
+
|
|
282
|
+
Large wheels, archives, safetensors, caches, overlays, and extracted models are
|
|
283
|
+
ignored by Git. Published artifacts must carry checksums, compatibility data,
|
|
284
|
+
and upstream attribution. Never commit Kaggle, GitHub, or Hugging Face tokens.
|
|
285
|
+
The Qwen persistent checkpoint remains governed by the non-commercial Qwen
|
|
286
|
+
Research License included with the model, not this repository's Apache-2.0
|
|
287
|
+
license.
|
|
288
|
+
|
|
289
|
+
## Known limitations
|
|
290
|
+
|
|
291
|
+
- Validation is specific to the tabled Kaggle environment and CPython 3.12 ABI.
|
|
292
|
+
- The SDK supports Python 3.10+, but the published native wheel profile is
|
|
293
|
+
Linux x86_64 CPython 3.12 only.
|
|
294
|
+
- No local GPU test is claimed; GPU results come from archived Kaggle evidence.
|
|
295
|
+
- The persistent model is TP-topology-aware and validated only at TP=2.
|
|
296
|
+
- The copied upstream HF weight index names original HF shards; standard
|
|
297
|
+
Transformers loading is not supported. Use vLLM `sharded_state`.
|
|
298
|
+
- Eager execution/custom all-reduce settings were conservative correctness
|
|
299
|
+
choices, not performance benchmarks.
|
|
300
|
+
- No arbitrary or uneven GPU-memory split API is provided.
|
|
301
|
+
- Qwen redistribution/use is non-commercial under its included license.
|
|
302
|
+
|
|
303
|
+
More detail: [architecture](docs/architecture.md), [runtime](docs/kaggle-runtime.md),
|
|
304
|
+
[tensor parallelism](docs/tensor-parallel.md), [validation](docs/validation.md),
|
|
305
|
+
and the [compatibility matrix](docs/compatibility-matrix.md).
|
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
# kaggle-vllm
|
|
2
|
+
|
|
3
|
+
`kaggle-vllm` is a lightweight Python SDK and compatibility toolkit around
|
|
4
|
+
[upstream vLLM](https://github.com/vllm-project/vllm) for Kaggle's NVIDIA Tesla
|
|
5
|
+
T4 environment. It validates the runtime, protects Kaggle's preinstalled
|
|
6
|
+
PyTorch/CUDA stack during explicit artifact staging, wraps `vllm.LLM`, inspects
|
|
7
|
+
vLLM-native persistent sharded checkpoints, and safely launches vLLM's
|
|
8
|
+
OpenAI-compatible server.
|
|
9
|
+
|
|
10
|
+
It is **not a fork, reimplementation, or replacement for vLLM**. Inference,
|
|
11
|
+
tensor parallelism, sharded-state persistence, and serving remain upstream vLLM
|
|
12
|
+
capabilities.
|
|
13
|
+
|
|
14
|
+
> **Status:** v0.1 release candidate. Functionally validated on the documented
|
|
15
|
+
> Kaggle dual-T4 environment. PyPI publication is pending Trusted Publisher
|
|
16
|
+
> configuration; it is not described as production-ready.
|
|
17
|
+
|
|
18
|
+
## Validated environment
|
|
19
|
+
|
|
20
|
+
The archived 2026-08-22/23 Kaggle runs recorded:
|
|
21
|
+
|
|
22
|
+
| Component | Validated value |
|
|
23
|
+
|---|---|
|
|
24
|
+
| Platform | Kaggle Notebook, Linux/glibc 2.35 |
|
|
25
|
+
| Python | 3.12.13 |
|
|
26
|
+
| PyTorch | 2.10.0+cu128 (preserved system install) |
|
|
27
|
+
| CUDA toolkit | 12.8.93 |
|
|
28
|
+
| Driver | 580.159.04; `nvidia-smi` CUDA capability 13.0 |
|
|
29
|
+
| GPU | 2 × NVIDIA Tesla T4, 15,360 MiB each |
|
|
30
|
+
| Compute capability | 7.5 / SM75 |
|
|
31
|
+
| NCCL | 2.27.5 |
|
|
32
|
+
| CMake / GCC | 3.31.10 / 11.4.0 |
|
|
33
|
+
| vLLM source | tag v0.18.1, commit `a26e8dc7ff2111a005144d775ecf9cebf56c45b2` |
|
|
34
|
+
|
|
35
|
+
The generated wheel is:
|
|
36
|
+
|
|
37
|
+
```text
|
|
38
|
+
vllm-0.18.2.dev0+ga26e8dc7f.d20260822.cu128-cp312-cp312-linux_x86_64.whl
|
|
39
|
+
SHA256 5a9bd710b8a19fdd23abb3442baad892da977466f996334decd533a225f5fd0c
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
The source identity and distribution version are not contradictory. The source
|
|
43
|
+
checkout is upstream `v0.18.1` at the commit above; the wheel filename/version
|
|
44
|
+
is generated build metadata from vLLM's `setuptools_scm` configuration, which
|
|
45
|
+
reported the next development version plus Git/date and local CUDA metadata.
|
|
46
|
+
It does not mean the source was the upstream v0.18.2 release.
|
|
47
|
+
|
|
48
|
+
## Install the lightweight SDK
|
|
49
|
+
|
|
50
|
+
The SDK intentionally has no hard dependency on vLLM, Torch, or CUDA. Once the
|
|
51
|
+
0.1.0 distribution is published to PyPI, the primary Kaggle flow is:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
pip install kaggle_vllm
|
|
55
|
+
kaggle-vllm bootstrap
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
The one-line form is:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install kaggle_vllm && kaggle-vllm bootstrap
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
The canonical distribution spelling is equivalent:
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
python -m pip install kaggle-vllm
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
`pip install kaggle_vllm` installs only the small `kaggle-vllm` distribution;
|
|
71
|
+
Python packaging normalizes `_` and `-` in project names. The explicit
|
|
72
|
+
`bootstrap` command then downloads the exact native wheel from the Hugging Face
|
|
73
|
+
Hub/Xet-backed repository, checks its immutable revision and SHA256, stages it
|
|
74
|
+
with `pip --target --no-deps`, and creates the validated dependency overlay.
|
|
75
|
+
It never replaces or reinstalls Kaggle's Torch packages.
|
|
76
|
+
|
|
77
|
+
Importing `kaggle_vllm` never downloads or installs anything. The native wheel
|
|
78
|
+
is CPython 3.12 (`cp312`) and bootstrap rejects Python 3.11 even though the
|
|
79
|
+
lightweight SDK itself can be developed and tested with Python 3.11. Until PyPI
|
|
80
|
+
publication completes, install a locally built SDK wheel or use the immutable
|
|
81
|
+
Hugging Face SDK fallback documented in [installation](docs/installation.md).
|
|
82
|
+
|
|
83
|
+
Inspect the complete plan without network or filesystem changes:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
kaggle-vllm bootstrap --dry-run --strict
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Python inference API
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from kaggle_vllm import KaggleLLM
|
|
93
|
+
|
|
94
|
+
llm = KaggleLLM(
|
|
95
|
+
model="Qwen/Qwen2.5-3B-Instruct",
|
|
96
|
+
tensor_parallel_size=2,
|
|
97
|
+
max_model_len=2048,
|
|
98
|
+
gpu_memory_utilization=0.70,
|
|
99
|
+
)
|
|
100
|
+
outputs = llm.generate(["Explain tensor parallelism."], sampling_params)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`KaggleLLM` lazily imports and wraps upstream `vllm.LLM`. It validates the TP
|
|
104
|
+
degree against visible GPUs and forwards advanced keyword arguments. The SDK
|
|
105
|
+
supplies the conservative settings validated on Kaggle T4 by default:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
dtype="float16"
|
|
109
|
+
enforce_eager=True
|
|
110
|
+
disable_custom_all_reduce=True
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
These are validated conservative defaults for this Kaggle T4 configuration,
|
|
114
|
+
not claims of universal optimality. Every value can be overridden explicitly.
|
|
115
|
+
|
|
116
|
+
## Tensor parallelism is not persistent sharding
|
|
117
|
+
|
|
118
|
+
Runtime tensor parallelism partitions model execution across visible devices:
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
KaggleLLM(model="Qwen/Qwen2.5-3B-Instruct", tensor_parallel_size=2)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
A persistent TP-aware checkpoint is a different artifact. The experiment used
|
|
125
|
+
vLLM's native `save_sharded_state` machinery to write rank-specific files:
|
|
126
|
+
|
|
127
|
+
```text
|
|
128
|
+
model-rank-0-part-0.safetensors
|
|
129
|
+
model-rank-0-part-1.safetensors
|
|
130
|
+
model-rank-1-part-0.safetensors
|
|
131
|
+
model-rank-1-part-1.safetensors
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Save and inspect one through the wrapper:
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
inspection = llm.save_sharded_model(
|
|
138
|
+
"/kaggle/working/qwen2.5-3b-t4x2-sharded"
|
|
139
|
+
)
|
|
140
|
+
print(inspection.rank_count) # 2
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Reload using the topology for which it was created:
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
llm = KaggleLLM(
|
|
147
|
+
model="/kaggle/input/qwen2.5-3b-t4x2-sharded",
|
|
148
|
+
tensor_parallel_size=2,
|
|
149
|
+
load_format="sharded_state",
|
|
150
|
+
max_model_len=2048,
|
|
151
|
+
gpu_memory_utilization=0.70,
|
|
152
|
+
)
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
This is not arbitrary tensor splitting, uneven 1/3–2/3 GPU allocation, or a
|
|
156
|
+
claim of topology-independent portability. See [persistent sharded state](docs/sharded-state.md).
|
|
157
|
+
|
|
158
|
+
## Explicit native bootstrap and activation
|
|
159
|
+
|
|
160
|
+
Normal dependency resolution can replace Kaggle's tightly coupled Torch/CUDA
|
|
161
|
+
packages. Bootstrap uses the packaged `kaggle-t4x2-cu128` profile and pins the
|
|
162
|
+
native artifact to Hugging Face commit
|
|
163
|
+
`f6b4f10de54924ed6fe9e28cceab84eca7276ab6`:
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
kaggle-vllm bootstrap --strict
|
|
167
|
+
eval "$(kaggle-vllm env)" # optional for subsequent shell commands
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
By default it uses `/kaggle/working/vllm-staged`,
|
|
171
|
+
`/kaggle/working/vllm-runtime-overlay`, and
|
|
172
|
+
`/kaggle/working/kaggle-vllm-cache`; every path is overridable. The packaged
|
|
173
|
+
overlay lock is the exact small reproducibility input from the successful
|
|
174
|
+
Kaggle recovery. Bootstrap rejects `torch`, `torchvision`, and `torchaudio`
|
|
175
|
+
entries, writes a runtime manifest, and refuses incompatible non-empty runtime
|
|
176
|
+
directories. `KaggleLLM` may activate an already-completed default manifest,
|
|
177
|
+
but it never bootstraps implicitly. See [installation](docs/installation.md).
|
|
178
|
+
|
|
179
|
+
## Kaggle CUDA-driver discovery
|
|
180
|
+
|
|
181
|
+
The toolkit was at `/usr/local/cuda-12.8`, while the mounted live driver was
|
|
182
|
+
`/usr/local/nvidia/lib64/libcuda.so`. CMake found the toolkit but initially did
|
|
183
|
+
not expose `CUDA::cuda_driver`. The successful build made the driver directory
|
|
184
|
+
visible with:
|
|
185
|
+
|
|
186
|
+
```bash
|
|
187
|
+
export CMAKE_LIBRARY_PATH=/usr/local/nvidia/lib64
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
The source-build scripts retain this workaround. The wheel itself is excluded
|
|
191
|
+
from Git.
|
|
192
|
+
|
|
193
|
+
## OpenAI-compatible serving
|
|
194
|
+
|
|
195
|
+
The server helper creates an argument array and invokes upstream `vllm serve`
|
|
196
|
+
without a shell:
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
kaggle-vllm serve /kaggle/input/qwen2.5-3b-t4x2-sharded \
|
|
200
|
+
--served-model-name qwen2.5-3b-kaggle-t4x2 \
|
|
201
|
+
--load-format sharded_state \
|
|
202
|
+
--tensor-parallel-size 2 \
|
|
203
|
+
--dtype float16 \
|
|
204
|
+
--max-model-len 2048 \
|
|
205
|
+
--gpu-memory-utilization 0.70 \
|
|
206
|
+
--host 127.0.0.1 \
|
|
207
|
+
--port 8001
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
The archived Qwen run returned HTTP 200 from both `GET /v1/models` and
|
|
211
|
+
`POST /v1/chat/completions`. See [OpenAI serving](docs/openai-serving.md).
|
|
212
|
+
|
|
213
|
+
## What was functionally validated
|
|
214
|
+
|
|
215
|
+
- CUDA-enabled vLLM wheel build and SHA256 verification
|
|
216
|
+
- staged native imports (`vllm._C`, `vllm._moe_C`, allocator)
|
|
217
|
+
- isolated dependency overlay while preserving system PyTorch
|
|
218
|
+
- single-T4 FP16 inference with `facebook/opt-125m`
|
|
219
|
+
- raw two-rank NCCL all-reduce (`3.0` on both ranks)
|
|
220
|
+
- real vLLM TP=2 inference with `facebook/opt-125m`
|
|
221
|
+
- Qwen/Qwen2.5-3B-Instruct FP16 TP=2 inference
|
|
222
|
+
- persistent TP=2 sharded-state creation and reload
|
|
223
|
+
- OpenAI-compatible TP=2 serving from the sharded Qwen checkpoint
|
|
224
|
+
|
|
225
|
+
The curated evidence is in [`artifacts/kaggle-2026-08-23`](artifacts/kaggle-2026-08-23/README.md),
|
|
226
|
+
with the larger immutable evidence and model archives kept outside Git.
|
|
227
|
+
|
|
228
|
+
## Tesla T4 / SM75 behavior
|
|
229
|
+
|
|
230
|
+
FlashAttention 2 requires compute capability 8.0 or newer and was unavailable
|
|
231
|
+
on SM75. vLLM selected `TRITON_ATTN` in the recorded runs. SymmMem communicator
|
|
232
|
+
warnings are also expected because that capability is unavailable on SM75;
|
|
233
|
+
ordinary NCCL communication and TP=2 inference still completed successfully.
|
|
234
|
+
|
|
235
|
+
## CLI
|
|
236
|
+
|
|
237
|
+
```text
|
|
238
|
+
kaggle-vllm doctor
|
|
239
|
+
kaggle-vllm fingerprint
|
|
240
|
+
kaggle-vllm bootstrap [--strict] [--dry-run]
|
|
241
|
+
kaggle-vllm env [--manifest PATH]
|
|
242
|
+
kaggle-vllm verify-gpus --tensor-parallel-size 2
|
|
243
|
+
kaggle-vllm inspect-shards PATH --json
|
|
244
|
+
kaggle-vllm verify-wheel PATH [--sha256 DIGEST]
|
|
245
|
+
kaggle-vllm stage-wheel PATH --target TARGET [--sha256 DIGEST]
|
|
246
|
+
kaggle-vllm serve MODEL ...
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
## Artifact distribution and security
|
|
250
|
+
|
|
251
|
+
Verified release artifacts are published separately from the source repository:
|
|
252
|
+
|
|
253
|
+
- [validated Kaggle dual-T4 vLLM wheel and metadata](https://huggingface.co/waqasm86/vllm-kaggle-binaries)
|
|
254
|
+
- [Qwen2.5-3B-Instruct TP=2 persistent sharded state](https://huggingface.co/waqasm86/vllm-kaggle-models)
|
|
255
|
+
|
|
256
|
+
Large wheels, archives, safetensors, caches, overlays, and extracted models are
|
|
257
|
+
ignored by Git. Published artifacts must carry checksums, compatibility data,
|
|
258
|
+
and upstream attribution. Never commit Kaggle, GitHub, or Hugging Face tokens.
|
|
259
|
+
The Qwen persistent checkpoint remains governed by the non-commercial Qwen
|
|
260
|
+
Research License included with the model, not this repository's Apache-2.0
|
|
261
|
+
license.
|
|
262
|
+
|
|
263
|
+
## Known limitations
|
|
264
|
+
|
|
265
|
+
- Validation is specific to the tabled Kaggle environment and CPython 3.12 ABI.
|
|
266
|
+
- The SDK supports Python 3.10+, but the published native wheel profile is
|
|
267
|
+
Linux x86_64 CPython 3.12 only.
|
|
268
|
+
- No local GPU test is claimed; GPU results come from archived Kaggle evidence.
|
|
269
|
+
- The persistent model is TP-topology-aware and validated only at TP=2.
|
|
270
|
+
- The copied upstream HF weight index names original HF shards; standard
|
|
271
|
+
Transformers loading is not supported. Use vLLM `sharded_state`.
|
|
272
|
+
- Eager execution/custom all-reduce settings were conservative correctness
|
|
273
|
+
choices, not performance benchmarks.
|
|
274
|
+
- No arbitrary or uneven GPU-memory split API is provided.
|
|
275
|
+
- Qwen redistribution/use is non-commercial under its included license.
|
|
276
|
+
|
|
277
|
+
More detail: [architecture](docs/architecture.md), [runtime](docs/kaggle-runtime.md),
|
|
278
|
+
[tensor parallelism](docs/tensor-parallel.md), [validation](docs/validation.md),
|
|
279
|
+
and the [compatibility matrix](docs/compatibility-matrix.md).
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "kaggle-vllm"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "A lightweight Kaggle compatibility SDK around upstream vLLM for Tesla T4 GPUs."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "Apache-2.0"
|
|
12
|
+
authors = [{name = "kaggle-vllm contributors"}]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 3 - Alpha",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.10",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
]
|
|
20
|
+
dependencies = []
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
test = ["pytest>=8"]
|
|
24
|
+
hub = ["huggingface_hub>=0.36"]
|
|
25
|
+
|
|
26
|
+
[project.urls]
|
|
27
|
+
Documentation = "https://github.com/kaggle-vllm/kaggle-vllm/tree/main/docs"
|
|
28
|
+
Repository = "https://github.com/kaggle-vllm/kaggle-vllm"
|
|
29
|
+
Issues = "https://github.com/kaggle-vllm/kaggle-vllm/issues"
|
|
30
|
+
Changelog = "https://github.com/kaggle-vllm/kaggle-vllm/blob/main/CHANGELOG.md"
|
|
31
|
+
"Hugging Face binaries" = "https://huggingface.co/waqasm86/vllm-kaggle-binaries"
|
|
32
|
+
"Hugging Face TP=2 model" = "https://huggingface.co/waqasm86/vllm-kaggle-models"
|
|
33
|
+
|
|
34
|
+
[project.scripts]
|
|
35
|
+
kaggle-vllm = "kaggle_vllm.cli:main"
|
|
36
|
+
|
|
37
|
+
[tool.setuptools.packages.find]
|
|
38
|
+
where = ["src"]
|
|
39
|
+
|
|
40
|
+
[tool.setuptools.package-data]
|
|
41
|
+
kaggle_vllm = ["profiles/*/*.json", "profiles/*/*.txt"]
|
|
42
|
+
|
|
43
|
+
[tool.pytest.ini_options]
|
|
44
|
+
testpaths = ["tests"]
|
|
45
|
+
markers = [
|
|
46
|
+
"gpu: requires a real CUDA GPU runtime",
|
|
47
|
+
"kaggle: requires the documented Kaggle environment",
|
|
48
|
+
]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Kaggle compatibility helpers around upstream vLLM."""
|
|
2
|
+
|
|
3
|
+
from .bootstrap import activate_runtime, bootstrap
|
|
4
|
+
from .llm import KaggleLLM
|
|
5
|
+
from .profiles import BootstrapProfile, load_profile
|
|
6
|
+
from .sharding import ShardedModelInspection, inspect_sharded_model
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"BootstrapProfile",
|
|
10
|
+
"KaggleLLM",
|
|
11
|
+
"ShardedModelInspection",
|
|
12
|
+
"activate_runtime",
|
|
13
|
+
"bootstrap",
|
|
14
|
+
"inspect_sharded_model",
|
|
15
|
+
"load_profile",
|
|
16
|
+
]
|
|
17
|
+
__version__ = "0.1.0"
|