kaggle-vllm 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. kaggle_vllm-0.1.0/LICENSE +17 -0
  2. kaggle_vllm-0.1.0/PKG-INFO +305 -0
  3. kaggle_vllm-0.1.0/README.md +279 -0
  4. kaggle_vllm-0.1.0/pyproject.toml +48 -0
  5. kaggle_vllm-0.1.0/setup.cfg +4 -0
  6. kaggle_vllm-0.1.0/src/kaggle_vllm/__init__.py +17 -0
  7. kaggle_vllm-0.1.0/src/kaggle_vllm/bootstrap.py +465 -0
  8. kaggle_vllm-0.1.0/src/kaggle_vllm/checksums.py +33 -0
  9. kaggle_vllm-0.1.0/src/kaggle_vllm/cli.py +215 -0
  10. kaggle_vllm-0.1.0/src/kaggle_vllm/doctor.py +105 -0
  11. kaggle_vllm-0.1.0/src/kaggle_vllm/download.py +122 -0
  12. kaggle_vllm-0.1.0/src/kaggle_vllm/environment.py +169 -0
  13. kaggle_vllm-0.1.0/src/kaggle_vllm/exceptions.py +37 -0
  14. kaggle_vllm-0.1.0/src/kaggle_vllm/installation.py +138 -0
  15. kaggle_vllm-0.1.0/src/kaggle_vllm/llm.py +114 -0
  16. kaggle_vllm-0.1.0/src/kaggle_vllm/profiles/kaggle-t4x2-cu128/overlay-lock.txt +47 -0
  17. kaggle_vllm-0.1.0/src/kaggle_vllm/profiles/kaggle-t4x2-cu128/overlay-requirements.txt +54 -0
  18. kaggle_vllm-0.1.0/src/kaggle_vllm/profiles/kaggle-t4x2-cu128/profile.json +40 -0
  19. kaggle_vllm-0.1.0/src/kaggle_vllm/profiles.py +117 -0
  20. kaggle_vllm-0.1.0/src/kaggle_vllm/runtime.py +42 -0
  21. kaggle_vllm-0.1.0/src/kaggle_vllm/server.py +88 -0
  22. kaggle_vllm-0.1.0/src/kaggle_vllm/sharding.py +142 -0
  23. kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/PKG-INFO +305 -0
  24. kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/SOURCES.txt +39 -0
  25. kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/dependency_links.txt +1 -0
  26. kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/entry_points.txt +2 -0
  27. kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/requires.txt +6 -0
  28. kaggle_vllm-0.1.0/src/kaggle_vllm.egg-info/top_level.txt +1 -0
  29. kaggle_vllm-0.1.0/tests/test_bootstrap.py +196 -0
  30. kaggle_vllm-0.1.0/tests/test_checksums.py +21 -0
  31. kaggle_vllm-0.1.0/tests/test_cli.py +77 -0
  32. kaggle_vllm-0.1.0/tests/test_doctor.py +34 -0
  33. kaggle_vllm-0.1.0/tests/test_download.py +68 -0
  34. kaggle_vllm-0.1.0/tests/test_environment.py +54 -0
  35. kaggle_vllm-0.1.0/tests/test_installation.py +66 -0
  36. kaggle_vllm-0.1.0/tests/test_integration_gpu.py +13 -0
  37. kaggle_vllm-0.1.0/tests/test_llm.py +120 -0
  38. kaggle_vllm-0.1.0/tests/test_profiles.py +22 -0
  39. kaggle_vllm-0.1.0/tests/test_runtime.py +19 -0
  40. kaggle_vllm-0.1.0/tests/test_server.py +30 -0
  41. kaggle_vllm-0.1.0/tests/test_sharding.py +31 -0
@@ -0,0 +1,17 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ Copyright 2026 kaggle-vllm contributors
6
+
7
+ Licensed under the Apache License, Version 2.0 (the "License");
8
+ you may not use this file except in compliance with the License.
9
+ You may obtain a copy of the License at
10
+
11
+ http://www.apache.org/licenses/LICENSE-2.0
12
+
13
+ Unless required by applicable law or agreed to in writing, software
14
+ distributed under the License is distributed on an "AS IS" BASIS,
15
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
16
+ See the License for the specific language governing permissions and
17
+ limitations under the License.
@@ -0,0 +1,305 @@
1
+ Metadata-Version: 2.4
2
+ Name: kaggle-vllm
3
+ Version: 0.1.0
4
+ Summary: A lightweight Kaggle compatibility SDK around upstream vLLM for Tesla T4 GPUs.
5
+ Author: kaggle-vllm contributors
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Documentation, https://github.com/kaggle-vllm/kaggle-vllm/tree/main/docs
8
+ Project-URL: Repository, https://github.com/kaggle-vllm/kaggle-vllm
9
+ Project-URL: Issues, https://github.com/kaggle-vllm/kaggle-vllm/issues
10
+ Project-URL: Changelog, https://github.com/kaggle-vllm/kaggle-vllm/blob/main/CHANGELOG.md
11
+ Project-URL: Hugging Face binaries, https://huggingface.co/waqasm86/vllm-kaggle-binaries
12
+ Project-URL: Hugging Face TP=2 model, https://huggingface.co/waqasm86/vllm-kaggle-models
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Provides-Extra: test
22
+ Requires-Dist: pytest>=8; extra == "test"
23
+ Provides-Extra: hub
24
+ Requires-Dist: huggingface_hub>=0.36; extra == "hub"
25
+ Dynamic: license-file
26
+
27
+ # kaggle-vllm
28
+
29
+ `kaggle-vllm` is a lightweight Python SDK and compatibility toolkit around
30
+ [upstream vLLM](https://github.com/vllm-project/vllm) for Kaggle's NVIDIA Tesla
31
+ T4 environment. It validates the runtime, protects Kaggle's preinstalled
32
+ PyTorch/CUDA stack during explicit artifact staging, wraps `vllm.LLM`, inspects
33
+ vLLM-native persistent sharded checkpoints, and safely launches vLLM's
34
+ OpenAI-compatible server.
35
+
36
+ It is **not a fork, reimplementation, or replacement for vLLM**. Inference,
37
+ tensor parallelism, sharded-state persistence, and serving remain upstream vLLM
38
+ capabilities.
39
+
40
+ > **Status:** v0.1 release candidate. Functionally validated on the documented
41
+ > Kaggle dual-T4 environment. PyPI publication is pending Trusted Publisher
42
+ > configuration; it is not described as production-ready.
43
+
44
+ ## Validated environment
45
+
46
+ The archived 2026-08-22/23 Kaggle runs recorded:
47
+
48
+ | Component | Validated value |
49
+ |---|---|
50
+ | Platform | Kaggle Notebook, Linux/glibc 2.35 |
51
+ | Python | 3.12.13 |
52
+ | PyTorch | 2.10.0+cu128 (preserved system install) |
53
+ | CUDA toolkit | 12.8.93 |
54
+ | Driver | 580.159.04; `nvidia-smi` CUDA capability 13.0 |
55
+ | GPU | 2 × NVIDIA Tesla T4, 15,360 MiB each |
56
+ | Compute capability | 7.5 / SM75 |
57
+ | NCCL | 2.27.5 |
58
+ | CMake / GCC | 3.31.10 / 11.4.0 |
59
+ | vLLM source | tag v0.18.1, commit `a26e8dc7ff2111a005144d775ecf9cebf56c45b2` |
60
+
61
+ The generated wheel is:
62
+
63
+ ```text
64
+ vllm-0.18.2.dev0+ga26e8dc7f.d20260822.cu128-cp312-cp312-linux_x86_64.whl
65
+ SHA256 5a9bd710b8a19fdd23abb3442baad892da977466f996334decd533a225f5fd0c
66
+ ```
67
+
68
+ The source identity and distribution version are not contradictory. The source
69
+ checkout is upstream `v0.18.1` at the commit above; the wheel filename/version
70
+ is generated build metadata from vLLM's `setuptools_scm` configuration, which
71
+ reported the next development version plus Git/date and local CUDA metadata.
72
+ It does not mean the source was the upstream v0.18.2 release.
73
+
74
+ ## Install the lightweight SDK
75
+
76
+ The SDK intentionally has no hard dependency on vLLM, Torch, or CUDA. Once the
77
+ 0.1.0 distribution is published to PyPI, the primary Kaggle flow is:
78
+
79
+ ```bash
80
+ pip install kaggle_vllm
81
+ kaggle-vllm bootstrap
82
+ ```
83
+
84
+ The one-line form is:
85
+
86
+ ```bash
87
+ pip install kaggle_vllm && kaggle-vllm bootstrap
88
+ ```
89
+
90
+ The canonical distribution spelling is equivalent:
91
+
92
+ ```bash
93
+ python -m pip install kaggle-vllm
94
+ ```
95
+
96
+ `pip install kaggle_vllm` installs only the small `kaggle-vllm` distribution;
97
+ Python packaging normalizes `_` and `-` in project names. The explicit
98
+ `bootstrap` command then downloads the exact native wheel from the Hugging Face
99
+ Hub/Xet-backed repository, checks its immutable revision and SHA256, stages it
100
+ with `pip --target --no-deps`, and creates the validated dependency overlay.
101
+ It never replaces or reinstalls Kaggle's Torch packages.
102
+
103
+ Importing `kaggle_vllm` never downloads or installs anything. The native wheel
104
+ is CPython 3.12 (`cp312`) and bootstrap rejects Python 3.11 even though the
105
+ lightweight SDK itself can be developed and tested with Python 3.11. Until PyPI
106
+ publication completes, install a locally built SDK wheel or use the immutable
107
+ Hugging Face SDK fallback documented in [installation](docs/installation.md).
108
+
109
+ Inspect the complete plan without network or filesystem changes:
110
+
111
+ ```bash
112
+ kaggle-vllm bootstrap --dry-run --strict
113
+ ```
114
+
115
+ ## Python inference API
116
+
117
+ ```python
118
+ from kaggle_vllm import KaggleLLM
119
+
120
+ llm = KaggleLLM(
121
+ model="Qwen/Qwen2.5-3B-Instruct",
122
+ tensor_parallel_size=2,
123
+ max_model_len=2048,
124
+ gpu_memory_utilization=0.70,
125
+ )
126
+ outputs = llm.generate(["Explain tensor parallelism."], sampling_params)
127
+ ```
128
+
129
+ `KaggleLLM` lazily imports and wraps upstream `vllm.LLM`. It validates the TP
130
+ degree against visible GPUs and forwards advanced keyword arguments. The SDK
131
+ supplies the conservative settings validated on Kaggle T4 by default:
132
+
133
+ ```python
134
+ dtype="float16"
135
+ enforce_eager=True
136
+ disable_custom_all_reduce=True
137
+ ```
138
+
139
+ These are validated conservative defaults for this Kaggle T4 configuration,
140
+ not claims of universal optimality. Every value can be overridden explicitly.
141
+
142
+ ## Tensor parallelism is not persistent sharding
143
+
144
+ Runtime tensor parallelism partitions model execution across visible devices:
145
+
146
+ ```python
147
+ KaggleLLM(model="Qwen/Qwen2.5-3B-Instruct", tensor_parallel_size=2)
148
+ ```
149
+
150
+ A persistent TP-aware checkpoint is a different artifact. The experiment used
151
+ vLLM's native `save_sharded_state` machinery to write rank-specific files:
152
+
153
+ ```text
154
+ model-rank-0-part-0.safetensors
155
+ model-rank-0-part-1.safetensors
156
+ model-rank-1-part-0.safetensors
157
+ model-rank-1-part-1.safetensors
158
+ ```
159
+
160
+ Save and inspect one through the wrapper:
161
+
162
+ ```python
163
+ inspection = llm.save_sharded_model(
164
+ "/kaggle/working/qwen2.5-3b-t4x2-sharded"
165
+ )
166
+ print(inspection.rank_count) # 2
167
+ ```
168
+
169
+ Reload using the topology for which it was created:
170
+
171
+ ```python
172
+ llm = KaggleLLM(
173
+ model="/kaggle/input/qwen2.5-3b-t4x2-sharded",
174
+ tensor_parallel_size=2,
175
+ load_format="sharded_state",
176
+ max_model_len=2048,
177
+ gpu_memory_utilization=0.70,
178
+ )
179
+ ```
180
+
181
+ This is not arbitrary tensor splitting, uneven 1/3–2/3 GPU allocation, or a
182
+ claim of topology-independent portability. See [persistent sharded state](docs/sharded-state.md).
183
+
184
+ ## Explicit native bootstrap and activation
185
+
186
+ Normal dependency resolution can replace Kaggle's tightly coupled Torch/CUDA
187
+ packages. Bootstrap uses the packaged `kaggle-t4x2-cu128` profile and pins the
188
+ native artifact to Hugging Face commit
189
+ `f6b4f10de54924ed6fe9e28cceab84eca7276ab6`:
190
+
191
+ ```bash
192
+ kaggle-vllm bootstrap --strict
193
+ eval "$(kaggle-vllm env)" # optional for subsequent shell commands
194
+ ```
195
+
196
+ By default it uses `/kaggle/working/vllm-staged`,
197
+ `/kaggle/working/vllm-runtime-overlay`, and
198
+ `/kaggle/working/kaggle-vllm-cache`; every path is overridable. The packaged
199
+ overlay lock is the exact small reproducibility input from the successful
200
+ Kaggle recovery. Bootstrap rejects `torch`, `torchvision`, and `torchaudio`
201
+ entries, writes a runtime manifest, and refuses incompatible non-empty runtime
202
+ directories. `KaggleLLM` may activate an already-completed default manifest,
203
+ but it never bootstraps implicitly. See [installation](docs/installation.md).
204
+
205
+ ## Kaggle CUDA-driver discovery
206
+
207
+ The toolkit was at `/usr/local/cuda-12.8`, while the mounted live driver was
208
+ `/usr/local/nvidia/lib64/libcuda.so`. CMake found the toolkit but initially did
209
+ not expose `CUDA::cuda_driver`. The successful build made the driver directory
210
+ visible with:
211
+
212
+ ```bash
213
+ export CMAKE_LIBRARY_PATH=/usr/local/nvidia/lib64
214
+ ```
215
+
216
+ The source-build scripts retain this workaround. The wheel itself is excluded
217
+ from Git.
218
+
219
+ ## OpenAI-compatible serving
220
+
221
+ The server helper creates an argument array and invokes upstream `vllm serve`
222
+ without a shell:
223
+
224
+ ```bash
225
+ kaggle-vllm serve /kaggle/input/qwen2.5-3b-t4x2-sharded \
226
+ --served-model-name qwen2.5-3b-kaggle-t4x2 \
227
+ --load-format sharded_state \
228
+ --tensor-parallel-size 2 \
229
+ --dtype float16 \
230
+ --max-model-len 2048 \
231
+ --gpu-memory-utilization 0.70 \
232
+ --host 127.0.0.1 \
233
+ --port 8001
234
+ ```
235
+
236
+ The archived Qwen run returned HTTP 200 from both `GET /v1/models` and
237
+ `POST /v1/chat/completions`. See [OpenAI serving](docs/openai-serving.md).
238
+
239
+ ## What was functionally validated
240
+
241
+ - CUDA-enabled vLLM wheel build and SHA256 verification
242
+ - staged native imports (`vllm._C`, `vllm._moe_C`, allocator)
243
+ - isolated dependency overlay while preserving system PyTorch
244
+ - single-T4 FP16 inference with `facebook/opt-125m`
245
+ - raw two-rank NCCL all-reduce (`3.0` on both ranks)
246
+ - real vLLM TP=2 inference with `facebook/opt-125m`
247
+ - Qwen/Qwen2.5-3B-Instruct FP16 TP=2 inference
248
+ - persistent TP=2 sharded-state creation and reload
249
+ - OpenAI-compatible TP=2 serving from the sharded Qwen checkpoint
250
+
251
+ The curated evidence is in [`artifacts/kaggle-2026-08-23`](artifacts/kaggle-2026-08-23/README.md),
252
+ with the larger immutable evidence and model archives kept outside Git.
253
+
254
+ ## Tesla T4 / SM75 behavior
255
+
256
+ FlashAttention 2 requires compute capability 8.0 or newer and was unavailable
257
+ on SM75. vLLM selected `TRITON_ATTN` in the recorded runs. SymmMem communicator
258
+ warnings are also expected because that capability is unavailable on SM75;
259
+ ordinary NCCL communication and TP=2 inference still completed successfully.
260
+
261
+ ## CLI
262
+
263
+ ```text
264
+ kaggle-vllm doctor
265
+ kaggle-vllm fingerprint
266
+ kaggle-vllm bootstrap [--strict] [--dry-run]
267
+ kaggle-vllm env [--manifest PATH]
268
+ kaggle-vllm verify-gpus --tensor-parallel-size 2
269
+ kaggle-vllm inspect-shards PATH --json
270
+ kaggle-vllm verify-wheel PATH [--sha256 DIGEST]
271
+ kaggle-vllm stage-wheel PATH --target TARGET [--sha256 DIGEST]
272
+ kaggle-vllm serve MODEL ...
273
+ ```
274
+
275
+ ## Artifact distribution and security
276
+
277
+ Verified release artifacts are published separately from the source repository:
278
+
279
+ - [validated Kaggle dual-T4 vLLM wheel and metadata](https://huggingface.co/waqasm86/vllm-kaggle-binaries)
280
+ - [Qwen2.5-3B-Instruct TP=2 persistent sharded state](https://huggingface.co/waqasm86/vllm-kaggle-models)
281
+
282
+ Large wheels, archives, safetensors, caches, overlays, and extracted models are
283
+ ignored by Git. Published artifacts must carry checksums, compatibility data,
284
+ and upstream attribution. Never commit Kaggle, GitHub, or Hugging Face tokens.
285
+ The Qwen persistent checkpoint remains governed by the non-commercial Qwen
286
+ Research License included with the model, not this repository's Apache-2.0
287
+ license.
288
+
289
+ ## Known limitations
290
+
291
+ - Validation is specific to the tabled Kaggle environment and CPython 3.12 ABI.
292
+ - The SDK supports Python 3.10+, but the published native wheel profile is
293
+ Linux x86_64 CPython 3.12 only.
294
+ - No local GPU test is claimed; GPU results come from archived Kaggle evidence.
295
+ - The persistent model is TP-topology-aware and validated only at TP=2.
296
+ - The copied upstream HF weight index names original HF shards; standard
297
+ Transformers loading is not supported. Use vLLM `sharded_state`.
298
+ - Eager execution/custom all-reduce settings were conservative correctness
299
+ choices, not performance benchmarks.
300
+ - No arbitrary or uneven GPU-memory split API is provided.
301
+ - Qwen redistribution/use is non-commercial under its included license.
302
+
303
+ More detail: [architecture](docs/architecture.md), [runtime](docs/kaggle-runtime.md),
304
+ [tensor parallelism](docs/tensor-parallel.md), [validation](docs/validation.md),
305
+ and the [compatibility matrix](docs/compatibility-matrix.md).
@@ -0,0 +1,279 @@
1
+ # kaggle-vllm
2
+
3
+ `kaggle-vllm` is a lightweight Python SDK and compatibility toolkit around
4
+ [upstream vLLM](https://github.com/vllm-project/vllm) for Kaggle's NVIDIA Tesla
5
+ T4 environment. It validates the runtime, protects Kaggle's preinstalled
6
+ PyTorch/CUDA stack during explicit artifact staging, wraps `vllm.LLM`, inspects
7
+ vLLM-native persistent sharded checkpoints, and safely launches vLLM's
8
+ OpenAI-compatible server.
9
+
10
+ It is **not a fork, reimplementation, or replacement for vLLM**. Inference,
11
+ tensor parallelism, sharded-state persistence, and serving remain upstream vLLM
12
+ capabilities.
13
+
14
+ > **Status:** v0.1 release candidate. Functionally validated on the documented
15
+ > Kaggle dual-T4 environment. PyPI publication is pending Trusted Publisher
16
+ > configuration; it is not described as production-ready.
17
+
18
+ ## Validated environment
19
+
20
+ The archived 2026-08-22/23 Kaggle runs recorded:
21
+
22
+ | Component | Validated value |
23
+ |---|---|
24
+ | Platform | Kaggle Notebook, Linux/glibc 2.35 |
25
+ | Python | 3.12.13 |
26
+ | PyTorch | 2.10.0+cu128 (preserved system install) |
27
+ | CUDA toolkit | 12.8.93 |
28
+ | Driver | 580.159.04; `nvidia-smi` CUDA capability 13.0 |
29
+ | GPU | 2 × NVIDIA Tesla T4, 15,360 MiB each |
30
+ | Compute capability | 7.5 / SM75 |
31
+ | NCCL | 2.27.5 |
32
+ | CMake / GCC | 3.31.10 / 11.4.0 |
33
+ | vLLM source | tag v0.18.1, commit `a26e8dc7ff2111a005144d775ecf9cebf56c45b2` |
34
+
35
+ The generated wheel is:
36
+
37
+ ```text
38
+ vllm-0.18.2.dev0+ga26e8dc7f.d20260822.cu128-cp312-cp312-linux_x86_64.whl
39
+ SHA256 5a9bd710b8a19fdd23abb3442baad892da977466f996334decd533a225f5fd0c
40
+ ```
41
+
42
+ The source identity and distribution version are not contradictory. The source
43
+ checkout is upstream `v0.18.1` at the commit above; the wheel filename/version
44
+ is generated build metadata from vLLM's `setuptools_scm` configuration, which
45
+ reported the next development version plus Git/date and local CUDA metadata.
46
+ It does not mean the source was the upstream v0.18.2 release.
47
+
48
+ ## Install the lightweight SDK
49
+
50
+ The SDK intentionally has no hard dependency on vLLM, Torch, or CUDA. Once the
51
+ 0.1.0 distribution is published to PyPI, the primary Kaggle flow is:
52
+
53
+ ```bash
54
+ pip install kaggle_vllm
55
+ kaggle-vllm bootstrap
56
+ ```
57
+
58
+ The one-line form is:
59
+
60
+ ```bash
61
+ pip install kaggle_vllm && kaggle-vllm bootstrap
62
+ ```
63
+
64
+ The canonical distribution spelling is equivalent:
65
+
66
+ ```bash
67
+ python -m pip install kaggle-vllm
68
+ ```
69
+
70
+ `pip install kaggle_vllm` installs only the small `kaggle-vllm` distribution;
71
+ Python packaging normalizes `_` and `-` in project names. The explicit
72
+ `bootstrap` command then downloads the exact native wheel from the Hugging Face
73
+ Hub/Xet-backed repository, checks its immutable revision and SHA256, stages it
74
+ with `pip --target --no-deps`, and creates the validated dependency overlay.
75
+ It never replaces or reinstalls Kaggle's Torch packages.
76
+
77
+ Importing `kaggle_vllm` never downloads or installs anything. The native wheel
78
+ is CPython 3.12 (`cp312`) and bootstrap rejects Python 3.11 even though the
79
+ lightweight SDK itself can be developed and tested with Python 3.11. Until PyPI
80
+ publication completes, install a locally built SDK wheel or use the immutable
81
+ Hugging Face SDK fallback documented in [installation](docs/installation.md).
82
+
83
+ Inspect the complete plan without network or filesystem changes:
84
+
85
+ ```bash
86
+ kaggle-vllm bootstrap --dry-run --strict
87
+ ```
88
+
89
+ ## Python inference API
90
+
91
+ ```python
92
+ from kaggle_vllm import KaggleLLM
93
+
94
+ llm = KaggleLLM(
95
+ model="Qwen/Qwen2.5-3B-Instruct",
96
+ tensor_parallel_size=2,
97
+ max_model_len=2048,
98
+ gpu_memory_utilization=0.70,
99
+ )
100
+ outputs = llm.generate(["Explain tensor parallelism."], sampling_params)
101
+ ```
102
+
103
+ `KaggleLLM` lazily imports and wraps upstream `vllm.LLM`. It validates the TP
104
+ degree against visible GPUs and forwards advanced keyword arguments. The SDK
105
+ supplies the conservative settings validated on Kaggle T4 by default:
106
+
107
+ ```python
108
+ dtype="float16"
109
+ enforce_eager=True
110
+ disable_custom_all_reduce=True
111
+ ```
112
+
113
+ These are validated conservative defaults for this Kaggle T4 configuration,
114
+ not claims of universal optimality. Every value can be overridden explicitly.
115
+
116
+ ## Tensor parallelism is not persistent sharding
117
+
118
+ Runtime tensor parallelism partitions model execution across visible devices:
119
+
120
+ ```python
121
+ KaggleLLM(model="Qwen/Qwen2.5-3B-Instruct", tensor_parallel_size=2)
122
+ ```
123
+
124
+ A persistent TP-aware checkpoint is a different artifact. The experiment used
125
+ vLLM's native `save_sharded_state` machinery to write rank-specific files:
126
+
127
+ ```text
128
+ model-rank-0-part-0.safetensors
129
+ model-rank-0-part-1.safetensors
130
+ model-rank-1-part-0.safetensors
131
+ model-rank-1-part-1.safetensors
132
+ ```
133
+
134
+ Save and inspect one through the wrapper:
135
+
136
+ ```python
137
+ inspection = llm.save_sharded_model(
138
+ "/kaggle/working/qwen2.5-3b-t4x2-sharded"
139
+ )
140
+ print(inspection.rank_count) # 2
141
+ ```
142
+
143
+ Reload using the topology for which it was created:
144
+
145
+ ```python
146
+ llm = KaggleLLM(
147
+ model="/kaggle/input/qwen2.5-3b-t4x2-sharded",
148
+ tensor_parallel_size=2,
149
+ load_format="sharded_state",
150
+ max_model_len=2048,
151
+ gpu_memory_utilization=0.70,
152
+ )
153
+ ```
154
+
155
+ This is not arbitrary tensor splitting, uneven 1/3–2/3 GPU allocation, or a
156
+ claim of topology-independent portability. See [persistent sharded state](docs/sharded-state.md).
157
+
158
+ ## Explicit native bootstrap and activation
159
+
160
+ Normal dependency resolution can replace Kaggle's tightly coupled Torch/CUDA
161
+ packages. Bootstrap uses the packaged `kaggle-t4x2-cu128` profile and pins the
162
+ native artifact to Hugging Face commit
163
+ `f6b4f10de54924ed6fe9e28cceab84eca7276ab6`:
164
+
165
+ ```bash
166
+ kaggle-vllm bootstrap --strict
167
+ eval "$(kaggle-vllm env)" # optional for subsequent shell commands
168
+ ```
169
+
170
+ By default it uses `/kaggle/working/vllm-staged`,
171
+ `/kaggle/working/vllm-runtime-overlay`, and
172
+ `/kaggle/working/kaggle-vllm-cache`; every path is overridable. The packaged
173
+ overlay lock is the exact small reproducibility input from the successful
174
+ Kaggle recovery. Bootstrap rejects `torch`, `torchvision`, and `torchaudio`
175
+ entries, writes a runtime manifest, and refuses incompatible non-empty runtime
176
+ directories. `KaggleLLM` may activate an already-completed default manifest,
177
+ but it never bootstraps implicitly. See [installation](docs/installation.md).
178
+
179
+ ## Kaggle CUDA-driver discovery
180
+
181
+ The toolkit was at `/usr/local/cuda-12.8`, while the mounted live driver was
182
+ `/usr/local/nvidia/lib64/libcuda.so`. CMake found the toolkit but initially did
183
+ not expose `CUDA::cuda_driver`. The successful build made the driver directory
184
+ visible with:
185
+
186
+ ```bash
187
+ export CMAKE_LIBRARY_PATH=/usr/local/nvidia/lib64
188
+ ```
189
+
190
+ The source-build scripts retain this workaround. The wheel itself is excluded
191
+ from Git.
192
+
193
+ ## OpenAI-compatible serving
194
+
195
+ The server helper creates an argument array and invokes upstream `vllm serve`
196
+ without a shell:
197
+
198
+ ```bash
199
+ kaggle-vllm serve /kaggle/input/qwen2.5-3b-t4x2-sharded \
200
+ --served-model-name qwen2.5-3b-kaggle-t4x2 \
201
+ --load-format sharded_state \
202
+ --tensor-parallel-size 2 \
203
+ --dtype float16 \
204
+ --max-model-len 2048 \
205
+ --gpu-memory-utilization 0.70 \
206
+ --host 127.0.0.1 \
207
+ --port 8001
208
+ ```
209
+
210
+ The archived Qwen run returned HTTP 200 from both `GET /v1/models` and
211
+ `POST /v1/chat/completions`. See [OpenAI serving](docs/openai-serving.md).
212
+
213
+ ## What was functionally validated
214
+
215
+ - CUDA-enabled vLLM wheel build and SHA256 verification
216
+ - staged native imports (`vllm._C`, `vllm._moe_C`, allocator)
217
+ - isolated dependency overlay while preserving system PyTorch
218
+ - single-T4 FP16 inference with `facebook/opt-125m`
219
+ - raw two-rank NCCL all-reduce (`3.0` on both ranks)
220
+ - real vLLM TP=2 inference with `facebook/opt-125m`
221
+ - Qwen/Qwen2.5-3B-Instruct FP16 TP=2 inference
222
+ - persistent TP=2 sharded-state creation and reload
223
+ - OpenAI-compatible TP=2 serving from the sharded Qwen checkpoint
224
+
225
+ The curated evidence is in [`artifacts/kaggle-2026-08-23`](artifacts/kaggle-2026-08-23/README.md),
226
+ with the larger immutable evidence and model archives kept outside Git.
227
+
228
+ ## Tesla T4 / SM75 behavior
229
+
230
+ FlashAttention 2 requires compute capability 8.0 or newer and was unavailable
231
+ on SM75. vLLM selected `TRITON_ATTN` in the recorded runs. SymmMem communicator
232
+ warnings are also expected because that capability is unavailable on SM75;
233
+ ordinary NCCL communication and TP=2 inference still completed successfully.
234
+
235
+ ## CLI
236
+
237
+ ```text
238
+ kaggle-vllm doctor
239
+ kaggle-vllm fingerprint
240
+ kaggle-vllm bootstrap [--strict] [--dry-run]
241
+ kaggle-vllm env [--manifest PATH]
242
+ kaggle-vllm verify-gpus --tensor-parallel-size 2
243
+ kaggle-vllm inspect-shards PATH --json
244
+ kaggle-vllm verify-wheel PATH [--sha256 DIGEST]
245
+ kaggle-vllm stage-wheel PATH --target TARGET [--sha256 DIGEST]
246
+ kaggle-vllm serve MODEL ...
247
+ ```
248
+
249
+ ## Artifact distribution and security
250
+
251
+ Verified release artifacts are published separately from the source repository:
252
+
253
+ - [validated Kaggle dual-T4 vLLM wheel and metadata](https://huggingface.co/waqasm86/vllm-kaggle-binaries)
254
+ - [Qwen2.5-3B-Instruct TP=2 persistent sharded state](https://huggingface.co/waqasm86/vllm-kaggle-models)
255
+
256
+ Large wheels, archives, safetensors, caches, overlays, and extracted models are
257
+ ignored by Git. Published artifacts must carry checksums, compatibility data,
258
+ and upstream attribution. Never commit Kaggle, GitHub, or Hugging Face tokens.
259
+ The Qwen persistent checkpoint remains governed by the non-commercial Qwen
260
+ Research License included with the model, not this repository's Apache-2.0
261
+ license.
262
+
263
+ ## Known limitations
264
+
265
+ - Validation is specific to the tabled Kaggle environment and CPython 3.12 ABI.
266
+ - The SDK supports Python 3.10+, but the published native wheel profile is
267
+ Linux x86_64 CPython 3.12 only.
268
+ - No local GPU test is claimed; GPU results come from archived Kaggle evidence.
269
+ - The persistent model is TP-topology-aware and validated only at TP=2.
270
+ - The copied upstream HF weight index names original HF shards; standard
271
+ Transformers loading is not supported. Use vLLM `sharded_state`.
272
+ - Eager execution/custom all-reduce settings were conservative correctness
273
+ choices, not performance benchmarks.
274
+ - No arbitrary or uneven GPU-memory split API is provided.
275
+ - Qwen redistribution/use is non-commercial under its included license.
276
+
277
+ More detail: [architecture](docs/architecture.md), [runtime](docs/kaggle-runtime.md),
278
+ [tensor parallelism](docs/tensor-parallel.md), [validation](docs/validation.md),
279
+ and the [compatibility matrix](docs/compatibility-matrix.md).
@@ -0,0 +1,48 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "kaggle-vllm"
7
+ version = "0.1.0"
8
+ description = "A lightweight Kaggle compatibility SDK around upstream vLLM for Tesla T4 GPUs."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "Apache-2.0"
12
+ authors = [{name = "kaggle-vllm contributors"}]
13
+ classifiers = [
14
+ "Development Status :: 3 - Alpha",
15
+ "Programming Language :: Python :: 3",
16
+ "Programming Language :: Python :: 3.10",
17
+ "Programming Language :: Python :: 3.11",
18
+ "Programming Language :: Python :: 3.12",
19
+ ]
20
+ dependencies = []
21
+
22
+ [project.optional-dependencies]
23
+ test = ["pytest>=8"]
24
+ hub = ["huggingface_hub>=0.36"]
25
+
26
+ [project.urls]
27
+ Documentation = "https://github.com/kaggle-vllm/kaggle-vllm/tree/main/docs"
28
+ Repository = "https://github.com/kaggle-vllm/kaggle-vllm"
29
+ Issues = "https://github.com/kaggle-vllm/kaggle-vllm/issues"
30
+ Changelog = "https://github.com/kaggle-vllm/kaggle-vllm/blob/main/CHANGELOG.md"
31
+ "Hugging Face binaries" = "https://huggingface.co/waqasm86/vllm-kaggle-binaries"
32
+ "Hugging Face TP=2 model" = "https://huggingface.co/waqasm86/vllm-kaggle-models"
33
+
34
+ [project.scripts]
35
+ kaggle-vllm = "kaggle_vllm.cli:main"
36
+
37
+ [tool.setuptools.packages.find]
38
+ where = ["src"]
39
+
40
+ [tool.setuptools.package-data]
41
+ kaggle_vllm = ["profiles/*/*.json", "profiles/*/*.txt"]
42
+
43
+ [tool.pytest.ini_options]
44
+ testpaths = ["tests"]
45
+ markers = [
46
+ "gpu: requires a real CUDA GPU runtime",
47
+ "kaggle: requires the documented Kaggle environment",
48
+ ]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,17 @@
1
+ """Kaggle compatibility helpers around upstream vLLM."""
2
+
3
+ from .bootstrap import activate_runtime, bootstrap
4
+ from .llm import KaggleLLM
5
+ from .profiles import BootstrapProfile, load_profile
6
+ from .sharding import ShardedModelInspection, inspect_sharded_model
7
+
8
+ __all__ = [
9
+ "BootstrapProfile",
10
+ "KaggleLLM",
11
+ "ShardedModelInspection",
12
+ "activate_runtime",
13
+ "bootstrap",
14
+ "inspect_sharded_model",
15
+ "load_profile",
16
+ ]
17
+ __version__ = "0.1.0"