infer-stack 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. infer_stack/__init__.py +2 -0
  2. infer_stack/backends/__init__.py +7 -0
  3. infer_stack/backends/compose_renderer.py +243 -0
  4. infer_stack/backends/kubeai_renderer.py +202 -0
  5. infer_stack/benchmark.py +38 -0
  6. infer_stack/catalog.py +438 -0
  7. infer_stack/cli/__init__.py +169 -0
  8. infer_stack/cli/__main__.py +4 -0
  9. infer_stack/cli/commands_profile.py +467 -0
  10. infer_stack/cli/commands_runtime.py +719 -0
  11. infer_stack/cli/commands_smoke.py +691 -0
  12. infer_stack/cli/compose.py +755 -0
  13. infer_stack/cli/context.py +471 -0
  14. infer_stack/cli/options.py +134 -0
  15. infer_stack/cli/probes.py +178 -0
  16. infer_stack/config.py +450 -0
  17. infer_stack/contracts.py +223 -0
  18. infer_stack/diff_prompt.py +117 -0
  19. infer_stack/docker_utils.py +230 -0
  20. infer_stack/env_utils.py +97 -0
  21. infer_stack/experimental/model_catalog_discover.py +1155 -0
  22. infer_stack/experimental/model_memory_estimator.py +1264 -0
  23. infer_stack/experimental/stress_test_long_context.py +397 -0
  24. infer_stack/hardware.py +70 -0
  25. infer_stack/kubeai_ops.py +76 -0
  26. infer_stack/paths.py +87 -0
  27. infer_stack/profile_runtime.py +46 -0
  28. infer_stack/renderer.py +19 -0
  29. infer_stack/resolver.py +1092 -0
  30. infer_stack/templates/default-models.yaml +674 -0
  31. infer_stack/templates/default-ollama-models.yaml +31 -0
  32. infer_stack/templates/default-profiles.yaml +1731 -0
  33. infer_stack/templates/default-vllm-models.yaml +714 -0
  34. infer_stack/templates/docker-compose.yml.j2 +430 -0
  35. infer_stack/templates/litellm_config.yaml.j2 +44 -0
  36. infer_stack/templates/nginx.conf.j2 +84 -0
  37. infer_stack/tuning.py +3 -0
  38. infer_stack/validator.py +314 -0
  39. infer_stack/verification.py +46 -0
  40. infer_stack-0.6.0.dist-info/METADATA +1034 -0
  41. infer_stack-0.6.0.dist-info/RECORD +44 -0
  42. infer_stack-0.6.0.dist-info/WHEEL +5 -0
  43. infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
  44. infer_stack-0.6.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,674 @@
1
+ models:
2
+ # Smallest Qwen3.5 — intended for fast smoke tests, local development,
3
+ # and high-concurrency utility tasks. Fits on a single 8 GiB GPU at FP16
4
+ # with room for activations and a modest KV cache.
5
+ qwen3.5-0.8b:
6
+ hf_model_id: Qwen/Qwen3.5-0.8B
7
+ served_model_name: qwen3.5-0.8b
8
+ family: qwen3.5
9
+ modalities: [text]
10
+ memory_class_gib: 4
11
+ min_vram_gib_per_replica: 4
12
+ preferred_gpu_count: 1
13
+ context_window: 32768
14
+ reasoning:
15
+ enabled: true
16
+ parser: qwen3
17
+ expose_to_openwebui: true
18
+ defaults:
19
+ max_model_len: 32768
20
+ gpu_memory_utilization: 0.9
21
+ enable_prefix_caching: true
22
+ max_num_batched_tokens: 4096
23
+ max_num_seqs: 32
24
+ qwen3.5-2b:
25
+ hf_model_id: Qwen/Qwen3.5-2B
26
+ served_model_name: qwen3.5-2b
27
+ family: qwen3.5
28
+ modalities: [text]
29
+ memory_class_gib: 8
30
+ min_vram_gib_per_replica: 8
31
+ preferred_gpu_count: 1
32
+ context_window: 32768
33
+ reasoning:
34
+ enabled: true
35
+ parser: qwen3
36
+ expose_to_openwebui: true
37
+ defaults:
38
+ max_model_len: 32768
39
+ gpu_memory_utilization: 0.9
40
+ enable_prefix_caching: true
41
+ max_num_batched_tokens: 4096
42
+ max_num_seqs: 32
43
+ qwen3.5-4b:
44
+ hf_model_id: Qwen/Qwen3.5-4B
45
+ served_model_name: qwen3.5-4b
46
+ family: qwen3.5
47
+ modalities: [text]
48
+ memory_class_gib: 12
49
+ min_vram_gib_per_replica: 12
50
+ preferred_gpu_count: 1
51
+ context_window: 32768
52
+ reasoning:
53
+ enabled: true
54
+ parser: qwen3
55
+ expose_to_openwebui: true
56
+ defaults:
57
+ max_model_len: 32768
58
+ gpu_memory_utilization: 0.9
59
+ enable_prefix_caching: true
60
+ max_num_batched_tokens: 4096
61
+ max_num_seqs: 24
62
+ qwen3.5-9b:
63
+ hf_model_id: Qwen/Qwen3.5-9B
64
+ served_model_name: qwen3.5-9b
65
+ family: qwen3.5
66
+ modalities: [text]
67
+ memory_class_gib: 20
68
+ min_vram_gib_per_replica: 20
69
+ preferred_gpu_count: 1
70
+ context_window: 32768
71
+ reasoning:
72
+ enabled: true
73
+ parser: qwen3
74
+ expose_to_openwebui: true
75
+ defaults:
76
+ max_model_len: 32768
77
+ gpu_memory_utilization: 0.9
78
+ enable_prefix_caching: true
79
+ max_num_batched_tokens: 8192
80
+ max_num_seqs: 16
81
+ qwen3.5-27b:
82
+ hf_model_id: Qwen/Qwen3.5-27B
83
+ served_model_name: qwen3.5-27b
84
+ family: qwen3.5
85
+ modalities: [text]
86
+ memory_class_gib: 48
87
+ min_vram_gib_per_replica: 48
88
+ preferred_gpu_count: 1
89
+ context_window: 32768
90
+ reasoning:
91
+ enabled: true
92
+ parser: qwen3
93
+ expose_to_openwebui: true
94
+ defaults:
95
+ max_model_len: 32768
96
+ gpu_memory_utilization: 0.9
97
+ enable_prefix_caching: true
98
+ max_num_batched_tokens: 8192
99
+ max_num_seqs: 16
100
+ qwen3.5-35b-a3b:
101
+ hf_model_id: Qwen/Qwen3.5-35B-A3B
102
+ served_model_name: qwen3.5-35b-a3b
103
+ family: qwen3.5
104
+ modalities: [text]
105
+ memory_class_gib: 48
106
+ min_vram_gib_per_replica: 48
107
+ preferred_gpu_count: 1
108
+ context_window: 32768
109
+ reasoning:
110
+ enabled: true
111
+ parser: qwen3
112
+ expose_to_openwebui: true
113
+ defaults:
114
+ max_model_len: 32768
115
+ gpu_memory_utilization: 0.9
116
+ enable_prefix_caching: true
117
+ max_num_batched_tokens: 8192
118
+ max_num_seqs: 16
119
+ qwen3.5-122b-a10b:
120
+ hf_model_id: Qwen/Qwen3.5-122B-A10B
121
+ served_model_name: qwen3.5-122b-a10b
122
+ family: qwen3.5
123
+ modalities: [text]
124
+ memory_class_gib: 80
125
+ min_vram_gib_per_replica: 80
126
+ preferred_gpu_count: 2
127
+ context_window: 131072
128
+ reasoning:
129
+ enabled: true
130
+ parser: qwen3
131
+ expose_to_openwebui: true
132
+ defaults:
133
+ max_model_len: 32768
134
+ gpu_memory_utilization: 0.9
135
+ enable_prefix_caching: true
136
+ max_num_batched_tokens: 4096
137
+ max_num_seqs: 8
138
+
139
+ qwen3.5-122b-a10b-fp8:
140
+ hf_model_id: Qwen/Qwen3.5-122B-A10B-FP8
141
+ served_model_name: qwen3.5-122b-a10b-fp8
142
+ family: qwen3.5
143
+ modalities: [text]
144
+ memory_class_gib: 80
145
+ min_vram_gib_per_replica: 80
146
+ preferred_gpu_count: 2
147
+ context_window: 262144
148
+ reasoning:
149
+ enabled: true
150
+ parser: qwen3
151
+ expose_to_openwebui: true
152
+ defaults:
153
+ max_model_len: 32768
154
+ gpu_memory_utilization: 0.9
155
+ enable_prefix_caching: true
156
+ max_num_batched_tokens: 4096
157
+ max_num_seqs: 8
158
+ tool_calling:
159
+ auto: true
160
+ parser: qwen3_xml
161
+
162
+ qwen3.6-35b-a3b:
163
+ hf_model_id: Qwen/Qwen3.6-35B-A3B
164
+ served_model_name: qwen3.6-35b-a3b
165
+ tokenizer_name: Qwen/Qwen3.6-35B-A3B
166
+ family: qwen3.6
167
+ modalities: [text]
168
+ memory_class_gib: 80
169
+ min_vram_gib_per_replica: 24
170
+ preferred_gpu_count: 2
171
+ context_window: 262144
172
+ supported_protocols: [chat, completions]
173
+ reasoning:
174
+ enabled: true
175
+ parser: qwen3
176
+ expose_to_openwebui: true
177
+ defaults:
178
+ max_model_len: 262144
179
+ gpu_memory_utilization: 0.95
180
+ enable_prefix_caching: false
181
+ max_num_batched_tokens: 8192
182
+ max_num_seqs: 8
183
+
184
+ gemma4-e2b:
185
+ hf_model_id: google/gemma-4-E2B-it
186
+ served_model_name: gemma4-e2b
187
+ family: gemma4
188
+ modalities: [text, image, audio]
189
+ memory_class_gib: 8
190
+ min_vram_gib_per_replica: 8
191
+ preferred_gpu_count: 1
192
+ context_window: 131072
193
+ thinking_history_policy: keep_final_only
194
+ notes:
195
+ - Historical turns should keep only final responses, not prior thinking content.
196
+ defaults:
197
+ max_model_len: 65536
198
+ gpu_memory_utilization: 0.9
199
+ enable_prefix_caching: true
200
+ max_num_batched_tokens: 4096
201
+ max_num_seqs: 16
202
+ tool_calling:
203
+ auto: true
204
+ parser: gemma4
205
+ gemma4-e4b:
206
+ hf_model_id: google/gemma-4-E4B-it
207
+ served_model_name: gemma4-e4b
208
+ family: gemma4
209
+ modalities: [text, image, audio]
210
+ memory_class_gib: 10
211
+ min_vram_gib_per_replica: 10
212
+ preferred_gpu_count: 1
213
+ context_window: 131072
214
+ thinking_history_policy: keep_final_only
215
+ notes:
216
+ - Historical turns should keep only final responses, not prior thinking content.
217
+ defaults:
218
+ max_model_len: 65536
219
+ gpu_memory_utilization: 0.9
220
+ enable_prefix_caching: true
221
+ max_num_batched_tokens: 4096
222
+ max_num_seqs: 16
223
+ tool_calling:
224
+ auto: true
225
+ parser: gemma4
226
+ gemma4-26b:
227
+ hf_model_id: google/gemma-4-26B-A4B-it
228
+ served_model_name: gemma4-26b
229
+ family: gemma4
230
+ modalities: [text, image]
231
+ memory_class_gib: 20
232
+ min_vram_gib_per_replica: 20
233
+ preferred_gpu_count: 1
234
+ context_window: 262144
235
+ thinking_history_policy: keep_final_only
236
+ notes:
237
+ - Historical turns should keep only final responses, not prior thinking content.
238
+ defaults:
239
+ max_model_len: 65536
240
+ gpu_memory_utilization: 0.9
241
+ enable_prefix_caching: true
242
+ max_num_batched_tokens: 4096
243
+ max_num_seqs: 12
244
+ tool_calling:
245
+ auto: true
246
+ parser: gemma4
247
+ gemma4-31b:
248
+ hf_model_id: google/gemma-4-31B-it
249
+ served_model_name: gemma4-31b
250
+ family: gemma4
251
+ modalities: [text, image]
252
+ memory_class_gib: 22
253
+ min_vram_gib_per_replica: 22
254
+ preferred_gpu_count: 1
255
+ context_window: 262144
256
+ thinking_history_policy: keep_final_only
257
+ notes:
258
+ - Historical turns should keep only final responses, not prior thinking content.
259
+ defaults:
260
+ max_model_len: 65536
261
+ gpu_memory_utilization: 0.9
262
+ enable_prefix_caching: true
263
+ max_num_batched_tokens: 4096
264
+ max_num_seqs: 12
265
+ tool_calling:
266
+ auto: true
267
+ parser: gemma4
268
+
269
+ # ── HELM evaluation models ────────────────────────────────────────────────
270
+ # served_model_name matches the HELM side tag so HELM can route correctly.
271
+
272
+ qwen2-72b-instruct:
273
+ hf_model_id: Qwen/Qwen2-72B-Instruct
274
+ served_model_name: qwen/qwen2-72b-instruct
275
+ logical_model_name: qwen/qwen2-72b-instruct
276
+ tokenizer_name: qwen/qwen2-72b-instruct
277
+ family: qwen2
278
+ modalities: [text]
279
+ memory_class_gib: 144
280
+ min_vram_gib_per_replica: 144
281
+ preferred_gpu_count: 2
282
+ context_window: 131072
283
+ defaults:
284
+ max_model_len: 32768
285
+ gpu_memory_utilization: 0.9
286
+ enable_prefix_caching: true
287
+ max_num_batched_tokens: 4096
288
+ max_num_seqs: 8
289
+
290
+ # Smallest *real* instruct-tuned Qwen — usable for true end-to-end
291
+ # smoke tests that actually pull weights and exercise vLLM. ~1 GiB fp16.
292
+ qwen2.5-0.5b-instruct:
293
+ hf_model_id: Qwen/Qwen2.5-0.5B-Instruct
294
+ served_model_name: qwen/qwen2.5-0.5b-instruct
295
+ logical_model_name: qwen/qwen2.5-0.5b-instruct
296
+ tokenizer_name: Qwen/Qwen2.5-0.5B-Instruct
297
+ family: qwen2.5
298
+ modalities: [text]
299
+ memory_class_gib: 2
300
+ min_vram_gib_per_replica: 2
301
+ preferred_gpu_count: 1
302
+ context_window: 32768
303
+ defaults:
304
+ max_model_len: 8192
305
+ gpu_memory_utilization: 0.85
306
+ enable_prefix_caching: true
307
+ max_num_batched_tokens: 4096
308
+ max_num_seqs: 16
309
+
310
+ qwen2.5-7b-instruct:
311
+ hf_model_id: Qwen/Qwen2.5-7B-Instruct
312
+ served_model_name: qwen/qwen2.5-7b-instruct-turbo
313
+ logical_model_name: qwen/qwen2.5-7b-instruct-turbo
314
+ tokenizer_name: qwen/qwen2.5-7b-instruct-turbo
315
+ family: qwen2.5
316
+ modalities: [text]
317
+ memory_class_gib: 16
318
+ min_vram_gib_per_replica: 16
319
+ preferred_gpu_count: 1
320
+ context_window: 131072
321
+ defaults:
322
+ max_model_len: 32768
323
+ gpu_memory_utilization: 0.9
324
+ enable_prefix_caching: true
325
+ max_num_batched_tokens: 8192
326
+ max_num_seqs: 16
327
+
328
+ qwen2.5-72b-instruct:
329
+ hf_model_id: Qwen/Qwen2.5-72B-Instruct
330
+ served_model_name: qwen/qwen2.5-72b-instruct-turbo
331
+ logical_model_name: qwen/qwen2.5-72b-instruct-turbo
332
+ tokenizer_name: qwen/qwen2.5-72b-instruct-turbo
333
+ family: qwen2.5
334
+ modalities: [text]
335
+ memory_class_gib: 144
336
+ min_vram_gib_per_replica: 144
337
+ preferred_gpu_count: 2
338
+ context_window: 131072
339
+ defaults:
340
+ max_model_len: 32768
341
+ gpu_memory_utilization: 0.9
342
+ enable_prefix_caching: true
343
+ max_num_batched_tokens: 4096
344
+ max_num_seqs: 8
345
+
346
+ gpt-oss-20b:
347
+ hf_model_id: openai/gpt-oss-20b
348
+ served_model_name: openai/gpt-oss-20b
349
+ logical_model_name: openai/gpt-oss-20b
350
+ tokenizer_name: openai/o200k_harmony
351
+ family: gpt-oss
352
+ modalities: [text]
353
+ memory_class_gib: 40
354
+ min_vram_gib_per_replica: 40
355
+ preferred_gpu_count: 1
356
+ context_window: 131072
357
+ defaults:
358
+ max_model_len: 32768
359
+ gpu_memory_utilization: 0.9
360
+ enable_prefix_caching: true
361
+ max_num_batched_tokens: 8192
362
+ max_num_seqs: 16
363
+
364
+ llama-3.1-8b-instruct:
365
+ hf_model_id: meta-llama/Meta-Llama-3.1-8B-Instruct
366
+ served_model_name: meta/llama-3.1-8b-instruct-turbo
367
+ family: llama3.1
368
+ modalities: [text]
369
+ memory_class_gib: 18
370
+ min_vram_gib_per_replica: 18
371
+ preferred_gpu_count: 1
372
+ context_window: 131072
373
+ defaults:
374
+ max_model_len: 32768
375
+ gpu_memory_utilization: 0.9
376
+ enable_prefix_caching: true
377
+ max_num_batched_tokens: 8192
378
+ max_num_seqs: 16
379
+
380
+ llama-3-8b-instruct:
381
+ hf_model_id: meta-llama/Meta-Llama-3-8B-Instruct
382
+ served_model_name: meta/llama-3-8b-chat
383
+ family: llama3
384
+ modalities: [text]
385
+ memory_class_gib: 18
386
+ min_vram_gib_per_replica: 18
387
+ preferred_gpu_count: 1
388
+ context_window: 8192
389
+ defaults:
390
+ max_model_len: 8192
391
+ gpu_memory_utilization: 0.9
392
+ enable_prefix_caching: true
393
+ max_num_batched_tokens: 8192
394
+ max_num_seqs: 16
395
+
396
+ llama-2-7b:
397
+ hf_model_id: meta-llama/Llama-2-7b-hf
398
+ served_model_name: meta/llama-2-7b
399
+ family: llama2
400
+ modalities: [text]
401
+ supported_protocols: [completions]
402
+ memory_class_gib: 16
403
+ min_vram_gib_per_replica: 16
404
+ preferred_gpu_count: 1
405
+ context_window: 4096
406
+ defaults:
407
+ max_model_len: 4096
408
+ gpu_memory_utilization: 0.9
409
+ enable_prefix_caching: true
410
+ max_num_batched_tokens: 4096
411
+ max_num_seqs: 16
412
+
413
+ llama-2-13b:
414
+ hf_model_id: meta-llama/Llama-2-13b-hf
415
+ served_model_name: meta/llama-2-13b
416
+ family: llama2
417
+ modalities: [text]
418
+ supported_protocols: [completions]
419
+ memory_class_gib: 28
420
+ min_vram_gib_per_replica: 28
421
+ preferred_gpu_count: 1
422
+ context_window: 4096
423
+ defaults:
424
+ max_model_len: 4096
425
+ gpu_memory_utilization: 0.9
426
+ enable_prefix_caching: true
427
+ max_num_batched_tokens: 4096
428
+ max_num_seqs: 16
429
+
430
+ # LLaMA-2-70B at fp16 needs ~140 GB; the default tp=2 layout fits on
431
+ # two 96 GB GPUs with headroom for KV cache. Public HELM ran fp16,
432
+ # so the audit recipe matches that — no quantization in this profile.
433
+ llama-2-70b:
434
+ hf_model_id: meta-llama/Llama-2-70b-hf
435
+ served_model_name: meta/llama-2-70b
436
+ family: llama2
437
+ modalities: [text]
438
+ supported_protocols: [completions]
439
+ memory_class_gib: 140
440
+ min_vram_gib_per_replica: 70
441
+ preferred_gpu_count: 2
442
+ context_window: 4096
443
+ defaults:
444
+ max_model_len: 4096
445
+ gpu_memory_utilization: 0.9
446
+ enable_prefix_caching: true
447
+ max_num_batched_tokens: 4096
448
+ max_num_seqs: 8
449
+
450
+ mistral-7b-instruct-v0.3:
451
+ hf_model_id: mistralai/Mistral-7B-Instruct-v0.3
452
+ served_model_name: mistralai/mistral-7b-instruct-v0.3
453
+ family: mistral
454
+ modalities: [text]
455
+ memory_class_gib: 16
456
+ min_vram_gib_per_replica: 16
457
+ preferred_gpu_count: 1
458
+ context_window: 32768
459
+ defaults:
460
+ max_model_len: 32768
461
+ gpu_memory_utilization: 0.9
462
+ enable_prefix_caching: true
463
+ max_num_batched_tokens: 8192
464
+ max_num_seqs: 16
465
+
466
+ mistral-7b-v0.1:
467
+ hf_model_id: mistralai/Mistral-7B-v0.1
468
+ served_model_name: mistralai/mistral-7b-v0.1
469
+ family: mistral
470
+ modalities: [text]
471
+ supported_protocols: [completions]
472
+ memory_class_gib: 16
473
+ min_vram_gib_per_replica: 16
474
+ preferred_gpu_count: 1
475
+ context_window: 32768
476
+ defaults:
477
+ max_model_len: 32768
478
+ gpu_memory_utilization: 0.9
479
+ enable_prefix_caching: true
480
+ max_num_batched_tokens: 8192
481
+ max_num_seqs: 16
482
+
483
+ # Falcon uses ALiBi attention: prefix caching must be disabled.
484
+ falcon-7b:
485
+ hf_model_id: tiiuae/falcon-7b
486
+ served_model_name: tiiuae/falcon-7b
487
+ family: falcon
488
+ modalities: [text]
489
+ supported_protocols: [completions]
490
+ memory_class_gib: 16
491
+ min_vram_gib_per_replica: 16
492
+ preferred_gpu_count: 1
493
+ context_window: 2048
494
+ defaults:
495
+ max_model_len: 2048
496
+ gpu_memory_utilization: 0.9
497
+ enable_prefix_caching: false
498
+ max_num_batched_tokens: 2048
499
+ max_num_seqs: 16
500
+
501
+ falcon-7b-instruct:
502
+ hf_model_id: tiiuae/falcon-7b-instruct
503
+ served_model_name: tiiuae/falcon-7b-instruct
504
+ family: falcon
505
+ modalities: [text]
506
+ memory_class_gib: 16
507
+ min_vram_gib_per_replica: 16
508
+ preferred_gpu_count: 1
509
+ context_window: 2048
510
+ defaults:
511
+ max_model_len: 2048
512
+ gpu_memory_utilization: 0.9
513
+ enable_prefix_caching: false
514
+ max_num_batched_tokens: 2048
515
+ max_num_seqs: 16
516
+
517
+ # GPT-2 base, 124M params (~250 MB fp16). Completions-only — no chat
518
+ # template. Useful as a near-zero-cost plumbing smoke test: download is
519
+ # fast, no HF_TOKEN required, and the request path covers the whole stack.
520
+ gpt2:
521
+ hf_model_id: openai-community/gpt2
522
+ served_model_name: gpt2
523
+ logical_model_name: gpt2
524
+ tokenizer_name: openai-community/gpt2
525
+ family: gpt2
526
+ modalities: [text]
527
+ supported_protocols: [completions]
528
+ memory_class_gib: 1
529
+ min_vram_gib_per_replica: 1
530
+ preferred_gpu_count: 1
531
+ context_window: 1024
532
+ defaults:
533
+ max_model_len: 1024
534
+ gpu_memory_utilization: 0.3
535
+ enable_prefix_caching: true
536
+ max_num_batched_tokens: 1024
537
+ max_num_seqs: 16
538
+
539
+ # Smallest practical instruct-tuned model on HuggingFace. ~270 MB fp16.
540
+ # Ideal for CI smoke tests that need real chat-template output.
541
+ smollm2-135m-instruct:
542
+ hf_model_id: HuggingFaceTB/SmolLM2-135M-Instruct
543
+ served_model_name: huggingfacetb/smollm2-135m-instruct
544
+ logical_model_name: huggingfacetb/smollm2-135m-instruct
545
+ tokenizer_name: HuggingFaceTB/SmolLM2-135M-Instruct
546
+ family: smollm2
547
+ modalities: [text]
548
+ memory_class_gib: 1
549
+ min_vram_gib_per_replica: 1
550
+ preferred_gpu_count: 1
551
+ context_window: 8192
552
+ defaults:
553
+ max_model_len: 4096
554
+ gpu_memory_utilization: 0.6
555
+ enable_prefix_caching: true
556
+ max_num_batched_tokens: 2048
557
+ max_num_seqs: 16
558
+
559
+ # Pythia uses ALiBi attention: prefix caching must be disabled.
560
+ pythia-160m:
561
+ hf_model_id: EleutherAI/pythia-160m
562
+ served_model_name: eleutherai/pythia-160m
563
+ logical_model_name: eleutherai/pythia-160m
564
+ tokenizer_name: EleutherAI/pythia-160m
565
+ family: pythia
566
+ modalities: [text]
567
+ supported_protocols: [completions]
568
+ memory_class_gib: 1
569
+ min_vram_gib_per_replica: 1
570
+ preferred_gpu_count: 1
571
+ context_window: 2048
572
+ defaults:
573
+ max_model_len: 2048
574
+ gpu_memory_utilization: 0.6
575
+ enable_prefix_caching: false
576
+ max_num_batched_tokens: 2048
577
+ max_num_seqs: 16
578
+
579
+ pythia-410m:
580
+ hf_model_id: EleutherAI/pythia-410m
581
+ served_model_name: eleutherai/pythia-410m
582
+ logical_model_name: eleutherai/pythia-410m
583
+ tokenizer_name: EleutherAI/pythia-410m
584
+ family: pythia
585
+ modalities: [text]
586
+ supported_protocols: [completions]
587
+ memory_class_gib: 2
588
+ min_vram_gib_per_replica: 2
589
+ preferred_gpu_count: 1
590
+ context_window: 2048
591
+ defaults:
592
+ max_model_len: 2048
593
+ gpu_memory_utilization: 0.7
594
+ enable_prefix_caching: false
595
+ max_num_batched_tokens: 2048
596
+ max_num_seqs: 16
597
+
598
+ pythia-6.9b:
599
+ hf_model_id: EleutherAI/pythia-6.9b
600
+ served_model_name: eleutherai/pythia-6.9b
601
+ logical_model_name: eleutherai/pythia-6.9b
602
+ tokenizer_name: eleutherai/pythia-6.9b
603
+ family: pythia
604
+ modalities: [text]
605
+ supported_protocols: [completions]
606
+ memory_class_gib: 16
607
+ min_vram_gib_per_replica: 16
608
+ preferred_gpu_count: 1
609
+ context_window: 2048
610
+ defaults:
611
+ max_model_len: 2048
612
+ gpu_memory_utilization: 0.9
613
+ enable_prefix_caching: false
614
+ max_num_batched_tokens: 2048
615
+ max_num_seqs: 16
616
+
617
+ pythia-2.8b-v0:
618
+ hf_model_id: EleutherAI/pythia-2.8b-v0
619
+ served_model_name: eleutherai/pythia-2.8b-v0
620
+ logical_model_name: eleutherai/pythia-2.8b-v0
621
+ tokenizer_name: eleutherai/pythia-2.8b-v0
622
+ family: pythia
623
+ modalities: [text]
624
+ supported_protocols: [completions]
625
+ memory_class_gib: 8
626
+ min_vram_gib_per_replica: 8
627
+ preferred_gpu_count: 1
628
+ context_window: 2048
629
+ defaults:
630
+ max_model_len: 2048
631
+ gpu_memory_utilization: 0.9
632
+ enable_prefix_caching: false
633
+ max_num_batched_tokens: 2048
634
+ max_num_seqs: 16
635
+
636
+ pythia-1b-v0:
637
+ hf_model_id: EleutherAI/pythia-1b-v0
638
+ served_model_name: eleutherai/pythia-1b-v0
639
+ logical_model_name: eleutherai/pythia-1b-v0
640
+ tokenizer_name: eleutherai/pythia-1b-v0
641
+ family: pythia
642
+ modalities: [text]
643
+ supported_protocols: [completions]
644
+ memory_class_gib: 4
645
+ min_vram_gib_per_replica: 4
646
+ preferred_gpu_count: 1
647
+ context_window: 2048
648
+ defaults:
649
+ max_model_len: 2048
650
+ gpu_memory_utilization: 0.9
651
+ enable_prefix_caching: false
652
+ max_num_batched_tokens: 2048
653
+ max_num_seqs: 16
654
+
655
+ vicuna-7b-v1.3:
656
+ hf_model_id: lmsys/vicuna-7b-v1.3
657
+ served_model_name: lmsys/vicuna-7b-v1.3
658
+ logical_model_name: lmsys/vicuna-7b-v1.3
659
+ tokenizer_name: lmsys/vicuna-7b-v1.3
660
+ family: vicuna
661
+ supported_protocols: [completions]
662
+ notes:
663
+ - Use a no-chat-template style profile when reproducing legacy completion-style prompting.
664
+ modalities: [text]
665
+ memory_class_gib: 16
666
+ min_vram_gib_per_replica: 16
667
+ preferred_gpu_count: 1
668
+ context_window: 2048
669
+ defaults:
670
+ max_model_len: 2048
671
+ gpu_memory_utilization: 0.9
672
+ enable_prefix_caching: false
673
+ max_num_batched_tokens: 2048
674
+ max_num_seqs: 16
@@ -0,0 +1,31 @@
1
+ ollama_models:
2
+ qwen3.5-2b:
3
+ tag: qwen3.5:2b
4
+ served_model_name: qwen3.5-2b
5
+ defaults:
6
+ num_ctx: 4096
7
+ temperature: 0.2
8
+ qwen3.5-4b:
9
+ tag: qwen3.5:4b
10
+ served_model_name: qwen3.5-4b
11
+ defaults:
12
+ num_ctx: 4096
13
+ temperature: 0.2
14
+ qwen3.5-9b-q4:
15
+ tag: qwen3.5:9b-q4_K_M
16
+ served_model_name: qwen3.5-9b-q4
17
+ defaults:
18
+ num_ctx: 4096
19
+ temperature: 0.2
20
+ gemma4-e2b:
21
+ tag: gemma4:e2b
22
+ served_model_name: gemma4-e2b
23
+ defaults:
24
+ num_ctx: 4096
25
+ temperature: 0.2
26
+ smollm2-135m:
27
+ tag: smollm2:135m
28
+ served_model_name: smollm2-135m
29
+ defaults:
30
+ num_ctx: 4096
31
+ temperature: 0.2