infer-stack 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. infer_stack/__init__.py +2 -0
  2. infer_stack/backends/__init__.py +7 -0
  3. infer_stack/backends/compose_renderer.py +243 -0
  4. infer_stack/backends/kubeai_renderer.py +202 -0
  5. infer_stack/benchmark.py +38 -0
  6. infer_stack/catalog.py +438 -0
  7. infer_stack/cli/__init__.py +169 -0
  8. infer_stack/cli/__main__.py +4 -0
  9. infer_stack/cli/commands_profile.py +467 -0
  10. infer_stack/cli/commands_runtime.py +719 -0
  11. infer_stack/cli/commands_smoke.py +691 -0
  12. infer_stack/cli/compose.py +755 -0
  13. infer_stack/cli/context.py +471 -0
  14. infer_stack/cli/options.py +134 -0
  15. infer_stack/cli/probes.py +178 -0
  16. infer_stack/config.py +450 -0
  17. infer_stack/contracts.py +223 -0
  18. infer_stack/diff_prompt.py +117 -0
  19. infer_stack/docker_utils.py +230 -0
  20. infer_stack/env_utils.py +97 -0
  21. infer_stack/experimental/model_catalog_discover.py +1155 -0
  22. infer_stack/experimental/model_memory_estimator.py +1264 -0
  23. infer_stack/experimental/stress_test_long_context.py +397 -0
  24. infer_stack/hardware.py +70 -0
  25. infer_stack/kubeai_ops.py +76 -0
  26. infer_stack/paths.py +87 -0
  27. infer_stack/profile_runtime.py +46 -0
  28. infer_stack/renderer.py +19 -0
  29. infer_stack/resolver.py +1092 -0
  30. infer_stack/templates/default-models.yaml +674 -0
  31. infer_stack/templates/default-ollama-models.yaml +31 -0
  32. infer_stack/templates/default-profiles.yaml +1731 -0
  33. infer_stack/templates/default-vllm-models.yaml +714 -0
  34. infer_stack/templates/docker-compose.yml.j2 +430 -0
  35. infer_stack/templates/litellm_config.yaml.j2 +44 -0
  36. infer_stack/templates/nginx.conf.j2 +84 -0
  37. infer_stack/tuning.py +3 -0
  38. infer_stack/validator.py +314 -0
  39. infer_stack/verification.py +46 -0
  40. infer_stack-0.6.0.dist-info/METADATA +1034 -0
  41. infer_stack-0.6.0.dist-info/RECORD +44 -0
  42. infer_stack-0.6.0.dist-info/WHEEL +5 -0
  43. infer_stack-0.6.0.dist-info/entry_points.txt +2 -0
  44. infer_stack-0.6.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,714 @@
1
+ vllm_models:
2
+ qwen3.5-0.8b:
3
+ hf_model_id: Qwen/Qwen3.5-0.8B
4
+ served_model_name: qwen3.5-0.8b
5
+ family: qwen3.5
6
+ modalities:
7
+ - text
8
+ memory_class_gib: 4
9
+ min_vram_gib_per_replica: 4
10
+ preferred_gpu_count: 1
11
+ context_window: 32768
12
+ reasoning:
13
+ enabled: true
14
+ parser: qwen3
15
+ expose_to_openwebui: true
16
+ defaults:
17
+ max_model_len: 32768
18
+ gpu_memory_utilization: 0.9
19
+ enable_prefix_caching: true
20
+ max_num_batched_tokens: 4096
21
+ max_num_seqs: 32
22
+ qwen3.5-2b:
23
+ hf_model_id: Qwen/Qwen3.5-2B
24
+ served_model_name: qwen3.5-2b
25
+ family: qwen3.5
26
+ modalities:
27
+ - text
28
+ memory_class_gib: 8
29
+ min_vram_gib_per_replica: 8
30
+ preferred_gpu_count: 1
31
+ context_window: 32768
32
+ reasoning:
33
+ enabled: true
34
+ parser: qwen3
35
+ expose_to_openwebui: true
36
+ defaults:
37
+ max_model_len: 32768
38
+ gpu_memory_utilization: 0.9
39
+ enable_prefix_caching: true
40
+ max_num_batched_tokens: 4096
41
+ max_num_seqs: 32
42
+ qwen3.5-4b:
43
+ hf_model_id: Qwen/Qwen3.5-4B
44
+ served_model_name: qwen3.5-4b
45
+ family: qwen3.5
46
+ modalities:
47
+ - text
48
+ memory_class_gib: 12
49
+ min_vram_gib_per_replica: 12
50
+ preferred_gpu_count: 1
51
+ context_window: 32768
52
+ reasoning:
53
+ enabled: true
54
+ parser: qwen3
55
+ expose_to_openwebui: true
56
+ defaults:
57
+ max_model_len: 32768
58
+ gpu_memory_utilization: 0.9
59
+ enable_prefix_caching: true
60
+ max_num_batched_tokens: 4096
61
+ max_num_seqs: 24
62
+ qwen3.5-9b:
63
+ hf_model_id: Qwen/Qwen3.5-9B
64
+ served_model_name: qwen3.5-9b
65
+ family: qwen3.5
66
+ modalities:
67
+ - text
68
+ memory_class_gib: 20
69
+ min_vram_gib_per_replica: 20
70
+ preferred_gpu_count: 1
71
+ context_window: 32768
72
+ reasoning:
73
+ enabled: true
74
+ parser: qwen3
75
+ expose_to_openwebui: true
76
+ defaults:
77
+ max_model_len: 32768
78
+ gpu_memory_utilization: 0.9
79
+ enable_prefix_caching: true
80
+ max_num_batched_tokens: 8192
81
+ max_num_seqs: 16
82
+ qwen3.5-27b:
83
+ hf_model_id: Qwen/Qwen3.5-27B
84
+ served_model_name: qwen3.5-27b
85
+ family: qwen3.5
86
+ modalities:
87
+ - text
88
+ memory_class_gib: 48
89
+ min_vram_gib_per_replica: 48
90
+ preferred_gpu_count: 1
91
+ context_window: 32768
92
+ reasoning:
93
+ enabled: true
94
+ parser: qwen3
95
+ expose_to_openwebui: true
96
+ defaults:
97
+ max_model_len: 32768
98
+ gpu_memory_utilization: 0.9
99
+ enable_prefix_caching: true
100
+ max_num_batched_tokens: 8192
101
+ max_num_seqs: 16
102
+ qwen3.5-35b-a3b:
103
+ hf_model_id: Qwen/Qwen3.5-35B-A3B
104
+ served_model_name: qwen3.5-35b-a3b
105
+ family: qwen3.5
106
+ modalities:
107
+ - text
108
+ memory_class_gib: 48
109
+ min_vram_gib_per_replica: 48
110
+ preferred_gpu_count: 1
111
+ context_window: 32768
112
+ reasoning:
113
+ enabled: true
114
+ parser: qwen3
115
+ expose_to_openwebui: true
116
+ defaults:
117
+ max_model_len: 32768
118
+ gpu_memory_utilization: 0.9
119
+ enable_prefix_caching: true
120
+ max_num_batched_tokens: 8192
121
+ max_num_seqs: 16
122
+ qwen3.5-122b-a10b:
123
+ hf_model_id: Qwen/Qwen3.5-122B-A10B
124
+ served_model_name: qwen3.5-122b-a10b
125
+ family: qwen3.5
126
+ modalities:
127
+ - text
128
+ memory_class_gib: 80
129
+ min_vram_gib_per_replica: 80
130
+ preferred_gpu_count: 2
131
+ context_window: 131072
132
+ reasoning:
133
+ enabled: true
134
+ parser: qwen3
135
+ expose_to_openwebui: true
136
+ defaults:
137
+ max_model_len: 32768
138
+ gpu_memory_utilization: 0.9
139
+ enable_prefix_caching: true
140
+ max_num_batched_tokens: 4096
141
+ max_num_seqs: 8
142
+ qwen3.5-122b-a10b-fp8:
143
+ hf_model_id: Qwen/Qwen3.5-122B-A10B-FP8
144
+ served_model_name: qwen3.5-122b-a10b-fp8
145
+ family: qwen3.5
146
+ modalities:
147
+ - text
148
+ memory_class_gib: 80
149
+ min_vram_gib_per_replica: 80
150
+ preferred_gpu_count: 2
151
+ context_window: 262144
152
+ reasoning:
153
+ enabled: true
154
+ parser: qwen3
155
+ expose_to_openwebui: true
156
+ defaults:
157
+ max_model_len: 32768
158
+ gpu_memory_utilization: 0.9
159
+ enable_prefix_caching: true
160
+ max_num_batched_tokens: 4096
161
+ max_num_seqs: 8
162
+ tool_calling:
163
+ auto: true
164
+ parser: qwen3_xml
165
+ qwen3.6-35b-a3b:
166
+ hf_model_id: Qwen/Qwen3.6-35B-A3B
167
+ served_model_name: qwen3.6-35b-a3b
168
+ tokenizer_name: Qwen/Qwen3.6-35B-A3B
169
+ family: qwen3.6
170
+ modalities:
171
+ - text
172
+ memory_class_gib: 80
173
+ min_vram_gib_per_replica: 24
174
+ preferred_gpu_count: 2
175
+ context_window: 262144
176
+ supported_protocols:
177
+ - chat
178
+ - completions
179
+ reasoning:
180
+ enabled: true
181
+ parser: qwen3
182
+ expose_to_openwebui: true
183
+ defaults:
184
+ max_model_len: 262144
185
+ gpu_memory_utilization: 0.95
186
+ enable_prefix_caching: false
187
+ max_num_batched_tokens: 8192
188
+ max_num_seqs: 8
189
+ qwen3.6-35b-a3b-fp8:
190
+ hf_model_id: Qwen/Qwen3.6-35B-A3B-FP8
191
+ served_model_name: qwen3.6-35b-a3b
192
+ tokenizer_name: Qwen/Qwen3.6-35B-A3B
193
+ family: qwen3.6
194
+ modalities:
195
+ - text
196
+ memory_class_gib: 40
197
+ min_vram_gib_per_replica: 24
198
+ preferred_gpu_count: 1
199
+ context_window: 262144
200
+ supported_protocols:
201
+ - chat
202
+ - completions
203
+ reasoning:
204
+ enabled: true
205
+ parser: qwen3
206
+ expose_to_openwebui: true
207
+ defaults:
208
+ max_model_len: 8192
209
+ gpu_memory_utilization: 0.92
210
+ enable_prefix_caching: false
211
+ max_num_batched_tokens: 2048
212
+ max_num_seqs: 2
213
+ tool_calling:
214
+ auto: true
215
+ parser: qwen3_coder
216
+ gemma4-e2b:
217
+ hf_model_id: google/gemma-4-E2B-it
218
+ served_model_name: gemma4-e2b
219
+ family: gemma4
220
+ modalities:
221
+ - text
222
+ - image
223
+ - audio
224
+ memory_class_gib: 8
225
+ min_vram_gib_per_replica: 8
226
+ preferred_gpu_count: 1
227
+ context_window: 131072
228
+ thinking_history_policy: keep_final_only
229
+ notes:
230
+ - Historical turns should keep only final responses, not prior thinking content.
231
+ defaults:
232
+ max_model_len: 65536
233
+ gpu_memory_utilization: 0.9
234
+ enable_prefix_caching: true
235
+ max_num_batched_tokens: 4096
236
+ max_num_seqs: 16
237
+ tool_calling:
238
+ auto: true
239
+ parser: gemma4
240
+ gemma4-e4b:
241
+ hf_model_id: google/gemma-4-E4B-it
242
+ served_model_name: gemma4-e4b
243
+ family: gemma4
244
+ modalities:
245
+ - text
246
+ - image
247
+ - audio
248
+ memory_class_gib: 10
249
+ min_vram_gib_per_replica: 10
250
+ preferred_gpu_count: 1
251
+ context_window: 131072
252
+ thinking_history_policy: keep_final_only
253
+ notes:
254
+ - Historical turns should keep only final responses, not prior thinking content.
255
+ defaults:
256
+ max_model_len: 65536
257
+ gpu_memory_utilization: 0.9
258
+ enable_prefix_caching: true
259
+ max_num_batched_tokens: 4096
260
+ max_num_seqs: 16
261
+ tool_calling:
262
+ auto: true
263
+ parser: gemma4
264
+ gemma4-26b:
265
+ hf_model_id: google/gemma-4-26B-A4B-it
266
+ served_model_name: gemma4-26b
267
+ family: gemma4
268
+ modalities:
269
+ - text
270
+ - image
271
+ memory_class_gib: 20
272
+ min_vram_gib_per_replica: 20
273
+ preferred_gpu_count: 1
274
+ context_window: 262144
275
+ thinking_history_policy: keep_final_only
276
+ notes:
277
+ - Historical turns should keep only final responses, not prior thinking content.
278
+ defaults:
279
+ max_model_len: 65536
280
+ gpu_memory_utilization: 0.9
281
+ enable_prefix_caching: true
282
+ max_num_batched_tokens: 4096
283
+ max_num_seqs: 12
284
+ tool_calling:
285
+ auto: true
286
+ parser: gemma4
287
+ gemma4-31b:
288
+ hf_model_id: google/gemma-4-31B-it
289
+ served_model_name: gemma4-31b
290
+ family: gemma4
291
+ modalities:
292
+ - text
293
+ - image
294
+ memory_class_gib: 22
295
+ min_vram_gib_per_replica: 22
296
+ preferred_gpu_count: 1
297
+ context_window: 262144
298
+ thinking_history_policy: keep_final_only
299
+ notes:
300
+ - Historical turns should keep only final responses, not prior thinking content.
301
+ defaults:
302
+ max_model_len: 65536
303
+ gpu_memory_utilization: 0.9
304
+ enable_prefix_caching: true
305
+ max_num_batched_tokens: 4096
306
+ max_num_seqs: 12
307
+ tool_calling:
308
+ auto: true
309
+ parser: gemma4
310
+ qwen2-72b-instruct:
311
+ hf_model_id: Qwen/Qwen2-72B-Instruct
312
+ served_model_name: qwen/qwen2-72b-instruct
313
+ logical_model_name: qwen/qwen2-72b-instruct
314
+ tokenizer_name: qwen/qwen2-72b-instruct
315
+ family: qwen2
316
+ modalities:
317
+ - text
318
+ memory_class_gib: 144
319
+ min_vram_gib_per_replica: 144
320
+ preferred_gpu_count: 2
321
+ context_window: 131072
322
+ defaults:
323
+ max_model_len: 32768
324
+ gpu_memory_utilization: 0.9
325
+ enable_prefix_caching: true
326
+ max_num_batched_tokens: 4096
327
+ max_num_seqs: 8
328
+ qwen2.5-0.5b-instruct:
329
+ hf_model_id: Qwen/Qwen2.5-0.5B-Instruct
330
+ served_model_name: qwen/qwen2.5-0.5b-instruct
331
+ logical_model_name: qwen/qwen2.5-0.5b-instruct
332
+ tokenizer_name: Qwen/Qwen2.5-0.5B-Instruct
333
+ family: qwen2.5
334
+ modalities:
335
+ - text
336
+ memory_class_gib: 2
337
+ min_vram_gib_per_replica: 2
338
+ preferred_gpu_count: 1
339
+ context_window: 32768
340
+ defaults:
341
+ max_model_len: 8192
342
+ gpu_memory_utilization: 0.85
343
+ enable_prefix_caching: true
344
+ max_num_batched_tokens: 4096
345
+ max_num_seqs: 16
346
+ qwen2.5-7b-instruct:
347
+ hf_model_id: Qwen/Qwen2.5-7B-Instruct
348
+ served_model_name: qwen/qwen2.5-7b-instruct-turbo
349
+ logical_model_name: qwen/qwen2.5-7b-instruct-turbo
350
+ tokenizer_name: qwen/qwen2.5-7b-instruct-turbo
351
+ family: qwen2.5
352
+ modalities:
353
+ - text
354
+ memory_class_gib: 16
355
+ min_vram_gib_per_replica: 16
356
+ preferred_gpu_count: 1
357
+ context_window: 131072
358
+ defaults:
359
+ max_model_len: 32768
360
+ gpu_memory_utilization: 0.9
361
+ enable_prefix_caching: true
362
+ max_num_batched_tokens: 8192
363
+ max_num_seqs: 16
364
+ qwen2.5-72b-instruct:
365
+ hf_model_id: Qwen/Qwen2.5-72B-Instruct
366
+ served_model_name: qwen/qwen2.5-72b-instruct-turbo
367
+ logical_model_name: qwen/qwen2.5-72b-instruct-turbo
368
+ tokenizer_name: qwen/qwen2.5-72b-instruct-turbo
369
+ family: qwen2.5
370
+ modalities:
371
+ - text
372
+ memory_class_gib: 144
373
+ min_vram_gib_per_replica: 144
374
+ preferred_gpu_count: 2
375
+ context_window: 131072
376
+ defaults:
377
+ max_model_len: 32768
378
+ gpu_memory_utilization: 0.9
379
+ enable_prefix_caching: true
380
+ max_num_batched_tokens: 4096
381
+ max_num_seqs: 8
382
+ gpt-oss-20b:
383
+ hf_model_id: openai/gpt-oss-20b
384
+ served_model_name: openai/gpt-oss-20b
385
+ logical_model_name: openai/gpt-oss-20b
386
+ tokenizer_name: openai/o200k_harmony
387
+ family: gpt-oss
388
+ modalities:
389
+ - text
390
+ memory_class_gib: 40
391
+ min_vram_gib_per_replica: 40
392
+ preferred_gpu_count: 1
393
+ context_window: 131072
394
+ defaults:
395
+ max_model_len: 32768
396
+ gpu_memory_utilization: 0.9
397
+ enable_prefix_caching: true
398
+ max_num_batched_tokens: 8192
399
+ max_num_seqs: 16
400
+ llama-3.1-8b-instruct:
401
+ hf_model_id: meta-llama/Meta-Llama-3.1-8B-Instruct
402
+ served_model_name: meta/llama-3.1-8b-instruct-turbo
403
+ family: llama3.1
404
+ modalities:
405
+ - text
406
+ memory_class_gib: 18
407
+ min_vram_gib_per_replica: 18
408
+ preferred_gpu_count: 1
409
+ context_window: 131072
410
+ defaults:
411
+ max_model_len: 32768
412
+ gpu_memory_utilization: 0.9
413
+ enable_prefix_caching: true
414
+ max_num_batched_tokens: 8192
415
+ max_num_seqs: 16
416
+ llama-3-8b-instruct:
417
+ hf_model_id: meta-llama/Meta-Llama-3-8B-Instruct
418
+ served_model_name: meta/llama-3-8b-chat
419
+ family: llama3
420
+ modalities:
421
+ - text
422
+ memory_class_gib: 18
423
+ min_vram_gib_per_replica: 18
424
+ preferred_gpu_count: 1
425
+ context_window: 8192
426
+ defaults:
427
+ max_model_len: 8192
428
+ gpu_memory_utilization: 0.9
429
+ enable_prefix_caching: true
430
+ max_num_batched_tokens: 8192
431
+ max_num_seqs: 16
432
+ llama-2-7b:
433
+ hf_model_id: meta-llama/Llama-2-7b-hf
434
+ served_model_name: meta/llama-2-7b
435
+ family: llama2
436
+ modalities:
437
+ - text
438
+ supported_protocols:
439
+ - completions
440
+ memory_class_gib: 16
441
+ min_vram_gib_per_replica: 16
442
+ preferred_gpu_count: 1
443
+ context_window: 4096
444
+ defaults:
445
+ max_model_len: 4096
446
+ gpu_memory_utilization: 0.9
447
+ enable_prefix_caching: true
448
+ max_num_batched_tokens: 4096
449
+ max_num_seqs: 16
450
+ llama-2-13b:
451
+ hf_model_id: meta-llama/Llama-2-13b-hf
452
+ served_model_name: meta/llama-2-13b
453
+ family: llama2
454
+ modalities:
455
+ - text
456
+ supported_protocols:
457
+ - completions
458
+ memory_class_gib: 28
459
+ min_vram_gib_per_replica: 28
460
+ preferred_gpu_count: 1
461
+ context_window: 4096
462
+ defaults:
463
+ max_model_len: 4096
464
+ gpu_memory_utilization: 0.9
465
+ enable_prefix_caching: true
466
+ max_num_batched_tokens: 4096
467
+ max_num_seqs: 16
468
+ llama-2-70b:
469
+ hf_model_id: meta-llama/Llama-2-70b-hf
470
+ served_model_name: meta/llama-2-70b
471
+ family: llama2
472
+ modalities:
473
+ - text
474
+ supported_protocols:
475
+ - completions
476
+ memory_class_gib: 140
477
+ min_vram_gib_per_replica: 70
478
+ preferred_gpu_count: 2
479
+ context_window: 4096
480
+ defaults:
481
+ max_model_len: 4096
482
+ gpu_memory_utilization: 0.9
483
+ enable_prefix_caching: true
484
+ max_num_batched_tokens: 4096
485
+ max_num_seqs: 8
486
+ mistral-7b-instruct-v0.3:
487
+ hf_model_id: mistralai/Mistral-7B-Instruct-v0.3
488
+ served_model_name: mistralai/mistral-7b-instruct-v0.3
489
+ family: mistral
490
+ modalities:
491
+ - text
492
+ memory_class_gib: 16
493
+ min_vram_gib_per_replica: 16
494
+ preferred_gpu_count: 1
495
+ context_window: 32768
496
+ defaults:
497
+ max_model_len: 32768
498
+ gpu_memory_utilization: 0.9
499
+ enable_prefix_caching: true
500
+ max_num_batched_tokens: 8192
501
+ max_num_seqs: 16
502
+ mistral-7b-v0.1:
503
+ hf_model_id: mistralai/Mistral-7B-v0.1
504
+ served_model_name: mistralai/mistral-7b-v0.1
505
+ family: mistral
506
+ modalities:
507
+ - text
508
+ supported_protocols:
509
+ - completions
510
+ memory_class_gib: 16
511
+ min_vram_gib_per_replica: 16
512
+ preferred_gpu_count: 1
513
+ context_window: 32768
514
+ defaults:
515
+ max_model_len: 32768
516
+ gpu_memory_utilization: 0.9
517
+ enable_prefix_caching: true
518
+ max_num_batched_tokens: 8192
519
+ max_num_seqs: 16
520
+ falcon-7b:
521
+ hf_model_id: tiiuae/falcon-7b
522
+ served_model_name: tiiuae/falcon-7b
523
+ family: falcon
524
+ modalities:
525
+ - text
526
+ supported_protocols:
527
+ - completions
528
+ memory_class_gib: 16
529
+ min_vram_gib_per_replica: 16
530
+ preferred_gpu_count: 1
531
+ context_window: 2048
532
+ defaults:
533
+ max_model_len: 2048
534
+ gpu_memory_utilization: 0.9
535
+ enable_prefix_caching: false
536
+ max_num_batched_tokens: 2048
537
+ max_num_seqs: 16
538
+ falcon-7b-instruct:
539
+ hf_model_id: tiiuae/falcon-7b-instruct
540
+ served_model_name: tiiuae/falcon-7b-instruct
541
+ family: falcon
542
+ modalities:
543
+ - text
544
+ memory_class_gib: 16
545
+ min_vram_gib_per_replica: 16
546
+ preferred_gpu_count: 1
547
+ context_window: 2048
548
+ defaults:
549
+ max_model_len: 2048
550
+ gpu_memory_utilization: 0.9
551
+ enable_prefix_caching: false
552
+ max_num_batched_tokens: 2048
553
+ max_num_seqs: 16
554
+ gpt2:
555
+ hf_model_id: openai-community/gpt2
556
+ served_model_name: gpt2
557
+ logical_model_name: gpt2
558
+ tokenizer_name: openai-community/gpt2
559
+ family: gpt2
560
+ modalities:
561
+ - text
562
+ supported_protocols:
563
+ - completions
564
+ memory_class_gib: 1
565
+ min_vram_gib_per_replica: 1
566
+ preferred_gpu_count: 1
567
+ context_window: 1024
568
+ defaults:
569
+ max_model_len: 1024
570
+ gpu_memory_utilization: 0.3
571
+ enable_prefix_caching: true
572
+ max_num_batched_tokens: 1024
573
+ max_num_seqs: 16
574
+ smollm2-135m-instruct:
575
+ hf_model_id: HuggingFaceTB/SmolLM2-135M-Instruct
576
+ served_model_name: huggingfacetb/smollm2-135m-instruct
577
+ logical_model_name: huggingfacetb/smollm2-135m-instruct
578
+ tokenizer_name: HuggingFaceTB/SmolLM2-135M-Instruct
579
+ family: smollm2
580
+ modalities:
581
+ - text
582
+ memory_class_gib: 1
583
+ min_vram_gib_per_replica: 1
584
+ preferred_gpu_count: 1
585
+ context_window: 8192
586
+ defaults:
587
+ max_model_len: 4096
588
+ gpu_memory_utilization: 0.6
589
+ enable_prefix_caching: true
590
+ max_num_batched_tokens: 2048
591
+ max_num_seqs: 16
592
+ pythia-160m:
593
+ hf_model_id: EleutherAI/pythia-160m
594
+ served_model_name: eleutherai/pythia-160m
595
+ logical_model_name: eleutherai/pythia-160m
596
+ tokenizer_name: EleutherAI/pythia-160m
597
+ family: pythia
598
+ modalities:
599
+ - text
600
+ supported_protocols:
601
+ - completions
602
+ memory_class_gib: 1
603
+ min_vram_gib_per_replica: 1
604
+ preferred_gpu_count: 1
605
+ context_window: 2048
606
+ defaults:
607
+ max_model_len: 2048
608
+ gpu_memory_utilization: 0.6
609
+ enable_prefix_caching: false
610
+ max_num_batched_tokens: 2048
611
+ max_num_seqs: 16
612
+ pythia-410m:
613
+ hf_model_id: EleutherAI/pythia-410m
614
+ served_model_name: eleutherai/pythia-410m
615
+ logical_model_name: eleutherai/pythia-410m
616
+ tokenizer_name: EleutherAI/pythia-410m
617
+ family: pythia
618
+ modalities:
619
+ - text
620
+ supported_protocols:
621
+ - completions
622
+ memory_class_gib: 2
623
+ min_vram_gib_per_replica: 2
624
+ preferred_gpu_count: 1
625
+ context_window: 2048
626
+ defaults:
627
+ max_model_len: 2048
628
+ gpu_memory_utilization: 0.7
629
+ enable_prefix_caching: false
630
+ max_num_batched_tokens: 2048
631
+ max_num_seqs: 16
632
+ pythia-6.9b:
633
+ hf_model_id: EleutherAI/pythia-6.9b
634
+ served_model_name: eleutherai/pythia-6.9b
635
+ logical_model_name: eleutherai/pythia-6.9b
636
+ tokenizer_name: eleutherai/pythia-6.9b
637
+ family: pythia
638
+ modalities:
639
+ - text
640
+ supported_protocols:
641
+ - completions
642
+ memory_class_gib: 16
643
+ min_vram_gib_per_replica: 16
644
+ preferred_gpu_count: 1
645
+ context_window: 2048
646
+ defaults:
647
+ max_model_len: 2048
648
+ gpu_memory_utilization: 0.9
649
+ enable_prefix_caching: false
650
+ max_num_batched_tokens: 2048
651
+ max_num_seqs: 16
652
+ pythia-2.8b-v0:
653
+ hf_model_id: EleutherAI/pythia-2.8b-v0
654
+ served_model_name: eleutherai/pythia-2.8b-v0
655
+ logical_model_name: eleutherai/pythia-2.8b-v0
656
+ tokenizer_name: eleutherai/pythia-2.8b-v0
657
+ family: pythia
658
+ modalities:
659
+ - text
660
+ supported_protocols:
661
+ - completions
662
+ memory_class_gib: 8
663
+ min_vram_gib_per_replica: 8
664
+ preferred_gpu_count: 1
665
+ context_window: 2048
666
+ defaults:
667
+ max_model_len: 2048
668
+ gpu_memory_utilization: 0.9
669
+ enable_prefix_caching: false
670
+ max_num_batched_tokens: 2048
671
+ max_num_seqs: 16
672
+ pythia-1b-v0:
673
+ hf_model_id: EleutherAI/pythia-1b-v0
674
+ served_model_name: eleutherai/pythia-1b-v0
675
+ logical_model_name: eleutherai/pythia-1b-v0
676
+ tokenizer_name: eleutherai/pythia-1b-v0
677
+ family: pythia
678
+ modalities:
679
+ - text
680
+ supported_protocols:
681
+ - completions
682
+ memory_class_gib: 4
683
+ min_vram_gib_per_replica: 4
684
+ preferred_gpu_count: 1
685
+ context_window: 2048
686
+ defaults:
687
+ max_model_len: 2048
688
+ gpu_memory_utilization: 0.9
689
+ enable_prefix_caching: false
690
+ max_num_batched_tokens: 2048
691
+ max_num_seqs: 16
692
+ vicuna-7b-v1.3:
693
+ hf_model_id: lmsys/vicuna-7b-v1.3
694
+ served_model_name: lmsys/vicuna-7b-v1.3
695
+ logical_model_name: lmsys/vicuna-7b-v1.3
696
+ tokenizer_name: lmsys/vicuna-7b-v1.3
697
+ family: vicuna
698
+ supported_protocols:
699
+ - completions
700
+ notes:
701
+ - Use a no-chat-template style profile when reproducing legacy completion-style
702
+ prompting.
703
+ modalities:
704
+ - text
705
+ memory_class_gib: 16
706
+ min_vram_gib_per_replica: 16
707
+ preferred_gpu_count: 1
708
+ context_window: 2048
709
+ defaults:
710
+ max_model_len: 2048
711
+ gpu_memory_utilization: 0.9
712
+ enable_prefix_caching: false
713
+ max_num_batched_tokens: 2048
714
+ max_num_seqs: 16